akm-cli 0.9.0-rc.0 → 0.9.0-rc.13

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (598) hide show
  1. package/CHANGELOG.md +1283 -22
  2. package/README.md +62 -37
  3. package/SECURITY.md +46 -31
  4. package/dist/akm +162 -38
  5. package/dist/akm-migrate +44 -0
  6. package/dist/assets/backends/schtasks-template.xml +2 -1
  7. package/dist/assets/hints/cli-hints-full.md +268 -118
  8. package/dist/assets/hints/cli-hints-short.md +87 -24
  9. package/dist/assets/{profiles → improve-strategies}/catchup.json +3 -1
  10. package/dist/assets/{profiles → improve-strategies}/consolidate.json +3 -1
  11. package/dist/assets/{profiles → improve-strategies}/default.json +6 -7
  12. package/dist/assets/improve-strategies/frequent.json +15 -0
  13. package/dist/assets/{profiles → improve-strategies}/graph-refresh.json +4 -2
  14. package/dist/assets/{profiles → improve-strategies}/memory-focus.json +4 -1
  15. package/dist/assets/{profiles → improve-strategies}/proactive-maintenance.json +5 -5
  16. package/dist/assets/{profiles → improve-strategies}/quick.json +4 -2
  17. package/dist/assets/improve-strategies/reflect-distill.json +30 -0
  18. package/dist/assets/{profiles → improve-strategies}/thorough.json +1 -1
  19. package/dist/assets/prompts/consolidate-system.md +5 -5
  20. package/dist/assets/prompts/extract-session.md +2 -6
  21. package/dist/assets/prompts/memory-infer-user.md +2 -3
  22. package/dist/assets/prompts/reflect-llm-framed-contract.md +11 -0
  23. package/dist/assets/prompts/reflect-llm-schema-contract.md +3 -0
  24. package/dist/assets/prompts/reflect-output-repair.md +3 -0
  25. package/dist/assets/prompts/workflow-unit-preamble.md +26 -0
  26. package/dist/assets/stash-skeleton/README.md +38 -10
  27. package/dist/assets/stash-skeleton/facts/conventions/assets/agent.md +8 -0
  28. package/dist/assets/stash-skeleton/facts/conventions/assets/command.md +8 -0
  29. package/dist/assets/stash-skeleton/facts/conventions/assets/fact.md +14 -1
  30. package/dist/assets/stash-skeleton/facts/conventions/assets/knowledge.md +13 -1
  31. package/dist/assets/stash-skeleton/facts/conventions/assets/lesson.md +9 -1
  32. package/dist/assets/stash-skeleton/facts/conventions/assets/memory.md +11 -0
  33. package/dist/assets/stash-skeleton/facts/conventions/assets/script.md +9 -0
  34. package/dist/assets/stash-skeleton/facts/conventions/assets/skill.md +9 -0
  35. package/dist/assets/stash-skeleton/facts/conventions/assets/workflow.md +8 -0
  36. package/dist/assets/stash-skeleton/facts/conventions/backlinks.md +100 -0
  37. package/dist/assets/stash-skeleton/facts/conventions/domains.md +64 -0
  38. package/dist/assets/stash-skeleton/facts/conventions/organization.md +136 -0
  39. package/dist/assets/tasks/core/extract.yml +3 -2
  40. package/dist/assets/tasks/core/improve.yml +2 -1
  41. package/dist/assets/tasks/core/index-refresh.yml +1 -0
  42. package/dist/assets/tasks/core/sync.yml +1 -0
  43. package/dist/assets/tasks/core/version-check.yml +2 -1
  44. package/dist/assets/tasks/improve/akm-graph-refresh-weekly.yml +5 -0
  45. package/dist/assets/tasks/improve/akm-improve-catchup.yml +8 -0
  46. package/dist/assets/tasks/improve/akm-improve-consolidate.yml +5 -0
  47. package/dist/assets/tasks/improve/akm-improve-frequent.yml +5 -0
  48. package/dist/assets/tasks/improve/akm-improve-nightly.yml +5 -0
  49. package/dist/assets/templates/html/health.html +5 -4
  50. package/dist/assets/workflows/workflow-template.md +31 -15
  51. package/dist/cli/invocation.js +279 -0
  52. package/dist/cli/parse-args.js +5 -90
  53. package/dist/cli/retired-commands.js +78 -0
  54. package/dist/cli/shared.js +158 -48
  55. package/dist/cli-node.mjs +2 -1
  56. package/dist/cli.js +747 -293
  57. package/dist/commands/agent/agent-dispatch.js +19 -18
  58. package/dist/commands/agent/agent-support.js +0 -24
  59. package/dist/commands/agent/contribute-cli.js +43 -97
  60. package/dist/commands/completions.js +80 -23
  61. package/dist/commands/config-cli.js +44 -281
  62. package/dist/commands/env/env-binding.js +99 -0
  63. package/dist/commands/env/env-cli.js +84 -224
  64. package/dist/commands/env/env.js +12 -163
  65. package/dist/commands/env/marker-path.js +6 -0
  66. package/dist/commands/env/secret-cli.js +45 -61
  67. package/dist/commands/env/secret.js +32 -62
  68. package/dist/commands/feedback-cli.js +179 -85
  69. package/dist/commands/health/accept-rate.js +58 -0
  70. package/dist/commands/health/advisories.js +7 -8
  71. package/dist/commands/health/checks.js +279 -94
  72. package/dist/commands/health/html-report.js +197 -578
  73. package/dist/commands/health/improve-metrics.js +277 -246
  74. package/dist/commands/health/llm-usage.js +19 -19
  75. package/dist/commands/health/md-report.js +16 -7
  76. package/dist/commands/health/metrics.js +67 -32
  77. package/dist/commands/health/renderers.js +47 -0
  78. package/dist/commands/health/report-view-model.js +508 -0
  79. package/dist/commands/health/stash-exposure.js +1 -1
  80. package/dist/commands/health/surfaces.js +16 -56
  81. package/dist/commands/health/task-runs.js +3 -67
  82. package/dist/{migrate-storage-node.mjs → commands/health/types-checks.js} +1 -5
  83. package/dist/commands/health/types-improve.js +29 -0
  84. package/dist/{output/text/save.js → commands/health/types-metrics.js} +1 -2
  85. package/dist/commands/health/types-result.js +7 -0
  86. package/dist/commands/health/types-runs.js +4 -0
  87. package/dist/commands/health/types-session-log.js +4 -0
  88. package/dist/commands/health/types-windows.js +4 -0
  89. package/dist/commands/health/types.js +26 -21
  90. package/dist/commands/health/windows.js +2 -3
  91. package/dist/commands/health.js +296 -167
  92. package/dist/commands/improve/anti-collapse.js +5 -5
  93. package/dist/commands/improve/autonomy-gate.js +68 -0
  94. package/dist/commands/improve/collapse-detector.js +65 -52
  95. package/dist/commands/improve/consolidate/chunking.js +9 -7
  96. package/dist/commands/improve/consolidate/eligibility.js +1 -23
  97. package/dist/commands/improve/consolidate/merge.js +4 -0
  98. package/dist/commands/improve/consolidate.js +454 -1354
  99. package/dist/commands/improve/content-hash.js +39 -0
  100. package/dist/commands/improve/distill/content-repair.js +4 -10
  101. package/dist/commands/improve/distill/promote-memory.js +89 -64
  102. package/dist/commands/improve/distill/quality-gate.js +118 -42
  103. package/dist/commands/improve/distill-guards.js +1 -1
  104. package/dist/commands/improve/distill-promotion-policy.js +33 -888
  105. package/dist/commands/improve/distill.js +607 -363
  106. package/dist/commands/improve/eligibility.js +165 -79
  107. package/dist/commands/improve/extract-cli.js +35 -126
  108. package/dist/commands/improve/extract-prompt.js +6 -35
  109. package/dist/commands/improve/extract.js +640 -391
  110. package/dist/commands/improve/feedback-valence.js +2 -12
  111. package/dist/commands/improve/improve-cli.js +134 -135
  112. package/dist/commands/improve/improve-result-file.js +30 -50
  113. package/dist/commands/improve/improve-run-types.js +4 -0
  114. package/dist/commands/improve/improve-strategies.js +135 -0
  115. package/dist/commands/improve/improve.js +904 -701
  116. package/dist/commands/improve/locks.js +64 -111
  117. package/dist/commands/improve/loop-stages.js +1110 -923
  118. package/dist/commands/improve/memory/derived-ref.js +124 -0
  119. package/dist/commands/improve/memory/memory-belief.js +79 -7
  120. package/dist/commands/improve/memory/memory-contradiction-detect.js +49 -52
  121. package/dist/commands/improve/memory/memory-improve.js +25 -37
  122. package/dist/commands/improve/outcome-loop.js +25 -88
  123. package/dist/commands/improve/preparation.js +1034 -813
  124. package/dist/commands/improve/proactive-maintenance.js +34 -9
  125. package/dist/commands/improve/proposal-envelope.js +31 -0
  126. package/dist/commands/improve/reflect.js +983 -794
  127. package/dist/commands/improve/run-context.js +119 -0
  128. package/dist/commands/improve/salience.js +24 -127
  129. package/dist/commands/improve/session-asset.js +7 -3
  130. package/dist/commands/improve/shared.js +14 -34
  131. package/dist/commands/improve/source-identity.js +28 -0
  132. package/dist/commands/improve/triage.js +20 -17
  133. package/dist/commands/lint/base-linter.js +340 -313
  134. package/dist/commands/lint/env-key-rules.js +31 -47
  135. package/dist/commands/lint/index.js +185 -30
  136. package/dist/commands/{events.js → log.js} +28 -38
  137. package/dist/commands/migrate-cli.js +54 -0
  138. package/dist/commands/migration-tool.js +55 -0
  139. package/dist/commands/observability-cli.js +70 -208
  140. package/dist/commands/proposal/diff-format.js +50 -0
  141. package/dist/commands/proposal/drain-policies.js +0 -6
  142. package/dist/commands/proposal/drain.js +91 -40
  143. package/dist/commands/proposal/proposal-cli.js +134 -132
  144. package/dist/commands/proposal/proposal-types.js +56 -0
  145. package/dist/commands/proposal/proposal.js +83 -65
  146. package/dist/commands/proposal/propose-cli.js +88 -0
  147. package/dist/commands/proposal/propose.js +105 -88
  148. package/dist/commands/proposal/repository.js +1303 -278
  149. package/dist/commands/proposal/validators/proposal-quality-validators.js +16 -6
  150. package/dist/commands/proposal/validators/proposal-validators.js +61 -12
  151. package/dist/commands/proposal/validators/proposals.js +6 -8
  152. package/dist/commands/read/curate.js +78 -73
  153. package/dist/commands/read/knowledge.js +510 -13
  154. package/dist/commands/read/registry-search.js +2 -2
  155. package/dist/commands/read/remember-cli.js +84 -15
  156. package/dist/commands/read/search-cli.js +203 -96
  157. package/dist/commands/read/search.js +126 -94
  158. package/dist/commands/read/show.js +226 -250
  159. package/dist/commands/registry-cli.js +34 -60
  160. package/dist/commands/remember.js +18 -57
  161. package/dist/commands/sources/add-cli.js +104 -49
  162. package/dist/commands/sources/bundle-cli.js +166 -0
  163. package/dist/commands/sources/bundle-config-ops.js +63 -0
  164. package/dist/commands/sources/info.js +27 -15
  165. package/dist/commands/sources/init.js +30 -40
  166. package/dist/commands/sources/installed-stashes.js +469 -172
  167. package/dist/commands/sources/migration-help.js +7 -4
  168. package/dist/commands/sources/schema-repair.js +10 -9
  169. package/dist/commands/sources/self-update.js +182 -121
  170. package/dist/commands/sources/source-add.js +169 -178
  171. package/dist/commands/sources/source-clone.js +144 -41
  172. package/dist/commands/sources/source-manage.js +94 -59
  173. package/dist/commands/sources/sources-cli.js +64 -205
  174. package/dist/commands/sources/stash-cli.js +91 -54
  175. package/dist/commands/sources/stash-skeleton.js +1 -1
  176. package/dist/commands/tasks/tasks-cli.js +106 -104
  177. package/dist/commands/tasks/tasks.js +445 -262
  178. package/dist/commands/workflow-cli.js +232 -121
  179. package/dist/core/action-contributors.js +1 -1
  180. package/dist/core/activation-policy.js +49 -0
  181. package/dist/core/adapter/adapters/agent-skills-adapter.js +181 -0
  182. package/dist/core/adapter/adapters/akm-adapter.js +528 -0
  183. package/dist/core/adapter/adapters/akm-lint.js +392 -0
  184. package/dist/core/adapter/adapters/akm-metadata.js +387 -0
  185. package/dist/core/adapter/adapters/akm-task-adapter.js +149 -0
  186. package/dist/core/adapter/adapters/akm-workflow-adapter.js +180 -0
  187. package/dist/core/adapter/adapters/claude-adapter.js +61 -0
  188. package/dist/core/adapter/adapters/dotenv-adapter.js +187 -0
  189. package/dist/core/adapter/adapters/generic-files-adapter.js +119 -0
  190. package/dist/core/adapter/adapters/index.js +80 -0
  191. package/dist/core/adapter/adapters/llm-wiki-adapter.js +419 -0
  192. package/dist/core/adapter/adapters/okf-adapter.js +391 -0
  193. package/dist/core/adapter/adapters/opencode-adapter.js +68 -0
  194. package/dist/core/adapter/adapters/shared.js +286 -0
  195. package/dist/core/adapter/adapters/tool-dir-shared.js +217 -0
  196. package/dist/core/adapter/adapters/website-snapshot-adapter.js +155 -0
  197. package/dist/core/adapter/bundle-adapter.js +4 -0
  198. package/dist/core/adapter/detect-adapter.js +17 -0
  199. package/dist/core/adapter/recognize-match.js +44 -0
  200. package/dist/core/adapter/registry.js +56 -0
  201. package/dist/core/adapter/types.js +4 -0
  202. package/dist/core/asset/akm-markdown.js +30 -0
  203. package/dist/core/asset/asset-placement.js +243 -0
  204. package/dist/core/asset/asset-ref.js +110 -79
  205. package/dist/core/asset/asset-serialize.js +20 -0
  206. package/dist/core/asset/frontmatter.js +28 -12
  207. package/dist/core/asset/markdown.js +40 -51
  208. package/dist/core/asset/resolve-ref.js +274 -0
  209. package/dist/core/asset/stash-meta.js +2 -2
  210. package/dist/core/bundle-id.js +51 -0
  211. package/dist/core/common.js +281 -86
  212. package/dist/core/config/config-io.js +42 -128
  213. package/dist/core/config/config-schema.js +233 -834
  214. package/dist/core/config/config-sources.js +162 -39
  215. package/dist/core/config/config-types.js +16 -11
  216. package/dist/core/config/config-version.js +29 -0
  217. package/dist/core/config/config-walker.js +126 -37
  218. package/dist/core/config/config.js +154 -331
  219. package/dist/core/config/deep-merge.js +41 -0
  220. package/dist/core/config/engine-semantics.js +28 -0
  221. package/dist/core/config/experimental.js +21 -0
  222. package/dist/core/config/schema/embedding.js +38 -0
  223. package/dist/core/config/schema/engines.js +116 -0
  224. package/dist/core/config/schema/experimental.js +47 -0
  225. package/dist/core/config/schema/feedback.js +31 -0
  226. package/dist/core/config/schema/improve-processes.js +389 -0
  227. package/dist/core/config/schema/improve.js +94 -0
  228. package/dist/core/config/schema/index-config.js +176 -0
  229. package/dist/core/config/schema/output.js +18 -0
  230. package/dist/core/config/schema/primitives.js +94 -0
  231. package/dist/core/config/schema/search.js +30 -0
  232. package/dist/core/config/schema/setup.js +18 -0
  233. package/dist/core/config/schema/sources-bundles.js +169 -0
  234. package/dist/core/config/schema/workflow.js +29 -0
  235. package/dist/core/env-secret-ref.js +155 -20
  236. package/dist/core/errors.js +17 -15
  237. package/dist/core/events-types.js +4 -0
  238. package/dist/core/events.js +46 -128
  239. package/dist/core/extra-params.js +62 -0
  240. package/dist/core/file-change.js +17 -0
  241. package/dist/core/file-lock.js +202 -57
  242. package/dist/core/fs-txn.js +392 -0
  243. package/dist/core/git-message.js +59 -0
  244. package/dist/core/improve-result.js +167 -0
  245. package/dist/core/json-schema.js +142 -0
  246. package/dist/core/lesson-lint.js +1 -17
  247. package/dist/core/logs-db.js +1 -1
  248. package/dist/core/maintenance-barrier.js +135 -0
  249. package/dist/core/migration-operation.js +44 -0
  250. package/dist/core/mutation-target.js +78 -0
  251. package/dist/core/paths.js +22 -25
  252. package/dist/core/platform.js +10 -0
  253. package/dist/core/recognition-util.js +128 -0
  254. package/dist/core/redaction.js +392 -0
  255. package/dist/core/standards/resolve-standards-context.js +36 -65
  256. package/dist/core/standards/resolve-stash-standards.js +2 -2
  257. package/dist/core/standards/resolve-type-conventions.js +5 -5
  258. package/dist/core/state/migrations.js +242 -11
  259. package/dist/core/state-db.js +98 -10
  260. package/dist/core/structured.js +1 -1
  261. package/dist/core/subprocess.js +303 -0
  262. package/dist/core/text-truncation.js +9 -5
  263. package/dist/core/time.js +20 -0
  264. package/dist/core/type-presentation.js +130 -0
  265. package/dist/core/warn.js +0 -3
  266. package/dist/core/write-source.js +834 -118
  267. package/dist/indexer/bundle-identity-guard.js +92 -0
  268. package/dist/indexer/db/graph-db.js +1 -25
  269. package/dist/indexer/db/llm-cache.js +1 -1
  270. package/dist/indexer/ensure-index.js +30 -9
  271. package/dist/indexer/graph/graph-boost.js +9 -30
  272. package/dist/indexer/graph/graph-extraction.js +41 -27
  273. package/dist/indexer/graph/graph-types.js +4 -0
  274. package/dist/indexer/index-writer-lock.js +93 -49
  275. package/dist/indexer/index-written-assets.js +100 -53
  276. package/dist/indexer/indexer.js +746 -329
  277. package/dist/indexer/init.js +18 -25
  278. package/dist/indexer/installations.js +142 -0
  279. package/dist/indexer/passes/dir-staleness.js +18 -10
  280. package/dist/indexer/passes/memory-inference.js +25 -15
  281. package/dist/indexer/passes/metadata.js +412 -243
  282. package/dist/indexer/scan/doc-to-entry.js +160 -0
  283. package/dist/indexer/scan/drain-dir.js +134 -0
  284. package/dist/indexer/search/db-search.js +292 -108
  285. package/dist/indexer/search/fts-query.js +64 -0
  286. package/dist/indexer/search/ranking-contributors.js +145 -25
  287. package/dist/indexer/search/ranking-types.js +4 -0
  288. package/dist/indexer/search/ranking.js +28 -71
  289. package/dist/indexer/search/search-attribution.js +67 -0
  290. package/dist/indexer/search/search-fields.js +18 -3
  291. package/dist/indexer/search/search-hit-enrichers.js +30 -40
  292. package/dist/indexer/search/search-source.js +157 -111
  293. package/dist/indexer/search/semantic-status.js +4 -1
  294. package/dist/indexer/usage/usage-events.js +10 -30
  295. package/dist/indexer/walk/file-context.js +3 -45
  296. package/dist/indexer/walk/matchers.js +42 -34
  297. package/dist/indexer/walk/path-resolver.js +11 -5
  298. package/dist/indexer/walk/walker.js +42 -14
  299. package/dist/integrations/agent/builder-shared.js +7 -0
  300. package/dist/integrations/agent/builders.js +5 -56
  301. package/dist/integrations/agent/config.js +3 -143
  302. package/dist/integrations/agent/detect.js +17 -2
  303. package/dist/integrations/agent/engine-resolution.js +231 -0
  304. package/dist/integrations/agent/index.js +1 -2
  305. package/dist/integrations/agent/model-aliases.js +16 -2
  306. package/dist/integrations/agent/profiles.js +36 -62
  307. package/dist/integrations/agent/prompts.js +46 -18
  308. package/dist/integrations/agent/runner-dispatch.js +93 -4
  309. package/dist/integrations/agent/runner.js +76 -208
  310. package/dist/integrations/agent/spawn.js +88 -196
  311. package/dist/integrations/harnesses/aider/agent-builder.js +114 -0
  312. package/dist/integrations/harnesses/aider/index.js +48 -0
  313. package/dist/integrations/harnesses/aider/result-extractor.js +53 -0
  314. package/dist/integrations/harnesses/amazonq/agent-builder.js +147 -0
  315. package/dist/integrations/harnesses/amazonq/index.js +45 -0
  316. package/dist/integrations/harnesses/amazonq/result-extractor.js +48 -0
  317. package/dist/integrations/harnesses/claude/agent-builder.js +46 -8
  318. package/dist/integrations/harnesses/claude/config-import.js +1 -3
  319. package/dist/integrations/harnesses/claude/index.js +24 -35
  320. package/dist/integrations/harnesses/claude/result-extractor.js +52 -0
  321. package/dist/integrations/harnesses/claude/session-log.js +27 -75
  322. package/dist/integrations/harnesses/codex/agent-builder.js +138 -0
  323. package/dist/integrations/harnesses/codex/index.js +52 -0
  324. package/dist/integrations/harnesses/codex/result-extractor.js +73 -0
  325. package/dist/integrations/harnesses/copilot/agent-builder.js +122 -0
  326. package/dist/integrations/harnesses/copilot/index.js +48 -0
  327. package/dist/integrations/harnesses/copilot/result-extractor.js +151 -0
  328. package/dist/integrations/harnesses/gemini/agent-builder.js +120 -0
  329. package/dist/integrations/harnesses/gemini/index.js +48 -0
  330. package/dist/integrations/harnesses/gemini/result-extractor.js +121 -0
  331. package/dist/integrations/harnesses/ids.js +24 -0
  332. package/dist/integrations/harnesses/index.js +54 -34
  333. package/dist/integrations/harnesses/opencode/agent-builder.js +23 -5
  334. package/dist/integrations/harnesses/opencode/config-import.js +1 -3
  335. package/dist/integrations/harnesses/opencode/index.js +14 -32
  336. package/dist/integrations/harnesses/opencode/session-log.js +67 -125
  337. package/dist/integrations/harnesses/opencode-sdk/harness.js +51 -0
  338. package/dist/integrations/harnesses/opencode-sdk/sdk-runner.js +681 -108
  339. package/dist/integrations/harnesses/openhands/agent-builder.js +128 -0
  340. package/dist/integrations/harnesses/openhands/index.js +48 -0
  341. package/dist/integrations/harnesses/openhands/result-extractor.js +103 -0
  342. package/dist/integrations/harnesses/pi/agent-builder.js +97 -0
  343. package/dist/integrations/harnesses/pi/index.js +45 -0
  344. package/dist/integrations/harnesses/pi/result-extractor.js +135 -0
  345. package/dist/integrations/harnesses/shared.js +17 -0
  346. package/dist/integrations/harnesses/types.js +43 -32
  347. package/dist/integrations/lockfile.js +211 -24
  348. package/dist/integrations/session-logs/index.js +36 -39
  349. package/dist/integrations/session-logs/provider-base.js +113 -0
  350. package/dist/llm/client.js +182 -110
  351. package/dist/llm/embedders/deterministic.js +2 -2
  352. package/dist/llm/embedders/remote.js +21 -9
  353. package/dist/llm/feature-gate.js +17 -57
  354. package/dist/llm/graph-extract.js +12 -13
  355. package/dist/llm/index-passes.js +8 -42
  356. package/dist/llm/memory-infer.js +144 -1
  357. package/dist/llm/metadata-enhance.js +45 -30
  358. package/dist/llm/structured-call.js +16 -8
  359. package/dist/llm/usage-persist.js +30 -5
  360. package/dist/llm/usage-telemetry.js +59 -6
  361. package/dist/output/cli-hints.js +1 -2
  362. package/dist/output/command-registry.js +27 -0
  363. package/dist/output/context.js +22 -7
  364. package/dist/output/format-exempt.js +80 -0
  365. package/dist/output/generic-render.js +251 -0
  366. package/dist/output/html-render.js +11 -16
  367. package/dist/output/render-registry.js +57 -0
  368. package/dist/output/renderers.js +14 -279
  369. package/dist/output/shapes/curate.js +10 -1
  370. package/dist/output/shapes/events.js +12 -7
  371. package/dist/output/shapes/helpers.js +58 -84
  372. package/dist/output/shapes/passthrough.js +11 -39
  373. package/dist/output/shapes/proposal/producer.js +15 -7
  374. package/dist/output/shapes/registry.js +12 -6
  375. package/dist/output/shapes.js +0 -9
  376. package/dist/output/text/{init.js → bundle-create.js} +3 -1
  377. package/dist/output/text/bundle-show.js +7 -0
  378. package/dist/output/text/command-format.js +562 -0
  379. package/dist/output/text/env.js +1 -3
  380. package/dist/output/text/events.js +8 -7
  381. package/dist/output/text/helpers.js +15 -1164
  382. package/dist/output/text/proposal/producer.js +4 -2
  383. package/dist/output/text/proposal-format.js +202 -0
  384. package/dist/output/text/registry-commands.js +1 -2
  385. package/dist/output/text/registry.js +12 -6
  386. package/dist/output/text/show-directives.js +117 -0
  387. package/dist/output/text/show-format.js +103 -0
  388. package/dist/output/text/sync.js +5 -0
  389. package/dist/output/text/workflow-format.js +332 -0
  390. package/dist/output/text/workflow.js +3 -2
  391. package/dist/output/text.js +10 -19
  392. package/dist/registry/factory.js +4 -6
  393. package/dist/registry/origin-resolve.js +16 -27
  394. package/dist/registry/providers/skills-sh.js +3 -3
  395. package/dist/registry/providers/static-index.js +15 -25
  396. package/dist/registry/resolve.js +43 -94
  397. package/dist/registry/semver.js +43 -0
  398. package/dist/runtime.js +81 -12
  399. package/dist/scripts/akm-migrate.js +35529 -0
  400. package/dist/setup/detect.js +5 -7
  401. package/dist/setup/detected-engines.js +136 -0
  402. package/dist/setup/engine-config.js +100 -0
  403. package/dist/setup/registry-stash-loader.js +3 -3
  404. package/dist/setup/semantic-assets.js +12 -9
  405. package/dist/setup/setup.js +444 -208
  406. package/dist/setup/steps/connection-shared.js +120 -0
  407. package/dist/setup/steps/connection.js +108 -305
  408. package/dist/setup/steps/platforms.js +13 -12
  409. package/dist/setup/steps/semantic.js +15 -3
  410. package/dist/setup/steps/sources.js +21 -15
  411. package/dist/setup/steps/stashdir.js +6 -4
  412. package/dist/setup/steps/tasks.js +236 -119
  413. package/dist/setup/steps.js +3 -2
  414. package/dist/sources/freshness.js +39 -0
  415. package/dist/sources/provider-factory.js +11 -17
  416. package/dist/sources/providers/filesystem.js +2 -3
  417. package/dist/sources/providers/git-install.js +278 -34
  418. package/dist/sources/providers/git-provider.js +54 -56
  419. package/dist/sources/providers/git-stash.js +420 -91
  420. package/dist/sources/providers/git.js +2 -2
  421. package/dist/sources/providers/npm.js +16 -19
  422. package/dist/sources/providers/provider-utils.js +47 -22
  423. package/dist/sources/providers/sync-from-ref.js +3 -9
  424. package/dist/sources/providers/website.js +2 -2
  425. package/dist/sources/resolve.js +11 -10
  426. package/dist/sources/snapshot-fetchers/types.js +4 -0
  427. package/dist/sources/{website-ingest.js → snapshot-fetchers/website-ingest.js} +110 -41
  428. package/dist/storage/database.js +60 -4
  429. package/dist/storage/engines/sqlite-migrations.js +156 -5
  430. package/dist/storage/locations.js +1 -2
  431. package/dist/storage/repositories/canaries-repository.js +1 -1
  432. package/dist/storage/repositories/events-repository.js +51 -11
  433. package/dist/storage/repositories/improve-runs-repository.js +6 -32
  434. package/dist/storage/repositories/index-connection.js +79 -0
  435. package/dist/storage/repositories/index-db.js +4 -3
  436. package/dist/storage/repositories/index-entries-repository.js +863 -0
  437. package/dist/{indexer/db/entry-mapper.js → storage/repositories/index-entry-mapper.js} +19 -2
  438. package/dist/storage/repositories/index-entry-types.js +4 -0
  439. package/dist/storage/repositories/index-fts-repository.js +167 -0
  440. package/dist/storage/repositories/index-llm-cache-repository.js +108 -0
  441. package/dist/storage/repositories/index-meta-repository.js +49 -0
  442. package/dist/{indexer/db/schema.js → storage/repositories/index-schema.js} +226 -100
  443. package/dist/storage/repositories/index-sql.js +12 -0
  444. package/dist/storage/repositories/index-utility-repository.js +356 -0
  445. package/dist/storage/repositories/index-vec-repository.js +250 -0
  446. package/dist/storage/repositories/outcome-repository.js +119 -0
  447. package/dist/storage/repositories/proposals-repository.js +317 -75
  448. package/dist/storage/repositories/registry-cache.js +1 -1
  449. package/dist/storage/repositories/salience-repository.js +172 -0
  450. package/dist/storage/repositories/task-history-repository.js +110 -3
  451. package/dist/storage/repositories/workflow-runs-repository.js +240 -19
  452. package/dist/tasks/backends/cron.js +169 -46
  453. package/dist/tasks/backends/exec-utils.js +76 -3
  454. package/dist/tasks/backends/index.js +6 -9
  455. package/dist/tasks/backends/launchd.js +292 -55
  456. package/dist/tasks/backends/schtasks.js +557 -70
  457. package/dist/tasks/backends/types.js +4 -0
  458. package/dist/tasks/command-executable.js +93 -0
  459. package/dist/tasks/embedded.js +56 -38
  460. package/dist/tasks/parser.js +156 -64
  461. package/dist/tasks/resolve-akm-bin.js +144 -51
  462. package/dist/tasks/runner.js +377 -209
  463. package/dist/tasks/schedule.js +108 -19
  464. package/dist/tasks/scheduler-invocation.js +296 -0
  465. package/dist/tasks/schema.js +1 -1
  466. package/dist/tasks/task-id.js +35 -0
  467. package/dist/tasks/validator.js +30 -16
  468. package/dist/text-import-hook.mjs +1 -1
  469. package/dist/workflows/authoring/authoring.js +104 -43
  470. package/dist/workflows/authoring/scope-key.js +1 -1
  471. package/dist/workflows/cli.js +0 -16
  472. package/dist/workflows/concurrency-policy.js +15 -0
  473. package/dist/workflows/exec/brief.js +450 -0
  474. package/dist/workflows/exec/frozen-judge.js +47 -0
  475. package/dist/workflows/exec/native-executor.js +1038 -0
  476. package/dist/workflows/exec/param-secrets.js +115 -0
  477. package/dist/workflows/exec/report.js +1460 -0
  478. package/dist/workflows/exec/run-workflow.js +602 -0
  479. package/dist/workflows/exec/scheduler.js +71 -0
  480. package/dist/workflows/exec/step-work.js +1190 -0
  481. package/dist/workflows/exec/unit-writer.js +23 -0
  482. package/dist/workflows/exec/workflow-engine-gate.js +67 -0
  483. package/dist/workflows/exec/worktree.js +171 -0
  484. package/dist/workflows/ir/compile.js +246 -0
  485. package/dist/workflows/ir/freeze.js +233 -0
  486. package/dist/workflows/ir/params.js +54 -0
  487. package/dist/workflows/ir/plan-hash.js +68 -0
  488. package/dist/workflows/ir/schema.js +540 -0
  489. package/dist/workflows/parser.js +878 -304
  490. package/dist/workflows/program/expressions.js +181 -0
  491. package/dist/workflows/program/schema.js +51 -0
  492. package/dist/workflows/renderer.js +100 -45
  493. package/dist/workflows/resource-limits.js +22 -0
  494. package/dist/workflows/runtime/agent-identity.js +59 -14
  495. package/dist/workflows/runtime/checkin.js +1 -1
  496. package/dist/workflows/runtime/plan-classifier.js +131 -0
  497. package/dist/workflows/runtime/runs.js +376 -119
  498. package/dist/workflows/runtime/unit-checkin.js +45 -0
  499. package/dist/workflows/runtime/unit-phases.js +20 -0
  500. package/dist/workflows/runtime/workflow-asset-loader.js +241 -40
  501. package/dist/workflows/schema.js +1 -11
  502. package/dist/workflows/validate-summary.js +2 -3
  503. package/dist/workflows/validator.js +52 -30
  504. package/docs/README.md +42 -78
  505. package/docs/migration/README.md +8 -0
  506. package/docs/migration/release-notes/0.6.0.md +1 -1
  507. package/docs/migration/release-notes/0.7.0.md +9 -8
  508. package/docs/migration/release-notes/0.9.0.md +158 -14
  509. package/docs/migration/v0.7-to-v0.8.md +46 -47
  510. package/docs/migration/v0.8-to-v0.9.md +844 -0
  511. package/docs/reference/README.md +12 -0
  512. package/docs/reference/data-and-telemetry.md +333 -0
  513. package/package.json +21 -17
  514. package/schemas/akm-asset-envelope.json +93 -0
  515. package/schemas/akm-config.json +4636 -0
  516. package/schemas/akm-task.json +87 -0
  517. package/schemas/akm-workflow.json +373 -0
  518. package/dist/akm-migrate-storage +0 -38
  519. package/dist/assets/help/help-accept.md +0 -12
  520. package/dist/assets/help/help-improve.md +0 -84
  521. package/dist/assets/help/help-proposals.md +0 -17
  522. package/dist/assets/help/help-propose.md +0 -17
  523. package/dist/assets/help/help-reject.md +0 -11
  524. package/dist/assets/profiles/frequent.json +0 -13
  525. package/dist/assets/profiles/recombine-only.json +0 -21
  526. package/dist/assets/profiles/reflect-distill.json +0 -30
  527. package/dist/assets/profiles/synthesize.json +0 -15
  528. package/dist/assets/prompts/procedural-system.md +0 -44
  529. package/dist/assets/prompts/recombine-system.md +0 -40
  530. package/dist/assets/prompts/staleness-detect-system.md +0 -6
  531. package/dist/assets/tasks/core/backup.yml +0 -4
  532. package/dist/assets/tasks/graph-refresh-weekly.yml +0 -10
  533. package/dist/assets/templates/html/default.html +0 -78
  534. package/dist/assets/templates/html/vendor/echarts.min.js +0 -45
  535. package/dist/assets/wiki/index-template.md +0 -12
  536. package/dist/assets/wiki/ingest-workflow-template.md +0 -83
  537. package/dist/assets/wiki/log-template.md +0 -8
  538. package/dist/assets/wiki/schema-template.md +0 -61
  539. package/dist/cli/config-migrate.js +0 -150
  540. package/dist/cli/config-validate.js +0 -39
  541. package/dist/commands/graph/graph-cli.js +0 -124
  542. package/dist/commands/graph/graph.js +0 -487
  543. package/dist/commands/improve/calibration.js +0 -161
  544. package/dist/commands/improve/dedup.js +0 -482
  545. package/dist/commands/improve/extract-watch.js +0 -140
  546. package/dist/commands/improve/hot-probation.js +0 -45
  547. package/dist/commands/improve/improve-auto-accept.js +0 -276
  548. package/dist/commands/improve/improve-profiles.js +0 -168
  549. package/dist/commands/improve/procedural.js +0 -398
  550. package/dist/commands/improve/recombine.js +0 -818
  551. package/dist/commands/improve/schema-similarity-gate.js +0 -168
  552. package/dist/commands/lint/agent-linter.js +0 -44
  553. package/dist/commands/lint/command-linter.js +0 -44
  554. package/dist/commands/lint/default-linter.js +0 -16
  555. package/dist/commands/lint/fact-linter.js +0 -39
  556. package/dist/commands/lint/knowledge-linter.js +0 -16
  557. package/dist/commands/lint/memory-linter.js +0 -61
  558. package/dist/commands/lint/registry.js +0 -41
  559. package/dist/commands/lint/skill-linter.js +0 -45
  560. package/dist/commands/lint/task-linter.js +0 -50
  561. package/dist/commands/lint/workflow-linter.js +0 -81
  562. package/dist/commands/proposal/legacy-import.js +0 -115
  563. package/dist/commands/sources/history.js +0 -196
  564. package/dist/commands/tasks/default-tasks.js +0 -186
  565. package/dist/commands/wiki-cli.js +0 -292
  566. package/dist/core/asset/asset-registry.js +0 -76
  567. package/dist/core/asset/asset-spec.js +0 -259
  568. package/dist/core/config/config-migration.js +0 -602
  569. package/dist/core/deep-merge.js +0 -38
  570. package/dist/core/eval/rank-metrics.js +0 -113
  571. package/dist/core/ripgrep/install.js +0 -163
  572. package/dist/core/ripgrep/resolve.js +0 -81
  573. package/dist/indexer/db/db.js +0 -1413
  574. package/dist/indexer/manifest.js +0 -170
  575. package/dist/indexer/passes/metadata-contributors.js +0 -31
  576. package/dist/indexer/usage/unmigrated-vaults-guard.js +0 -94
  577. package/dist/integrations/harnesses/opencode-sdk/index.js +0 -49
  578. package/dist/llm/call-ai.js +0 -62
  579. package/dist/llm/memory-infer-impl.js +0 -138
  580. package/dist/output/shapes/distill.js +0 -14
  581. package/dist/output/shapes/history.js +0 -11
  582. package/dist/output/text/distill.js +0 -6
  583. package/dist/output/text/enable-disable.js +0 -8
  584. package/dist/output/text/history.js +0 -6
  585. package/dist/output/text/wiki.js +0 -16
  586. package/dist/registry/build-index.js +0 -386
  587. package/dist/scripts/migrate-storage.js +0 -19108
  588. package/dist/scripts/migrations/import-fs-improve-runs-to-db.js +0 -9411
  589. package/dist/scripts/migrations/v16-to-v17.js +0 -141
  590. package/dist/setup/legacy-config.js +0 -106
  591. package/dist/storage/repositories/consolidation-repository.js +0 -38
  592. package/dist/storage/repositories/recombine-repository.js +0 -213
  593. package/dist/wiki/wiki-templates.js +0 -15
  594. package/dist/wiki/wiki.js +0 -1012
  595. package/dist/workflows/db.js +0 -215
  596. package/docs/data-and-telemetry.md +0 -226
  597. /package/dist/sources/{wiki-fetchers → snapshot-fetchers}/registry.js +0 -0
  598. /package/dist/sources/{wiki-fetchers → snapshot-fetchers}/youtube.js +0 -0
@@ -3,196 +3,104 @@
3
3
  // file, You can obtain one at https://mozilla.org/MPL/2.0/.
4
4
  import fs from "node:fs";
5
5
  import path from "node:path";
6
- import { parseAssetRef } from "../../core/asset/asset-ref.js";
6
+ import { makeBundleRef } from "../../core/asset/asset-ref.js";
7
7
  import { parseFrontmatter } from "../../core/asset/frontmatter.js";
8
+ import { typeNameFromConceptId } from "../../core/asset/resolve-ref.js";
8
9
  import { daysToMs } from "../../core/common.js";
9
- import { getDefaultLlmConfig, loadConfig } from "../../core/config/config.js";
10
- import { rethrowIfTestIsolationError } from "../../core/errors.js";
10
+ import { loadConfig } from "../../core/config/config.js";
11
+ import { ConfigError, rethrowIfTestIsolationError } from "../../core/errors.js";
11
12
  import { appendEvent, readEvents } from "../../core/events.js";
12
13
  import { openStateDatabase, withStateDb } from "../../core/state-db.js";
13
14
  import { info, warn } from "../../core/warn.js";
14
- import { closeDatabase, getRetrievalCounts, getZeroResultSearches, openExistingDatabase } from "../../indexer/db/db.js";
15
15
  import { countUsageEventsByType } from "../../indexer/usage/usage-events.js";
16
+ import { materializeLlmRunnerConnection } from "../../integrations/agent/runner.js";
16
17
  import { getAvailableHarnesses } from "../../integrations/session-logs/index.js";
17
18
  import { withLlmStage } from "../../llm/usage-telemetry.js";
18
- import { persistPhaseThreshold } from "../../storage/repositories/improve-runs-repository.js";
19
- import { listProposalGateDecisions, listStateProposals } from "../../storage/repositories/proposals-repository.js";
19
+ import { closeDatabase, openExistingDatabase } from "../../storage/repositories/index-connection.js";
20
+ import { getZeroResultSearches } from "../../storage/repositories/index-entries-repository.js";
21
+ import { getRetrievalCounts } from "../../storage/repositories/index-utility-repository.js";
22
+ import { listStateProposals } from "../../storage/repositories/proposals-repository.js";
20
23
  import { akmLint } from "../lint/index.js";
21
- import { getProposal, listProposals } from "../proposal/repository.js";
22
24
  import { runSchemaRepairPass } from "../sources/schema-repair.js";
23
- import { computeThresholdAutoTune, gateDecisionsToSamples, summarizeCalibration, } from "./calibration.js";
25
+ import { isAutonomyLaneAllowed } from "./autonomy-gate.js";
24
26
  import { akmConsolidate } from "./consolidate.js";
25
27
  // Eligibility / candidate-selection predicates live in ./eligibility.
26
28
  import { buildLatestFeedbackTsMap, buildLatestProposalTsMap, buildUtilityMap, dedupeRefs, findAssetFilePath, isDistillCandidateRef, isLessonCandidate, isSignalDeltaEligible, } from "./eligibility.js";
27
29
  import { akmExtract, countNewExtractCandidates } from "./extract.js";
28
30
  import { computeValenceScore, FEEDBACK_WEIGHT, UTILITY_WEIGHT } from "./feedback-valence.js";
29
- import { makeGateConfig, resolveExtractConfidence, runAutoAcceptGate } from "./improve-auto-accept.js";
30
- import { resolveProcessEnabled } from "./improve-profiles.js";
31
31
  import { applyMemoryCleanup } from "./memory/memory-improve.js";
32
32
  import { computeProxyAdequacy, getAllAssetOutcomes, getOutcomeScoresByRef, OUTCOME_SCORE_MAX, outcomeScoreToSalience, updateAssetOutcome, } from "./outcome-loop.js";
33
33
  import { DEFAULT_DUE_DAYS, DEFAULT_MAX_PER_RUN, selectProactiveMaintenanceRefs } from "./proactive-maintenance.js";
34
- import { buildRankChangeReport, computeSalience, getAllRankScores, getAssetSalience, getConsecutiveNoOps, getLastUseMsByRef, isContentEncodingRow, SALIENCE_NO_OP_DAMPEN_FACTOR, SALIENCE_NO_OP_DAMPEN_THRESHOLD, upsertAssetSalience, } from "./salience.js";
35
- // ── improve preparation stage ───────────────────────
36
- // The pre-loop preparation pipeline (consolidation, session-extract, validation/
37
- // repair, eligibility partitioning, selectors) extracted from improve.ts.
38
- /**
39
- * #612 / WS-4 — bounded, opt-in per-phase auto-accept threshold auto-tune.
40
- *
41
- * Reads `improve.calibration` from config. When `autoTune` is enabled, computes
42
- * the calibration of recent gate decisions, derives a bounded threshold
43
- * adjustment (clamped into the configured band, capped per step), logs it, and
44
- * records a `calibration_autotune` event. Returns the new threshold (integer
45
- * 0-100) when an adjustment was made, or `undefined` to leave the caller's
46
- * threshold unchanged.
47
- *
48
- * WS-4 change: accepts an optional `phase` parameter. When provided, the tuned
49
- * threshold is persisted to `improve_gate_thresholds` (state.db Migration 012)
50
- * keyed by phase so `makeGateConfig` can read it back on the next run and each
51
- * phase maintains its own calibrated threshold rather than a shared global.
52
- *
53
- * WS-4 ceiling: `maxThreshold` defaults to 85 (not 100) to prevent the gate
54
- * converging to pure exploitation and shutting down Gap-3/4 novelty throughput.
55
- *
56
- * DEFAULT OFF: with no `improve.calibration` block (or `autoTune: false`) this
57
- * returns `undefined` immediately, so the gate threshold is unchanged and
58
- * behaviour is byte-identical to today.
59
- */
60
- export function maybeAutoTuneThreshold(currentThreshold, config, stateDbPath, ctx, phase) {
61
- const cal = config.improve?.calibration;
62
- if (!cal?.autoTune)
63
- return undefined;
64
- // WS-4: default maxThreshold is now 85 (ceiling to prevent pure exploitation).
65
- // Callers that explicitly set maxThreshold in config override this default.
66
- const tuneConfig = {
67
- autoTune: true,
68
- minThreshold: cal.minThreshold ?? 0,
69
- maxThreshold: cal.maxThreshold ?? 85,
70
- maxStep: cal.maxStep ?? 5,
71
- minSamples: cal.minSamples ?? 20,
72
- targetAcceptRate: cal.targetAcceptRate ?? 0.9,
73
- };
74
- // Defensive: an inverted band disables tuning rather than clamping to nonsense.
75
- if (tuneConfig.minThreshold > tuneConfig.maxThreshold)
76
- return undefined;
77
- const summary = withStateDb((db) => {
78
- const allDecisions = listProposalGateDecisions(db);
79
- // WS-4 fix: when called with a phase label, restrict calibration to that
80
- // phase's decision pool so a reflect-dominated run cannot tighten the
81
- // consolidate gate (or vice-versa). The gate field is `improve:<phase>`,
82
- // matching what improve-auto-accept.ts stamps at line ~163.
83
- const gateLabel = phase ? `improve:${phase}` : undefined;
84
- const decisions = gateLabel ? allDecisions.filter((d) => d.gate === gateLabel) : allDecisions;
85
- return summarizeCalibration(gateDecisionsToSamples(decisions));
86
- }, { path: stateDbPath });
87
- const result = computeThresholdAutoTune(currentThreshold, summary, tuneConfig);
88
- if (!result.adjusted)
89
- return undefined;
90
- const appendEventFn = ctx?.appendEventFn ?? appendEvent;
91
- const phaseLabel = phase ?? "global";
92
- info(`[improve] calibration auto-tune (${phaseLabel}): threshold ${result.previousThreshold} -> ${result.newThreshold} ` +
93
- `(${result.reason}; samples=${summary.samples}, acceptRate=${summary.overallAcceptRate}, ` +
94
- `gap=${summary.calibrationGap}, band=[${tuneConfig.minThreshold},${tuneConfig.maxThreshold}])`);
95
- try {
96
- appendEventFn({
97
- eventType: "calibration_autotune",
98
- ref: "improve:calibration",
99
- metadata: {
100
- phase: phaseLabel,
101
- previousThreshold: result.previousThreshold,
102
- newThreshold: result.newThreshold,
103
- delta: result.delta,
104
- reason: result.reason,
105
- samples: summary.samples,
106
- overallAcceptRate: summary.overallAcceptRate,
107
- calibrationGap: summary.calibrationGap,
108
- minThreshold: tuneConfig.minThreshold,
109
- maxThreshold: tuneConfig.maxThreshold,
110
- },
111
- });
112
- }
113
- catch (err) {
114
- warn(`[improve] calibration auto-tune event not recorded: ${err instanceof Error ? err.message : String(err)}`);
115
- }
116
- // WS-4: Persist the per-phase threshold so makeGateConfig reads it on the
117
- // next run. Best-effort — a write failure must not abort the improve run.
118
- if (phase) {
119
- try {
120
- withStateDb((persistDb) => persistPhaseThreshold(persistDb, phase, result.newThreshold), {
121
- path: stateDbPath,
122
- });
123
- }
124
- catch (err) {
125
- warn(`[improve] calibration auto-tune: failed to persist phase threshold for ${phase}: ${err instanceof Error ? err.message : String(err)}`);
126
- }
34
+ import { buildRankChangeReport, computeSalience, getAllRankScores, getAssetSalience, getLastUseMsByRef, isContentEncodingRow, SALIENCE_NO_OP_DAMPEN_FACTOR, SALIENCE_NO_OP_DAMPEN_THRESHOLD, upsertAssetSalience, } from "./salience.js";
35
+ import { bareImproveRef, improveStateReadRefs } from "./source-identity.js";
36
+ function readAssetSalienceForImproveRef(db, ref, itemRef) {
37
+ for (const key of improveStateReadRefs(ref, itemRef)) {
38
+ const row = getAssetSalience(db, key);
39
+ if (row)
40
+ return row;
127
41
  }
128
- return result.newThreshold;
42
+ return undefined;
43
+ }
44
+ function readConsecutiveNoOpsForImproveRef(db, ref, itemRef) {
45
+ return readAssetSalienceForImproveRef(db, ref, itemRef)?.consecutive_no_ops ?? 0;
46
+ }
47
+ // ── Durable-state write keys ──────────────────────────────────────────────────
48
+ //
49
+ // The durable improve-state writers (salience + outcome) key by the resolved
50
+ // index entry's `item_ref` (`<bundle>//<conceptId>`) when the planner resolved
51
+ // one (`ImproveEligibleRef.itemRef`). Both writers share one key expression,
52
+ // `itemRefByRef.get(ref) ?? ref`.
53
+ //
54
+ // A direct `--scope <ref>` candidate does not flow through
55
+ // `collectEligibleRefsFromIndex`, so its conceptId `ref` is the write key.
56
+ //
57
+ // itemRefByRef is `ref → item_ref | undefined`, built once per pass from the
58
+ // candidate set.
59
+ /** `ref → item_ref | undefined` for a run's candidate set. */
60
+ function buildItemRefByRef(refs) {
61
+ const m = new Map();
62
+ for (const r of refs)
63
+ m.set(r.ref, r.itemRef);
64
+ return m;
65
+ }
66
+ /** Durable `asset_salience` write key: the entry's item_ref, else its conceptId `ref` (scope-ref fallback). */
67
+ function salienceWriteKey(ref, itemRefByRef) {
68
+ return itemRefByRef.get(ref) ?? ref;
69
+ }
70
+ /** Durable `asset_outcome` write key: the entry's item_ref, else its conceptId `ref` (scope-ref fallback). */
71
+ function outcomeWriteKey(ref, itemRefByRef) {
72
+ return itemRefByRef.get(ref) ?? ref;
73
+ }
74
+ /** Resolve an AKM asset type from a short or bundle-qualified conceptId. */
75
+ function assetTypeOf(ref) {
76
+ const tail = ref.includes("//") ? ref.slice(ref.indexOf("//") + 2) : ref;
77
+ return typeNameFromConceptId(tail)?.type ?? "";
129
78
  }
130
79
  /**
131
- * Run (or gate-skip) the memory consolidation pass.
132
- *
133
- * #551 — two coordinated changes live here:
134
- *
135
- * 1. STRUCTURAL: this runs before extract in the improve pipeline (see
136
- * `runImprovePreparationStage`). Consolidation therefore only ever judges
137
- * PRIOR-run memories; current-run extract promotions are invisible to it.
138
- *
139
- * 2. SMARTER POOL-DELTA GATE: even among on-disk files, a memory whose only
140
- * post-`lastConsolidateTs` mtime bump came from its OWN auto-accept
141
- * promotion (i.e. it was just promoted by extract in the immediately
142
- * preceding run and has not had a full improve cycle to settle) does NOT
143
- * count as "work to do". We exclude those paths from the pool-delta check
144
- * using the `promoted` events already emitted with each promotion's
145
- * `assetPath`. A genuinely-settled prior memory — one edited by feedback,
146
- * reflect, manual edit, or simply older than the last consolidate — still
147
- * triggers the run. This is gate-option (a) from the issue (same-run /
148
- * adjacent-run promotion exclusion), chosen over option (b) because there
149
- * is no `extract_completed` event in the data model to gate against;
150
- * `promoted` events with `assetPath` already carry exactly the signal we
151
- * need, so the fix is non-invasive and provably correct.
80
+ * Evaluate the consolidation gate flags (volume trigger, #551 pool-delta
81
+ * cooldown, profile disable, #553 min-pool-size) up front, before any LLM call.
82
+ * Extracted verbatim from `runConsolidationPass` — logic is byte-identical.
152
83
  */
153
- export async function runConsolidationPass(args) {
154
- const { options, primaryStashDir, memorySummary, improveProfile, eventsCtx, budgetSignal, runBudgetMs } = args;
155
- const baseConfig = options.config ?? loadConfig();
84
+ function evaluateConsolidationEligibility(args) {
85
+ const { options, primaryStashDir, memorySummary, improveProfile, resolvedPlan } = args;
156
86
  const MEMORY_VOLUME_THRESHOLD = options.memoryVolumeConsolidationThreshold ?? 100;
157
- const hasLlm = !!(baseConfig.defaults?.llm || baseConfig.defaults?.agent);
87
+ const hasLlm = resolvedPlan.processes.consolidate.runner !== null;
158
88
  const volumeTriggered = typeof memorySummary.eligible === "number" && memorySummary.eligible > MEMORY_VOLUME_THRESHOLD && hasLlm;
159
- // When volume triggers a consolidation pass, force-enable the consolidate
160
- // process on the default improve profile so the gate accepts the run even
161
- // if the user's config disabled it. We synthesise a new profile override
162
- // rather than mutating connection settings.
163
- const consolidationConfig = volumeTriggered
164
- ? {
165
- ...baseConfig,
166
- profiles: {
167
- ...(baseConfig.profiles ?? {}),
168
- improve: {
169
- ...(baseConfig.profiles?.improve ?? {}),
170
- default: {
171
- ...(baseConfig.profiles?.improve?.default ?? {}),
172
- processes: {
173
- ...(baseConfig.profiles?.improve?.default?.processes ?? {}),
174
- consolidate: {
175
- ...(baseConfig.profiles?.improve?.default?.processes?.consolidate ?? {}),
176
- enabled: true,
177
- },
178
- },
179
- },
180
- },
181
- },
182
- }
183
- : baseConfig;
184
89
  // 0.8.0 pool-delta gate for consolidate: re-eligible iff at least one
185
90
  // memory file has been updated since the most recent successful
186
91
  // consolidate_completed event. Time-based cooldowns produced the same
187
92
  // synchronised-wave failure mode the reflect/distill cooldowns did; the
188
93
  // pool-delta gate ties consolidation to actual work-to-do.
94
+ const sourceName = options.sourceName ?? options.writeTarget?.source.name ?? options.config?.defaultBundle ?? "stash";
189
95
  const recentConsolidations = readEvents({ type: "consolidate_completed" });
190
96
  const lastConsolidation = recentConsolidations.events
191
- .filter((e) => e.metadata?.processed && Number(e.metadata.processed) > 0)
97
+ .filter((e) => e.metadata?.source === sourceName && Number(e.metadata?.processed) > 0)
192
98
  .sort((a, b) => new Date(b.ts ?? 0).getTime() - new Date(a.ts ?? 0).getTime())[0];
193
- const lastConsolidateTs = lastConsolidation?.ts;
99
+ const lastConsolidateTs = typeof lastConsolidation?.metadata?.completedThrough === "string"
100
+ ? lastConsolidation.metadata.completedThrough
101
+ : lastConsolidation?.ts;
194
102
  // #551 smarter gate: build the set of memory asset paths whose only delta
195
- // since the last consolidate is their OWN auto-accept promotion. Those files
103
+ // since the last consolidate is their OWN promotion. Those files
196
104
  // have not had a full improve cycle to settle, so they offer no merge /
197
105
  // contradiction candidates yet — excluding them stops the gate firing on
198
106
  // freshly-promoted single-source memories. We read `promoted` events emitted
@@ -235,21 +143,29 @@ export async function runConsolidationPass(args) {
235
143
  if (!fs.existsSync(memoriesDir))
236
144
  return false;
237
145
  try {
238
- return fs.readdirSync(memoriesDir).some((f) => {
239
- if (!f.endsWith(".md"))
240
- return false;
241
- const filePath = path.join(memoriesDir, f);
242
- // #551: skip files that were only touched by their own promotion this
243
- // cohort — they have no settled merge/contradiction candidates yet.
244
- if (promotedSinceConsolidate.has(path.resolve(filePath)))
245
- return false;
246
- try {
247
- return fs.statSync(filePath).mtime.toISOString() > lastConsolidateTs;
248
- }
249
- catch {
250
- return false;
146
+ const pending = [memoriesDir];
147
+ while (pending.length > 0) {
148
+ const current = pending.pop();
149
+ for (const entry of fs.readdirSync(current, { withFileTypes: true })) {
150
+ const filePath = path.join(current, entry.name);
151
+ if (entry.isDirectory()) {
152
+ pending.push(filePath);
153
+ continue;
154
+ }
155
+ if (!entry.isFile() || !entry.name.endsWith(".md"))
156
+ continue;
157
+ if (promotedSinceConsolidate.has(path.resolve(filePath)))
158
+ continue;
159
+ try {
160
+ if (fs.statSync(filePath).mtime.toISOString() > lastConsolidateTs)
161
+ return true;
162
+ }
163
+ catch {
164
+ // Ignore files that disappear during the scan.
165
+ }
251
166
  }
252
- });
167
+ }
168
+ return false;
253
169
  }
254
170
  catch {
255
171
  return false;
@@ -272,6 +188,21 @@ export async function runConsolidationPass(args) {
272
188
  // so a force-triggered run never trips the pool-size guard. The guard only
273
189
  // engages when minPoolSize > 0 and the eligible pool is strictly below it.
274
190
  const poolBelowMinSize = !volumeTriggered && minPoolSize > 0 && eligiblePoolSize < minPoolSize;
191
+ return {
192
+ volumeTriggered,
193
+ consolidationOnCooldown,
194
+ consolidateDisabledByProfile,
195
+ poolBelowMinSize,
196
+ eligiblePoolSize,
197
+ minPoolSize,
198
+ ...(lastConsolidateTs ? { lastConsolidationTs: lastConsolidateTs } : {}),
199
+ };
200
+ }
201
+ export async function runConsolidationPass(args) {
202
+ const { options, primaryStashDir, memorySummary, improveProfile, resolvedPlan, eventsCtx, budgetSignal, runBudgetMs, } = args;
203
+ const baseConfig = options.config ?? loadConfig();
204
+ const consolidationConfig = baseConfig;
205
+ const { volumeTriggered, consolidationOnCooldown, consolidateDisabledByProfile, poolBelowMinSize, eligiblePoolSize, minPoolSize, lastConsolidationTs, } = evaluateConsolidationEligibility({ options, primaryStashDir, memorySummary, improveProfile, resolvedPlan });
275
206
  let consolidation = {
276
207
  schemaVersion: 1,
277
208
  ok: true,
@@ -287,16 +218,6 @@ export async function runConsolidationPass(args) {
287
218
  warnings: [],
288
219
  durationMs: 0,
289
220
  };
290
- let gateAutoAcceptedCount = 0;
291
- let gateAutoAcceptFailedCount = 0;
292
- const consolidateGateCfg = makeGateConfig("consolidate", {
293
- globalThreshold: options.autoAccept,
294
- dryRun: options.dryRun ?? false,
295
- stashDir: primaryStashDir,
296
- config: consolidationConfig,
297
- eventsCtx,
298
- stateDbPath: eventsCtx?.dbPath,
299
- }, { minimumThreshold: 95 });
300
221
  if (consolidateDisabledByProfile) {
301
222
  info("[improve] consolidation skipped (disabled by improve profile)");
302
223
  }
@@ -306,7 +227,7 @@ export async function runConsolidationPass(args) {
306
227
  // it via the dynamic skipReasons aggregation under `pool_below_min_size`.
307
228
  appendEvent({
308
229
  eventType: "improve_skipped",
309
- ref: "memory:_consolidation",
230
+ ref: "memories/_consolidation",
310
231
  metadata: {
311
232
  reason: "pool_below_min_size",
312
233
  poolSize: eligiblePoolSize,
@@ -316,13 +237,18 @@ export async function runConsolidationPass(args) {
316
237
  info(`[improve] consolidation skipped (pool ${eligiblePoolSize} < minPoolSize ${minPoolSize})`);
317
238
  }
318
239
  else if (!consolidationOnCooldown) {
240
+ const consolidationStartedAt = new Date().toISOString();
319
241
  consolidation = await withLlmStage("consolidate", () => akmConsolidate({
320
242
  ...options.consolidateOptions,
321
243
  config: consolidationConfig,
244
+ dryRun: options.dryRun ?? false,
322
245
  stashDir: options.stashDir,
323
246
  // Active profile for this improve run — lets consolidate's secondary
324
247
  // process-config reads honor `--profile <name>` instead of `default`.
325
248
  improveProfile,
249
+ llmConfig: resolvedPlan.processes.consolidate.runner
250
+ ? materializeLlmRunnerConnection(resolvedPlan.processes.consolidate.runner)
251
+ : null,
326
252
  autoTriggered: volumeTriggered,
327
253
  // Tie consolidate proposals back to this improve invocation so
328
254
  // accept-rate-per-run aggregation works. Mirrors reflect/propose/extract.
@@ -331,25 +257,9 @@ export async function runConsolidationPass(args) {
331
257
  // recently-changed memories + graph neighbours — use this for frequent
332
258
  // passes (quick-shredder). Leave absent in the nightly default profile for
333
259
  // a full-pool sweep that catches stale-but-unmerged duplicates.
334
- incrementalSince: improveProfile?.processes?.consolidate?.incrementalSince,
335
260
  limit: improveProfile?.processes?.consolidate?.limit,
336
261
  neighborsPerChanged: improveProfile?.processes?.consolidate?.neighborsPerChanged,
337
262
  maxChunkSize: improveProfile?.processes?.consolidate?.maxChunkSize,
338
- // #617 — deterministic near-duplicate dedup pre-pass. DEFAULT OFF; only
339
- // runs when the profile explicitly sets `consolidate.dedup.enabled`.
340
- dedup: improveProfile?.processes?.consolidate?.dedup,
341
- // #581 — judged-state cache. DEFAULT OFF; only engages when the profile
342
- // explicitly sets `consolidate.judgedCache.enabled`. Skips memories
343
- // judged-unchanged since their last judge so one run sweeps the full
344
- // corpus instead of narrowing to a time-window slice.
345
- judgedCache: improveProfile?.processes?.consolidate?.judgedCache,
346
- // Honor profile.autoAccept (already merged into options.autoAccept at the
347
- // top of akmImprove). The CLI parser always supplies 90 when --auto-accept
348
- // is absent, so ?? 90 is not needed here and would prevent --auto-accept=false
349
- // (which maps to undefined) from disabling consolidation auto-accept.
350
- // options.consolidateOptions.autoAccept (if explicitly provided by caller)
351
- // still wins because the spread above runs first.
352
- autoAccept: options.consolidateOptions?.autoAccept ?? options.autoAccept,
353
263
  // WS-3a: forward budget signal for graceful abort on timeout, and pass
354
264
  // the profile's p90 estimate for cold-start budget reduction.
355
265
  signal: budgetSignal,
@@ -357,28 +267,25 @@ export async function runConsolidationPass(args) {
357
267
  // WS-5: pass total run budget so perfTelemetry.estimatedBudgetFractionUsed
358
268
  // can flag when consolidation alone exceeded the budget.
359
269
  runBudgetMs,
360
- }));
361
- {
362
- const consolidateGr = await runAutoAcceptGate(consolidation.promoted.map((proposalId) => {
363
- try {
364
- if (!primaryStashDir)
365
- return { proposalId, confidence: undefined };
366
- const proposal = getProposal(primaryStashDir, proposalId);
367
- return { proposalId, confidence: proposal.confidence };
368
- }
369
- catch {
370
- return { proposalId, confidence: undefined };
371
- }
372
- }), consolidateGateCfg);
373
- gateAutoAcceptedCount += consolidateGr.promoted.length;
374
- gateAutoAcceptFailedCount += consolidateGr.failed.length;
375
- }
376
- if (consolidation.processed > 0) {
270
+ }), { engine: resolvedPlan.processes.consolidate.runner?.engine, process: "consolidate" });
271
+ const sourceName = options.sourceName ?? options.writeTarget?.source.name ?? baseConfig.defaultBundle ?? "stash";
272
+ const complete = (consolidation.failedChunks ?? 0) === 0 &&
273
+ (consolidation.failedChunkMemories ?? 0) === 0 &&
274
+ (consolidation.failedPromotions ?? 0) === 0 &&
275
+ (consolidation.deferredMemories ?? 0) === 0;
276
+ const hasUnappliedAdvisoryOperations = consolidation.planned?.some((op) => op.op !== "promote") ?? false;
277
+ if (consolidation.ok &&
278
+ !consolidation.dryRun &&
279
+ complete &&
280
+ !hasUnappliedAdvisoryOperations &&
281
+ consolidation.processed > 0) {
377
282
  appendEvent({
378
283
  eventType: "consolidate_completed",
379
- ref: "memory:_consolidation",
284
+ ref: makeBundleRef(sourceName, "memories/_consolidation"),
380
285
  metadata: {
381
286
  processed: consolidation.processed,
287
+ source: sourceName,
288
+ completedThrough: consolidationStartedAt,
382
289
  merged: consolidation.merged,
383
290
  deleted: consolidation.deleted,
384
291
  contradicted: consolidation.contradicted,
@@ -391,38 +298,30 @@ export async function runConsolidationPass(args) {
391
298
  else {
392
299
  appendEvent({
393
300
  eventType: "improve_skipped",
394
- ref: "memory:_consolidation",
301
+ ref: "memories/_consolidation",
395
302
  metadata: {
396
303
  reason: "consolidation_no_memory_updates",
397
- lastEventTs: lastConsolidation?.ts ?? null,
304
+ lastEventTs: lastConsolidationTs ?? null,
398
305
  },
399
306
  }, eventsCtx);
400
307
  info("[improve] consolidation skipped (no memory updates since last run)");
401
308
  }
402
309
  // D9: track whether consolidation wrote any data so graph extraction can reindex if needed
403
- const consolidationRan = !consolidateDisabledByProfile && !poolBelowMinSize && !consolidationOnCooldown && consolidation.processed > 0;
404
- // WS-4: Per-phase threshold auto-tune for the consolidate phase.
405
- // Persists result for the NEXT run's makeGateConfig to read.
406
- const consolidateTuneDbPath = eventsCtx?.dbPath;
407
- if (options.autoAccept !== undefined && consolidateTuneDbPath) {
408
- try {
409
- maybeAutoTuneThreshold(consolidateGateCfg.phaseThreshold ?? options.autoAccept, consolidationConfig, consolidateTuneDbPath, undefined, "consolidate");
410
- }
411
- catch (err) {
412
- warn(`[improve] calibration auto-tune (consolidate) skipped: ${err instanceof Error ? err.message : String(err)}`);
413
- }
414
- }
415
- return { consolidation, consolidationRan, gateAutoAcceptedCount, gateAutoAcceptFailedCount };
310
+ const consolidationRan = !consolidateDisabledByProfile &&
311
+ !poolBelowMinSize &&
312
+ !consolidationOnCooldown &&
313
+ !consolidation.previewOnly &&
314
+ consolidation.processed > 0;
315
+ return { consolidation, consolidationRan };
416
316
  }
417
317
  /**
418
318
  * Phase 0.4 — session-extract pass. Reads native session files through the
419
- * SessionLogHarness registry, asks a bounded LLM for candidate proposals, gates
420
- * them, and drains the extract backlog. Failures are non-fatal (collected into
421
- * `warnings`). Returns the extract results + the gate counters seeded from the
422
- * consolidation pass and accumulated here.
319
+ * SessionLogHarness registry, and asks a bounded LLM for candidate proposals.
320
+ * Failures are non-fatal (collected into `warnings`). Returns the extract
321
+ * results + any warnings collected along the way.
423
322
  */
424
323
  async function runSessionExtractPass(args) {
425
- const { options, primaryStashDir, improveProfile, eventsCtx, seedGateAccepted, seedGateFailed } = args;
324
+ const { options, primaryStashDir, improveProfile, resolvedPlan, eventsCtx, budgetSignal } = args;
426
325
  const warnings = [];
427
326
  // Phase 0.4 — session-extract pass.
428
327
  //
@@ -433,9 +332,10 @@ async function runSessionExtractPass(args) {
433
332
  // / `akm feedback` invocations. Replaces the akm-plugin session-checkpoint
434
333
  // hook with an on-demand pull pipeline.
435
334
  //
436
- // Default-on; opt out via the ACTIVE profile's `processes.extract.enabled: false`
437
- // (#593: the gate respects the resolved improve profile, not just the
438
- // hardcoded `default` profile path the legacy feature flag reads).
335
+ // Runs only when the ACTIVE strategy resolves
336
+ // `processes.extract.enabled: true` (#593: the gate respects the resolved
337
+ // improve strategy, not just the hardcoded `default` path the legacy feature
338
+ // flag read). Shipped `default` and `frequent` strategies leave this off.
439
339
  // Each available harness gets one call with the default --since window;
440
340
  // already-seen sessions (tracked in state.db.extract_sessions_seen) are
441
341
  // skipped automatically so re-runs don't burn LLM calls on unchanged data.
@@ -443,30 +343,17 @@ async function runSessionExtractPass(args) {
443
343
  // Failures are non-fatal — one harness throwing doesn't abort improve.
444
344
  // The extract envelope's own `warnings` field surfaces what went wrong.
445
345
  let extractResults;
446
- // Seed the preparation-stage gate counters with consolidation's auto-accept
447
- // gate results (#551: consolidation now runs in this stage), then accumulate
448
- // extract's gate results on top.
449
- let gateAutoAcceptedCount = seedGateAccepted;
450
- let gateAutoAcceptFailedCount = seedGateFailed;
451
346
  const extractConfig = options.config ?? loadConfig();
452
- const extractGateCfg = makeGateConfig("extract", {
453
- globalThreshold: options.autoAccept,
454
- dryRun: options.dryRun ?? false,
455
- stashDir: primaryStashDir,
456
- config: extractConfig,
457
- eventsCtx,
458
- stateDbPath: eventsCtx?.dbPath,
459
- });
460
347
  // #554 minNewSessions gate: skip the entire extract pass (ensureIndex was
461
348
  // already done upstream; here we elide every akmExtract/processSession call)
462
349
  // when the NEW (unseen, in-window) candidate-session pool is below a minimum.
463
350
  // 22% of improve runs produce zero memory-inference writes because extract
464
351
  // finds no new sessions, yet still burns the full extract pipeline. Default 0
465
352
  // (disabled) preserves existing always-run behaviour; only opted-in profiles
466
- // (e.g. `frequent`) set it. Evaluated BEFORE any LLM call so a skip costs zero
467
- // LLM work AND writes nothing — which also means no extract auto-accept bumps
468
- // memory mtimes, so a skipped extract never flags work for the NEXT run's
469
- // consolidation mtime-gate (the downstream trigger #554 asks us to suppress).
353
+ // (e.g. a user-enabled `frequent` strategy) set it. Evaluated BEFORE any LLM
354
+ // call so a skip costs zero LLM work AND writes nothing. A skipped extract
355
+ // never flags work for the NEXT run's consolidation mtime-gate (the
356
+ // downstream trigger #554 asks us to suppress).
470
357
  const EXTRACT_DEFAULT_MIN_NEW_SESSIONS = 0;
471
358
  // Read from the ACTIVE resolved profile (not always `default`), matching how
472
359
  // `extract.enabled` resolves — otherwise a non-default profile (e.g.
@@ -476,11 +363,22 @@ async function runSessionExtractPass(args) {
476
363
  // #593/#594: the ACTIVE resolved improve profile is the single source of
477
364
  // truth for whether extract runs. (Previously this also ANDed in the legacy
478
365
  // `session_extraction` feature flag, which only reads
479
- // `profiles.improve.default.processes.extract.enabled`; that made the default
480
- // profile a global kill switch, so a non-default profile enabling extract was
481
- // silently overridden. The default profile is now just another profile.)
366
+ // a retired global feature path; the selected strategy is authoritative.)
482
367
  // `akmExtract` re-checks the same active profile internally via `improveProfile`.
483
- if (resolveProcessEnabled("extract", improveProfile)) {
368
+ if (resolvedPlan.processes.extract.enabled) {
369
+ const extractRunner = resolvedPlan.processes.extract.runner;
370
+ if (!extractRunner?.engine) {
371
+ throw new ConfigError("Resolved improve plan has no runner for enabled extract process.", "LLM_NOT_CONFIGURED");
372
+ }
373
+ const extractPlan = Object.freeze({
374
+ strategy: resolvedPlan.strategy.name,
375
+ engine: extractRunner.engine,
376
+ enabled: true,
377
+ process: resolvedPlan.processes.extract.config,
378
+ runner: extractRunner,
379
+ timeoutMs: extractRunner.timeoutMs === undefined ? 600_000 : extractRunner.timeoutMs,
380
+ embeddingConfig: Object.freeze(structuredClone(extractConfig.embedding)),
381
+ });
484
382
  const availableHarnesses = options.extractHarnesses ?? getAvailableHarnesses();
485
383
  // The guard engages only when minNewSessions > 0; 0 disables it entirely.
486
384
  let belowMinNewSessions = false;
@@ -503,7 +401,7 @@ async function runSessionExtractPass(args) {
503
401
  // skipReasons aggregation surfaces this under `below_min_new_sessions`.
504
402
  appendEvent({
505
403
  eventType: "improve_skipped",
506
- ref: "memory:_extract",
404
+ ref: "memories/_extract",
507
405
  metadata: {
508
406
  reason: "below_min_new_sessions",
509
407
  newSessions: newCandidateCount,
@@ -521,25 +419,17 @@ async function runSessionExtractPass(args) {
521
419
  type: h.name,
522
420
  ...(primaryStashDir !== undefined ? { stashDir: primaryStashDir } : {}),
523
421
  config: extractConfig,
524
- // Thread the ACTIVE profile so extract's internal gate + per-process
525
- // config read the running profile, not always `default`.
526
- improveProfile,
422
+ resolvedPlan: extractPlan,
527
423
  dryRun: options.dryRun ?? false,
424
+ signal: budgetSignal,
528
425
  ...(options.extractHarnesses ? { harnesses: options.extractHarnesses } : {}),
529
426
  // C2: pin extract's skip-tracking state.db open to the boundary path.
530
427
  ...(eventsCtx?.dbPath ? { stateDbPath: eventsCtx.dbPath } : {}),
531
- }));
428
+ // R25: extract's event emits reuse the run's events context
429
+ // (fast path when it carries the long-lived handle).
430
+ eventsCtx,
431
+ }), { engine: resolvedPlan.processes.extract.runner?.engine, process: "extract" });
532
432
  extractResults.push(result);
533
- {
534
- const gr = await runAutoAcceptGate(primaryStashDir
535
- ? result.proposals.map((proposalId) => {
536
- const proposal = getProposal(primaryStashDir, proposalId);
537
- return { proposalId, confidence: resolveExtractConfidence(proposal) };
538
- })
539
- : [], extractGateCfg);
540
- gateAutoAcceptedCount += gr.promoted.length;
541
- gateAutoAcceptFailedCount += gr.failed.length;
542
- }
543
433
  }
544
434
  catch (err) {
545
435
  const msg = err instanceof Error ? err.message : String(err);
@@ -553,73 +443,57 @@ async function runSessionExtractPass(args) {
553
443
  }
554
444
  }
555
445
  }
556
- // Backlog drain: gate any pending extract proposals that weren't created in
557
- // this run (i.e. pre-date the gate or were produced by a run that timed out
558
- // before the gate fired). Without this, eligible proposals accumulate
559
- // indefinitely — the fresh-gate only covers the current run's output.
560
- if (primaryStashDir && !options.dryRun && options.autoAccept !== undefined) {
561
- const freshIds = new Set((extractResults ?? []).flatMap((r) => r.proposals));
562
- const backlog = listProposals(primaryStashDir, { status: "pending" }).filter((p) => p.source === "extract" && !freshIds.has(p.id));
563
- if (backlog.length > 0) {
564
- const backlogCandidates = backlog.map((p) => ({
565
- proposalId: p.id,
566
- confidence: resolveExtractConfidence(p),
567
- }));
568
- const backlogGr = await runAutoAcceptGate(backlogCandidates, extractGateCfg);
569
- gateAutoAcceptedCount += backlogGr.promoted.length;
570
- gateAutoAcceptFailedCount += backlogGr.failed.length;
571
- }
572
- }
573
- return { extractResults, gateAutoAcceptedCount, gateAutoAcceptFailedCount, warnings, extractGateCfg };
446
+ return {
447
+ extractResults,
448
+ warnings,
449
+ };
574
450
  }
575
451
  /**
576
452
  * Phase 1 — validation + schema-repair pass. Scans postCleanupRefs for assets
577
453
  * with structural problems (missing file, missing lesson description), attempts
578
454
  * LLM schema repair, and returns the still-failing ref set + the repair records.
579
455
  */
580
- async function runValidationAndRepairPass(args) {
581
- const { postCleanupRefs, options, startMs, budgetMs, primaryStashDir } = args;
582
- const validationFailures = [];
583
- for (const candidate of postCleanupRefs) {
456
+ export async function runValidationAndRepairPass(args) {
457
+ const { postCleanupRefs, options, startMs, budgetMs, primaryStashDir, resolvedPlan, repairValidationFailures, schemaRepairFn = runSchemaRepairPass, } = args;
458
+ const validateCandidate = async (candidate) => {
584
459
  try {
585
- // #591: use the path pre-resolved at planning time when it is still on
586
- // disk — a serial async DB lookup per ref cost ~500 s on a 9 000-ref
587
- // stash. Fall back to findAssetFilePath only for refs that bypassed
588
- // collectEligibleRefs' index scan or whose file moved since planning.
589
460
  const filePath = candidate.filePath && fs.existsSync(candidate.filePath)
590
461
  ? candidate.filePath
591
462
  : await findAssetFilePath(candidate.ref, options.stashDir);
592
- if (!filePath) {
593
- validationFailures.push({ ref: candidate.ref, reason: "file not found on disk" });
594
- continue;
595
- }
596
- if (path.extname(filePath).toLowerCase() !== ".md") {
597
- continue;
598
- }
463
+ if (!filePath)
464
+ return "file not found on disk";
465
+ if (path.extname(filePath).toLowerCase() !== ".md")
466
+ return undefined;
599
467
  if (isLessonCandidate(candidate.ref)) {
600
- const raw = fs.readFileSync(filePath, "utf8");
601
- const fm = parseFrontmatter(raw).data;
468
+ const fm = parseFrontmatter(fs.readFileSync(filePath, "utf8")).data;
602
469
  if (!fm.description)
603
- validationFailures.push({ ref: candidate.ref, reason: "missing description" });
470
+ return "missing description";
604
471
  }
472
+ return undefined;
605
473
  }
606
- catch (e) {
607
- validationFailures.push({ ref: candidate.ref, reason: String(e) });
474
+ catch (error) {
475
+ return String(error);
608
476
  }
477
+ };
478
+ const validationFailures = [];
479
+ for (const candidate of postCleanupRefs) {
480
+ const reason = await validateCandidate(candidate);
481
+ if (reason)
482
+ validationFailures.push({ ref: candidate.ref, reason });
609
483
  }
610
484
  if (validationFailures.length > 0) {
611
- info(`[improve] ${validationFailures.length} assets have validation issues (will attempt schema repair):`);
485
+ info(`[improve] ${validationFailures.length} assets have validation issues${repairValidationFailures ? " (will attempt schema repair)" : ""}:`);
612
486
  for (const f of validationFailures)
613
487
  info(` ${f.ref}: ${f.reason}`);
614
488
  }
615
489
  let schemaRepairs = [];
616
- let repairedRefs = new Set();
490
+ const repairedRefs = new Set();
617
491
  // Schema repair pass: attempt to fix validation failures via LLM before skipping.
618
- if (validationFailures.length > 0 && options.repairValidationFailures !== false) {
619
- const baseConfigForRepair = options.config ?? loadConfig();
620
- const llmCfg = getDefaultLlmConfig(baseConfigForRepair);
492
+ if (validationFailures.length > 0) {
493
+ const validationRunner = resolvedPlan.processes.validation.runner;
494
+ const llmCfg = validationRunner ? materializeLlmRunnerConnection(validationRunner) : undefined;
621
495
  if (llmCfg) {
622
- const result = await runSchemaRepairPass(validationFailures, {
496
+ const result = await withLlmStage("validation", () => schemaRepairFn(validationFailures, {
623
497
  startMs,
624
498
  budgetMs,
625
499
  llmConfig: llmCfg,
@@ -633,9 +507,17 @@ async function runValidationAndRepairPass(args) {
633
507
  stashDir: primaryStashDir,
634
508
  findFilePath: findAssetFilePath,
635
509
  isLessonCandidateFn: isLessonCandidate,
636
- });
510
+ }), { engine: resolvedPlan.processes.validation.runner?.engine, process: "validation" });
637
511
  schemaRepairs = result.repairs;
638
- repairedRefs = result.repairedRefs;
512
+ // A repair result is advisory. Only a fresh structural read of the live
513
+ // asset can remove it from the failure set; queued content is not live.
514
+ const failedRefs = new Set(validationFailures.map((failure) => failure.ref));
515
+ const candidatesByRef = new Map(postCleanupRefs.map((candidate) => [candidate.ref, candidate]));
516
+ for (const ref of failedRefs) {
517
+ const candidate = candidatesByRef.get(ref);
518
+ if (candidate && !(await validateCandidate(candidate)))
519
+ repairedRefs.add(ref);
520
+ }
639
521
  }
640
522
  }
641
523
  const validationFailureRefs = new Set(validationFailures.filter((f) => !repairedRefs.has(f.ref)).map((f) => f.ref));
@@ -645,32 +527,18 @@ async function runValidationAndRepairPass(args) {
645
527
  return { validationFailures, validationFailureRefs, schemaRepairs };
646
528
  }
647
529
  export async function runImprovePreparationStage(args) {
648
- const { scope, options, plannedRefs, memoryCleanupPlan, primaryStashDir, memorySummary, reindexFn, startMs, budgetMs, eventsCtx, initialCleanupWarnings, improveProfile, budgetSignal, } = args;
530
+ const { scope, options, plannedRefs, memoryCleanupPlan, primaryStashDir, memorySummary, reindexFn, startMs, budgetMs, eventsCtx, initialCleanupWarnings, improveProfile, resolvedPlan, strategyName, budgetSignal, } = args;
649
531
  const actions = [];
650
532
  const cleanupWarnings = initialCleanupWarnings ? [...initialCleanupWarnings] : [];
651
- // Phase 0 — MEMORY.md budget check (200-line cap; warn at 180)
652
- let memoryIndexHealth;
653
- if (primaryStashDir) {
654
- const memoryMdPath = path.join(primaryStashDir, "memories", "MEMORY.md");
655
- if (fs.existsSync(memoryMdPath)) {
656
- try {
657
- const lines = fs.readFileSync(memoryMdPath, "utf8").split("\n").length;
658
- const overBudget = lines >= 180;
659
- memoryIndexHealth = { lineCount: lines, overBudget };
660
- if (overBudget) {
661
- cleanupWarnings.push(`MEMORY.md has ${lines} lines (budget: 200). Consolidation strongly recommended.`);
662
- }
663
- }
664
- catch {
665
- // best-effort
666
- }
667
- }
668
- }
533
+ const memoryBudget = assessMemoryIndexBudget(primaryStashDir);
534
+ const memoryIndexHealth = memoryBudget.memoryIndexHealth;
535
+ if (memoryBudget.warning)
536
+ cleanupWarnings.push(memoryBudget.warning);
669
537
  // Phase 0.3 — memory consolidation pass (#551).
670
538
  //
671
539
  // Consolidation runs BEFORE the session-extract pass. This is the structural
672
- // half of the #551 fix: extract auto-accept writes brand-new memory .md files
673
- // on every run, which previously made the consolidation pool-delta gate fire
540
+ // half of the #551 fix: extract promotions write brand-new memory .md files,
541
+ // which previously made the consolidation pool-delta gate fire
674
542
  // unconditionally (any new file => "memory updated since last consolidate").
675
543
  // By running consolidation first, the gate and akmConsolidate only ever see
676
544
  // memories that existed at the start of the run — current-run extract
@@ -681,6 +549,7 @@ export async function runImprovePreparationStage(args) {
681
549
  primaryStashDir,
682
550
  memorySummary,
683
551
  improveProfile,
552
+ resolvedPlan,
684
553
  eventsCtx,
685
554
  budgetSignal,
686
555
  runBudgetMs: budgetMs,
@@ -690,13 +559,11 @@ export async function runImprovePreparationStage(args) {
690
559
  options,
691
560
  primaryStashDir,
692
561
  improveProfile,
562
+ resolvedPlan,
693
563
  eventsCtx,
694
- seedGateAccepted: consolidationPass.gateAutoAcceptedCount,
695
- seedGateFailed: consolidationPass.gateAutoAcceptFailedCount,
564
+ budgetSignal,
696
565
  });
697
566
  const extractResults = extractPass.extractResults;
698
- const gateAutoAcceptedCount = extractPass.gateAutoAcceptedCount;
699
- const gateAutoAcceptFailedCount = extractPass.gateAutoAcceptFailedCount;
700
567
  if (extractPass.warnings.length > 0)
701
568
  cleanupWarnings.push(...extractPass.warnings);
702
569
  // eligibleCount = raw pre-filter count (before cooldown/signal/cleanup filters).
@@ -704,19 +571,167 @@ export async function runImprovePreparationStage(args) {
704
571
  appendEvent({
705
572
  eventType: "improve_invoked",
706
573
  ref: scope.mode === "ref" ? scope.value : `improve:${scope.mode}:${scope.value ?? "all"}`,
707
- metadata: { scope, dryRun: options.dryRun ?? false, eligibleCount: plannedRefs.length },
574
+ metadata: { strategy: strategyName, scope, dryRun: options.dryRun ?? false, eligibleCount: plannedRefs.length },
708
575
  }, eventsCtx);
709
576
  // ensureIndex now runs in akmImprove() BEFORE collectEligibleRefs so the
710
577
  // eligible-ref query sees a populated `entries` table on the very first
711
578
  // pass after a DB version upgrade (#339). Any failure messages from that
712
579
  // earlier call were threaded in via args.initialCleanupWarnings.
580
+ const cleanup = await applyCleanupPass({
581
+ primaryStashDir,
582
+ memoryCleanupPlan,
583
+ plannedRefs,
584
+ reindexFn,
585
+ budgetSignal,
586
+ allowApply: isAutonomyLaneAllowed("memoryCleanup", options.config ?? loadConfig()),
587
+ });
588
+ const appliedCleanup = cleanup.appliedCleanup;
589
+ const postCleanupRefs = cleanup.postCleanupRefs;
590
+ actions.push(...cleanup.pruneActions);
591
+ cleanupWarnings.push(...cleanup.warnings);
592
+ const { validationFailures, validationFailureRefs, schemaRepairs } = await runValidationAndRepairPass({
593
+ postCleanupRefs,
594
+ options,
595
+ startMs,
596
+ budgetMs,
597
+ primaryStashDir,
598
+ resolvedPlan,
599
+ repairValidationFailures: resolvedPlan.processes.validation.enabled && options.repairValidationFailures !== false,
600
+ });
601
+ // Phase 0.5 — structural hygiene pass
602
+ let lintSummary;
603
+ if (primaryStashDir) {
604
+ try {
605
+ const lintResult = akmLint({ fix: false, dir: primaryStashDir });
606
+ lintSummary = { fixed: lintResult.summary.fixed, flagged: lintResult.summary.flagged };
607
+ }
608
+ catch {
609
+ // lint is best-effort; never block improve
610
+ }
611
+ }
612
+ const recentErrors = seedRecentErrorWindows(schemaRepairs);
613
+ const snapshot = buildSnapshotManifest({ postCleanupRefs, validationFailureRefs });
614
+ const gathered = gatherCandidates({
615
+ scope,
616
+ options,
617
+ primaryStashDir,
618
+ eventsCtx,
619
+ improveProfile,
620
+ resolvedPlan,
621
+ postCleanupRefs,
622
+ validationFailureRefs,
623
+ snapshot,
624
+ });
625
+ actions.push(...gathered.actions);
626
+ const scored = scoreSalience({
627
+ scope,
628
+ options,
629
+ primaryStashDir,
630
+ eventsCtx,
631
+ mergedRefs: gathered.mergedRefs,
632
+ eligibilitySourceByRef: gathered.eligibilitySourceByRef,
633
+ feedbackSummary: gathered.feedbackSummary,
634
+ retrievalCounts: gathered.retrievalCounts,
635
+ signalFiltered: gathered.signalFiltered,
636
+ proactiveRefs: gathered.proactiveRefs,
637
+ highSalienceRefs: gathered.highSalienceRefs,
638
+ });
639
+ const filtered = await filterEligibility({
640
+ scope,
641
+ options,
642
+ plannedRefs,
643
+ eventsCtx,
644
+ mergedRefs: scored.mergedRefs,
645
+ salienceMap: scored.salienceMap,
646
+ eligibilitySourceByRef: gathered.eligibilitySourceByRef,
647
+ distillOnlyRefs: gathered.distillOnlyRefs,
648
+ validationFailureRefs,
649
+ summary: {
650
+ fullySkippedCount: gathered.fullySkippedCount,
651
+ preCooldownCount: gathered.preCooldownCount,
652
+ signalAndRetrievalRefs: gathered.signalAndRetrievalRefs,
653
+ signalFiltered: gathered.signalFiltered,
654
+ },
655
+ });
656
+ return {
657
+ actions,
658
+ cleanupWarnings,
659
+ appliedCleanup,
660
+ memoryIndexHealth,
661
+ extract: extractResults,
662
+ actionableRefs: filtered.actionableRefs,
663
+ signalBearingSet: gathered.signalBearingSet,
664
+ validationFailures,
665
+ schemaRepairs,
666
+ lintSummary,
667
+ loopRefs: filtered.loopRefs,
668
+ distillCooledRefs: gathered.distillCooledRefs,
669
+ distillOnlyRefs: filtered.distillOnlyRefs,
670
+ coverageGaps: filtered.coverageGaps,
671
+ recentErrors,
672
+ utilityMap: scored.utilityMap,
673
+ consolidation: consolidationPass.consolidation,
674
+ consolidationRan: consolidationPass.consolidationRan,
675
+ ...(gathered.proactiveMaintenanceSummary ? { proactiveMaintenance: gathered.proactiveMaintenanceSummary } : {}),
676
+ };
677
+ }
678
+ // ── preparation-stage passes (WI-7.6 decomposition, R31) ────────────────────
679
+ // The six-pass split prescribed by the chunk-7 brief §WI-7.6, adapted to the
680
+ // code as it exists at HEAD (anchors re-measured; see the chunk-7 ledger):
681
+ // snapshot-manifest → buildSnapshotManifest
682
+ // candidate-gather → gatherCandidates (+ its five lane/sub-passes)
683
+ // salience-score → scoreSalience (+ outcome/vector/persist sub-passes)
684
+ // valence-score → the two computeValenceScore call sites move VERBATIM
685
+ // inside the salience passes (pure fn; no separate pass)
686
+ // standards-context → does not exist in preparation.ts (assembly lives in
687
+ // extract.ts — recorded in the ledger, no empty pass)
688
+ // eligibility-filter → filterEligibility (+ replay/disk-check sub-passes)
689
+ // Every pass takes an args object and returns its results; the orchestrator
690
+ // folds them. Shared-by-reference structures (the ImproveEligibleRef objects,
691
+ // eligibilitySourceByRef, salienceMap, actions, recentErrors) keep their
692
+ // identity — attribution stamps must travel with the ref objects into the
693
+ // loop stage exactly as before.
694
+ /** Phase 0 — MEMORY.md budget check (200-line cap; warn at 180). */
695
+ function assessMemoryIndexBudget(primaryStashDir) {
696
+ let warning;
697
+ // Phase 0 — MEMORY.md budget check (200-line cap; warn at 180)
698
+ let memoryIndexHealth;
699
+ if (primaryStashDir) {
700
+ const memoryMdPath = path.join(primaryStashDir, "memories", "MEMORY.md");
701
+ if (fs.existsSync(memoryMdPath)) {
702
+ try {
703
+ const lines = fs.readFileSync(memoryMdPath, "utf8").split("\n").length;
704
+ const overBudget = lines >= 180;
705
+ memoryIndexHealth = { lineCount: lines, overBudget };
706
+ if (overBudget) {
707
+ warning = `MEMORY.md has ${lines} lines (budget: 200). Consolidation strongly recommended.`;
708
+ }
709
+ }
710
+ catch {
711
+ // best-effort
712
+ }
713
+ }
714
+ }
715
+ return { memoryIndexHealth, warning };
716
+ }
717
+ /**
718
+ * Memory-cleanup apply + prune-action recording + the post-cleanup reindex.
719
+ * Returns the surviving ref set and the prune actions/warnings for the
720
+ * orchestrator to fold (same order as the old inline pushes).
721
+ */
722
+ async function applyCleanupPass(args) {
723
+ const { primaryStashDir, memoryCleanupPlan, plannedRefs, reindexFn, budgetSignal, allowApply } = args;
724
+ const pruneActions = [];
725
+ const warnings = [];
713
726
  let appliedCleanup;
714
727
  try {
715
728
  appliedCleanup =
716
- primaryStashDir && memoryCleanupPlan ? applyMemoryCleanup(primaryStashDir, memoryCleanupPlan) : undefined;
729
+ primaryStashDir && memoryCleanupPlan && allowApply
730
+ ? applyMemoryCleanup(primaryStashDir, memoryCleanupPlan)
731
+ : undefined;
717
732
  }
718
733
  catch (err) {
719
- cleanupWarnings.push(`applyMemoryCleanup failed: ${err instanceof Error ? err.message : String(err)}`);
734
+ warnings.push(`applyMemoryCleanup failed: ${err instanceof Error ? err.message : String(err)}`);
720
735
  }
721
736
  const archivedRefs = appliedCleanup?.archived.map((record) => record.ref) ?? [];
722
737
  const removed = new Set(archivedRefs);
@@ -730,7 +745,7 @@ export async function runImprovePreparationStage(args) {
730
745
  const archived = appliedCleanup.archived.find((record) => record.ref === candidate.ref);
731
746
  if (!archived)
732
747
  continue;
733
- actions.push({
748
+ pruneActions.push({
734
749
  ref: candidate.ref,
735
750
  mode: "memory-prune",
736
751
  result: { ok: true, pruned: true, reason: candidate.reason },
@@ -738,31 +753,17 @@ export async function runImprovePreparationStage(args) {
738
753
  }
739
754
  if ((appliedCleanup.archived.length > 0 || appliedCleanup.beliefStateTransitions.length > 0) && primaryStashDir) {
740
755
  try {
741
- await reindexFn({ stashDir: primaryStashDir });
756
+ await reindexFn({ stashDir: primaryStashDir, signal: budgetSignal });
742
757
  }
743
758
  catch (err) {
744
- cleanupWarnings.push(`reindex after cleanup failed: ${err instanceof Error ? err.message : String(err)}`);
759
+ warnings.push(`reindex after cleanup failed: ${err instanceof Error ? err.message : String(err)}`);
745
760
  }
746
761
  }
747
762
  }
748
- const { validationFailures, validationFailureRefs, schemaRepairs } = await runValidationAndRepairPass({
749
- postCleanupRefs,
750
- options,
751
- startMs,
752
- budgetMs,
753
- primaryStashDir,
754
- });
755
- // Phase 0.5 — structural hygiene pass
756
- let lintSummary;
757
- if (primaryStashDir) {
758
- try {
759
- const lintResult = akmLint({ fix: true, dir: primaryStashDir });
760
- lintSummary = { fixed: lintResult.summary.fixed, flagged: lintResult.summary.flagged };
761
- }
762
- catch {
763
- // lint is best-effort; never block improve
764
- }
765
- }
763
+ return { appliedCleanup, postCleanupRefs, pruneActions, warnings };
764
+ }
765
+ /** Seed the per-originator rolling error windows from schema-repair errors. */
766
+ function seedRecentErrorWindows(schemaRepairs) {
766
767
  // O-5 / #378: Per-originator rolling error windows.
767
768
  // Reflexion (arXiv:2303.11366) warns that cross-task verbal critique
768
769
  // contamination degrades below single-shot baseline. Each originator key
@@ -785,6 +786,11 @@ export async function runImprovePreparationStage(args) {
785
786
  pushRecentError("schema-repair", errMsg);
786
787
  }
787
788
  }
789
+ return recentErrors;
790
+ }
791
+ /** Pass: snapshot-manifest — the three timestamp maps + the 30-day signal window. */
792
+ export function buildSnapshotManifest(args) {
793
+ const { postCleanupRefs, validationFailureRefs } = args;
788
794
  // ── Phase 2: signal-delta eligibility sets built EARLY ────────────────────
789
795
  // 0.8.0 replaces the flat time-based cooldowns (which produced synchronised
790
796
  // waves whenever many refs cooled at the same instant — see the 2026-05-26
@@ -801,69 +807,235 @@ export async function runImprovePreparationStage(args) {
801
807
  // The 30-day FEEDBACK_SIGNAL_WINDOW_DAYS bound still applies — only feedback
802
808
  // events newer than that count as "current signal". Ancient one-off
803
809
  // negatives don't permanently lock a ref into every run.
804
- //
805
- // High-retrieval refs (P0-A path) use a simpler "eligible once" rule: a
806
- // ref with no feedback signal but retrievalCount ≥ threshold is eligible
807
- // exactly once (no prior reflect proposal). Subsequent re-eligibility for
808
- // those refs requires either a new feedback event (then the normal
809
- // signal-delta gate applies) or human action. Documented limitation: this
810
- // path does not re-fire on retrieval-count growth alone in 0.8.0; storing
811
- // the retrieval count in proposal metadata for proper delta-tracking is
812
- // captured as future work.
813
810
  const FEEDBACK_SIGNAL_WINDOW_DAYS = 30;
814
811
  const feedbackSinceCutoff = new Date(Date.now() - daysToMs(FEEDBACK_SIGNAL_WINDOW_DAYS)).toISOString();
815
812
  // Build the three timestamp maps once across the entire postCleanupRefs set.
816
813
  // Per-ref queries would be N+1 and the planner is already the hottest path
817
814
  // in `akm improve`.
818
815
  const candidateRefs = postCleanupRefs.filter((r) => !validationFailureRefs.has(r.ref)).map((r) => r.ref);
819
- const latestFeedbackTs = buildLatestFeedbackTsMap(candidateRefs, feedbackSinceCutoff);
820
- const lastReflectProposalTs = buildLatestProposalTsMap(candidateRefs, "reflect");
821
- const lastDistillProposalTs = buildLatestProposalTsMap(candidateRefs, "distill");
822
- // Refs the distill signal-delta gate rejected at planning time. The main
823
- // loop reads this to skip distill for these refs without re-checking
824
- // eligibility per iteration.
825
- const distillCooledRefs = new Set();
826
- const preCooldownCount = postCleanupRefs.length;
827
- // ── Phase 3: partition postCleanupRefs by signal-delta eligibility ────────
828
- // Three buckets (validation failures are excluded entirely):
829
- // eligibleRefs — reflect signal-delta passes (full reflect+distill
830
- // loop path; distill guard remains in the loop for
831
- // refs that fail the distill signal-delta gate).
832
- // distillOnlyRefs — reflect blocked but distill signal-delta passes
833
- // AND ref is a distill candidate.
834
- // noFeedbackPool — neither signal-delta gate passes *and* the ref has
835
- // no recent feedback signal at all. These are NOT
836
- // skipped here: they are handed to the high-retrieval
837
- // fallback (P0-A) below so frequently-retrieved but
838
- // never-rated assets can still be improved. Only refs
839
- // that P0-A declines are ultimately fully skipped.
840
- // fullySkippedCount — has stale feedback but no signal delta → genuine
841
- // skip (counted, aggregated event emitted post-loop),
842
- // excluded from sort.
843
- const eligibleRefs = [];
844
- const distillOnlyRefs = [];
845
- // Zero-(recent-)feedback refs deferred to the P0-A high-retrieval fallback.
846
- const noFeedbackPool = [];
847
- let fullySkippedCount = 0;
848
- // O-2 (#365): explicit --scope <ref> bypasses every gate (user intent wins).
849
- const scopeRefBypass = scope.mode === "ref";
850
- for (const r of postCleanupRefs) {
851
- if (validationFailureRefs.has(r.ref))
816
+ // Carry each candidate's item_ref into the feedback/proposal timestamp reads.
817
+ const itemRefByRef = buildItemRefByRef(postCleanupRefs);
818
+ const latestFeedbackTs = buildLatestFeedbackTsMap(candidateRefs, feedbackSinceCutoff, itemRefByRef);
819
+ const lastReflectProposalTs = buildLatestProposalTsMap(candidateRefs, "reflect", itemRefByRef);
820
+ const lastDistillProposalTs = buildLatestProposalTsMap(candidateRefs, "distill", itemRefByRef);
821
+ return { feedbackSinceCutoff, latestFeedbackTs, lastReflectProposalTs, lastDistillProposalTs };
822
+ }
823
+ /**
824
+ * Pass: candidate-gather — the signal-delta partition, the bulk feedback
825
+ * summary, retrieval signals, the Layer-2 proactive and Layer-3 high-salience
826
+ * rescue lanes, the merged candidate set, and lane attribution stamping.
827
+ */
828
+ function gatherCandidates(args) {
829
+ const { scope, options, primaryStashDir, eventsCtx, improveProfile, resolvedPlan, postCleanupRefs } = args;
830
+ const { feedbackSinceCutoff, lastReflectProposalTs, lastDistillProposalTs } = args.snapshot;
831
+ const partition = partitionBySignalDelta({
832
+ scope,
833
+ options,
834
+ eventsCtx,
835
+ postCleanupRefs,
836
+ validationFailureRefs: args.validationFailureRefs,
837
+ snapshot: args.snapshot,
838
+ });
839
+ const actions = [...partition.actions];
840
+ const { distillCooledRefs, preCooldownCount, eligibleRefs, distillOnlyRefs, noFeedbackPool, fullySkippedCount } = partition;
841
+ // ── Phase 4: signal/feedback/utility/sort on the reduced set ──────────────
842
+ // Everything from here works on (eligibleRefs ∪ distillOnlyRefs) plus the
843
+ // deferred noFeedbackPool that may be rescued by the proactive-maintenance
844
+ // (Layer 2) or high-salience (Layer 3) fallbacks below. The fully-skipped
845
+ // bucket has already been routed and its aggregated event emitted; we
846
+ // deliberately avoid spending DB/CPU on refs that the signal-delta gate
847
+ // rejected with feedback already on record.
848
+ const processableRefs = [...eligibleRefs, ...distillOnlyRefs];
849
+ const feedbackSummary = buildFeedbackSummaryMap({
850
+ processableRefs,
851
+ noFeedbackPool,
852
+ eventsCtx,
853
+ feedbackSinceCutoff,
854
+ });
855
+ const signalFiltered = processableRefs.filter((candidate) => feedbackSummary.get(candidate.ref)?.hasSignal === true);
856
+ const signalBearingSet = new Set(signalFiltered.map((r) => r.ref));
857
+ // Zero-feedback candidates for the proactive/high-salience fallbacks:
858
+ // processableRefs without a recent signal, plus the deferred noFeedbackPool.
859
+ // Dedupe by ref (the two sources are disjoint by construction, but guard
860
+ // against overlap defensively).
861
+ const noFeedbackSeen = new Set();
862
+ const noFeedbackCandidates = [];
863
+ for (const r of [...processableRefs.filter((r) => !signalBearingSet.has(r.ref)), ...noFeedbackPool]) {
864
+ if (noFeedbackSeen.has(r.ref))
852
865
  continue;
853
- if (scopeRefBypass) {
854
- eligibleRefs.push(r);
866
+ noFeedbackSeen.add(r.ref);
867
+ noFeedbackCandidates.push(r);
868
+ }
869
+ const { retrievalCounts, lastUseMsForProactive } = fetchRetrievalSignals({
870
+ options,
871
+ primaryStashDir,
872
+ signalFiltered,
873
+ noFeedbackCandidates,
874
+ });
875
+ const proactive = selectProactiveMaintenanceLane({
876
+ scope,
877
+ improveProfile,
878
+ resolvedPlan,
879
+ eventsCtx,
880
+ noFeedbackCandidates,
881
+ lastReflectProposalTs,
882
+ lastDistillProposalTs,
883
+ retrievalCounts,
884
+ lastUseMsForProactive,
885
+ });
886
+ const proactiveRefs = proactive.proactiveRefs;
887
+ const proactiveMaintenanceSummary = proactive.proactiveMaintenanceSummary;
888
+ const highSalienceRefs = selectHighSalienceLane({
889
+ options,
890
+ improveProfile,
891
+ eventsCtx,
892
+ noFeedbackCandidates,
893
+ proactiveRefs,
894
+ lastReflectProposalTs,
895
+ });
896
+ // Record an in-memory skip action for every zero-feedback ref that the
897
+ // partition loop deferred to the proactive/high-salience fallbacks but those
898
+ // lanes then declined (not due, below threshold, or a prior reflect proposal
899
+ // already on record). These never make it into mergedRefs, so without this
900
+ // they would silently vanish from the run summary. No DB event is written
901
+ // here — these refs carry no signal at all, so there is nothing for the skip
902
+ // histogram to aggregate; the action log alone preserves the per-ref audit
903
+ // trail (mirrors the fully-skipped action above).
904
+ const rescuedSet = new Set([...proactiveRefs, ...highSalienceRefs].map((r) => r.ref));
905
+ for (const r of noFeedbackPool) {
906
+ if (rescuedSet.has(r.ref))
855
907
  continue;
856
- }
857
- const reflectOk = isSignalDeltaEligible(r.ref, latestFeedbackTs, lastReflectProposalTs);
858
- const distillOk = isSignalDeltaEligible(r.ref, latestFeedbackTs, lastDistillProposalTs);
859
- const isDistillCandidate = isDistillCandidateRef(r.ref, options.stashDir);
860
- if (reflectOk) {
861
- if (!distillOk && isDistillCandidate) {
862
- // Reflect passes the gate, distill does not — emit the synthetic
863
- // distill-skipped action and event up-front so the in-loop guard
864
- // does not have to re-derive eligibility.
865
- distillCooledRefs.add(r.ref);
866
- actions.push({ ref: r.ref, mode: "distill-skipped", result: { ok: true, reason: "distill signal-delta" } });
908
+ actions.push({
909
+ ref: r.ref,
910
+ mode: "distill-skipped",
911
+ result: { ok: true, reason: "no new signal since last proposal" },
912
+ });
913
+ }
914
+ // If the user explicitly scoped to a single ref, always act on it —
915
+ // skip the signal/retrieval filter entirely. The filter exists to avoid
916
+ // noisy "improve everything" runs; it should not gate an intentional
917
+ // per-ref invocation where the user's explicit choice is the signal.
918
+ //
919
+ // For type/all scope: only process refs with usage signals (recent feedback
920
+ // or a proactive/high-salience rescue). A stash with no signals has 0
921
+ // eligible refs — usage is the gate. Run `akm feedback <ref> --positive` or
922
+ // retrieve assets to bring them into the eligible pool.
923
+ // Layer-2 proactive refs join the eligible set alongside feedback-signal
924
+ // refs. The three sources are disjoint by construction (proactive draws from
925
+ // noFeedbackCandidates, and high-salience draws from the remainder), but
926
+ // dedupe defensively so a ref can never enter the loop twice.
927
+ // `requireFeedbackSignal` still suppresses all fallback sources for callers
928
+ // that want feedback-only runs.
929
+ const signalAndRetrievalRefs = dedupeRefs([...signalFiltered, ...proactiveRefs, ...highSalienceRefs]);
930
+ const mergedRefs = scope.mode === "ref" ? processableRefs : options.requireFeedbackSignal ? signalFiltered : signalAndRetrievalRefs;
931
+ // ── Attribution tagging: stamp each ref with the eligibility lane that
932
+ // selected it ──────────────────────────────────────────────────────────────
933
+ // Every reflect/distill proposal must record WHICH lane chose its source asset
934
+ // so downstream accept/reject/revert/retrieval outcomes can be sliced by lane
935
+ // (does the PROACTIVE lane produce value vs the reactive lanes?). We build the
936
+ // lane map here — the one place all three lanes are known — and stamp it onto
937
+ // each ImproveEligibleRef object. Because the ref objects are shared by
938
+ // reference across buckets, the stamp travels with the ref through the sort,
939
+ // disk-check, and loop stages down to the reflect/distill event emit sites and
940
+ // createProposal calls. See EligibilitySource for the lane vocabulary.
941
+ //
942
+ // Precedence (prefer the most specific reactive signal):
943
+ // scope > signal-delta > proactive > high-salience
944
+ // A ref with real feedback is attributed to feedback even if it was also due
945
+ // for proactive maintenance or had high encoding salience. We apply lanes
946
+ // weakest-first so the strongest overwrites; the explicit --scope <ref> bypass
947
+ // wins outright (user intent).
948
+ const eligibilitySourceByRef = new Map();
949
+ for (const r of highSalienceRefs)
950
+ eligibilitySourceByRef.set(r.ref, "high-salience");
951
+ for (const r of proactiveRefs)
952
+ eligibilitySourceByRef.set(r.ref, "proactive");
953
+ for (const r of signalFiltered)
954
+ eligibilitySourceByRef.set(r.ref, "signal-delta");
955
+ if (scope.mode === "ref") {
956
+ // O-2 (#365): explicit --scope <ref> bypass — every ref in processableRefs
957
+ // arrived via the scopeRefBypass branch, so attribute the whole set to scope.
958
+ for (const r of processableRefs)
959
+ eligibilitySourceByRef.set(r.ref, "scope");
960
+ }
961
+ for (const r of mergedRefs) {
962
+ // "unknown" is a genuine fallback, never a silent alias for signal-delta:
963
+ // only refs we truly cannot attribute land here (none in practice, since
964
+ // mergedRefs is always a subset of the four lanes above).
965
+ r.eligibilitySource = eligibilitySourceByRef.get(r.ref) ?? "unknown";
966
+ }
967
+ return {
968
+ actions,
969
+ distillCooledRefs,
970
+ preCooldownCount,
971
+ distillOnlyRefs,
972
+ fullySkippedCount,
973
+ feedbackSummary,
974
+ signalFiltered,
975
+ signalBearingSet,
976
+ retrievalCounts,
977
+ proactiveRefs,
978
+ proactiveMaintenanceSummary,
979
+ highSalienceRefs,
980
+ signalAndRetrievalRefs,
981
+ mergedRefs,
982
+ eligibilitySourceByRef,
983
+ };
984
+ }
985
+ /**
986
+ * The signal-delta partition of postCleanupRefs into the four buckets (pass:
987
+ * candidate-gather, phase 3). The 2026-05-26 54-ref incident semantics move
988
+ * VERBATIM — see the phase-2/3 comments inside.
989
+ */
990
+ export function partitionBySignalDelta(args) {
991
+ const { scope, options, eventsCtx, postCleanupRefs, validationFailureRefs } = args;
992
+ const { latestFeedbackTs, lastReflectProposalTs, lastDistillProposalTs } = args.snapshot;
993
+ const actions = [];
994
+ // Refs the distill signal-delta gate rejected at planning time. The main
995
+ // loop reads this to skip distill for these refs without re-checking
996
+ // eligibility per iteration.
997
+ const distillCooledRefs = new Set();
998
+ const preCooldownCount = postCleanupRefs.length;
999
+ // ── Phase 3: partition postCleanupRefs by signal-delta eligibility ────────
1000
+ // Three buckets (validation failures are excluded entirely):
1001
+ // eligibleRefs — reflect signal-delta passes (full reflect+distill
1002
+ // loop path; distill guard remains in the loop for
1003
+ // refs that fail the distill signal-delta gate).
1004
+ // distillOnlyRefs — reflect blocked but distill signal-delta passes
1005
+ // AND ref is a distill candidate.
1006
+ // noFeedbackPool — neither signal-delta gate passes *and* the ref has
1007
+ // no recent feedback signal at all. These are NOT
1008
+ // skipped here: they are handed to the proactive
1009
+ // (Layer 2) and high-salience (Layer 3) fallbacks
1010
+ // below so never-rated assets can still be improved.
1011
+ // Only refs those lanes decline are fully skipped.
1012
+ // fullySkippedCount — has stale feedback but no signal delta → genuine
1013
+ // skip (counted, aggregated event emitted post-loop),
1014
+ // excluded from sort.
1015
+ const eligibleRefs = [];
1016
+ const distillOnlyRefs = [];
1017
+ // Zero-(recent-)feedback refs deferred to the proactive/high-salience fallbacks.
1018
+ const noFeedbackPool = [];
1019
+ let fullySkippedCount = 0;
1020
+ // O-2 (#365): explicit --scope <ref> bypasses every gate (user intent wins).
1021
+ const scopeRefBypass = scope.mode === "ref";
1022
+ for (const r of postCleanupRefs) {
1023
+ if (validationFailureRefs.has(r.ref))
1024
+ continue;
1025
+ if (scopeRefBypass) {
1026
+ eligibleRefs.push(r);
1027
+ continue;
1028
+ }
1029
+ const reflectOk = isSignalDeltaEligible(r.ref, latestFeedbackTs, lastReflectProposalTs);
1030
+ const distillOk = isSignalDeltaEligible(r.ref, latestFeedbackTs, lastDistillProposalTs);
1031
+ const isDistillCandidate = isDistillCandidateRef(r.ref, options.stashDir);
1032
+ if (reflectOk) {
1033
+ if (!distillOk && isDistillCandidate) {
1034
+ // Reflect passes the gate, distill does not — emit the synthetic
1035
+ // distill-skipped action and event up-front so the in-loop guard
1036
+ // does not have to re-derive eligibility.
1037
+ distillCooledRefs.add(r.ref);
1038
+ actions.push({ ref: r.ref, mode: "distill-skipped", result: { ok: true, reason: "distill signal-delta" } });
867
1039
  appendEvent({
868
1040
  eventType: "improve_skipped",
869
1041
  ref: r.ref,
@@ -883,16 +1055,16 @@ export async function runImprovePreparationStage(args) {
883
1055
  }
884
1056
  else if (!latestFeedbackTs.has(r.ref)) {
885
1057
  // Neither signal-delta gate passes AND there is no recent feedback signal
886
- // at all. Rather than skip outright, defer to the high-retrieval fallback
887
- // (P0-A) below: a never-rated-but-frequently-retrieved asset is exactly
888
- // what that path is meant to rescue. Refs P0-A declines are skipped there.
1058
+ // at all. Rather than skip outright, defer to the proactive-maintenance
1059
+ // and high-salience fallbacks below: a never-rated asset is exactly what
1060
+ // those lanes are meant to rescue. Refs those lanes decline are skipped there.
889
1061
  noFeedbackPool.push(r);
890
1062
  }
891
1063
  else {
892
1064
  // Has feedback on record but no signal delta since the last proposal —
893
1065
  // genuinely fully skipped. Counted here; a single aggregated
894
1066
  // improve_skipped event is emitted after the loop (mirrors
895
- // profile_filtered_all_passes) instead of one event per ref.
1067
+ // strategy_filtered_all_passes) instead of one event per ref.
896
1068
  fullySkippedCount++;
897
1069
  actions.push({
898
1070
  ref: r.ref,
@@ -903,7 +1075,7 @@ export async function runImprovePreparationStage(args) {
903
1075
  }
904
1076
  // Emit ONE aggregated skip event for the fully-skipped bucket rather than one
905
1077
  // improve_skipped event per ref (#592 pattern, mirrors
906
- // profile_filtered_all_passes above). The per-ref loop previously produced
1078
+ // strategy_filtered_all_passes above). The per-ref loop previously produced
907
1079
  // ~11K state.db writes per run on a large stash, the dominant contributor to
908
1080
  // 900 s timeouts. The in-memory `actions` log keeps the per-ref detail for the
909
1081
  // run summary; no downstream consumer needs a per-ref DB audit trail (health's
@@ -918,26 +1090,23 @@ export async function runImprovePreparationStage(args) {
918
1090
  },
919
1091
  }, eventsCtx);
920
1092
  }
921
- // ── Phase 4: signal/feedback/utility/sort on the reduced set ──────────────
922
- // Everything from here works on (eligibleRefs ∪ distillOnlyRefs) plus the
923
- // deferred noFeedbackPool that may be rescued by the high-retrieval fallback
924
- // (P0-A). The fully-skipped bucket has already been routed and its aggregated
925
- // event emitted; we deliberately avoid spending DB/CPU on refs that the
926
- // signal-delta gate rejected with feedback already on record.
927
- const processableRefs = [...eligibleRefs, ...distillOnlyRefs];
928
- // Refs eligible for the high-retrieval fallback (P0-A): the signal-delta
929
- // partition above could not place these in a reflect/distill bucket, but they
930
- // may still qualify if they have been retrieved often enough. Two disjoint
931
- // sources feed this set:
932
- // 1. noFeedbackPool — refs with no recent feedback that the partition loop
933
- // deliberately deferred here (otherwise they would never reach P0-A).
934
- // 2. processableRefs entries that turn out to carry no recent feedback
935
- // *signal* once feedbackSummary is computed below.
936
- // (1) is added here; (2) is folded in after feedbackSummary is built.
1093
+ return {
1094
+ actions,
1095
+ distillCooledRefs,
1096
+ preCooldownCount,
1097
+ eligibleRefs,
1098
+ distillOnlyRefs,
1099
+ noFeedbackPool,
1100
+ fullySkippedCount,
1101
+ };
1102
+ }
1103
+ /** Bulk per-ref feedback summary in a SINGLE readEvents pass (candidate-gather). */
1104
+ function buildFeedbackSummaryMap(args) {
1105
+ const { processableRefs, noFeedbackPool, eventsCtx, feedbackSinceCutoff } = args;
937
1106
  // Gap 6: only surface feedback signals from the last 30 days so that
938
1107
  // ancient one-off feedback events don't permanently lock an asset into
939
1108
  // every improve run. Assets with only stale signals fall through to the
940
- // high-retrieval path (P0-A) or are skipped until new signals arrive.
1109
+ // proactive/high-salience fallbacks or are skipped until new signals arrive.
941
1110
  // (FEEDBACK_SIGNAL_WINDOW_DAYS / feedbackSinceCutoff are already defined in
942
1111
  // Phase 2 above for the signal-delta gate; we reuse them here.)
943
1112
  // Pre-compute feedback summary per ref in a SINGLE bulk read so we don't
@@ -946,22 +1115,25 @@ export async function runImprovePreparationStage(args) {
946
1115
  // above: one readEvents() call fetches ALL feedback events, then we aggregate
947
1116
  // in-memory by ref — O(1) DB opens regardless of candidate set size.
948
1117
  // Cover processableRefs *and* the deferred noFeedbackPool so utility/feedback
949
- // ratios are available for any noFeedbackPool ref that P0-A rescues below.
1118
+ // ratios are available for any noFeedbackPool ref the fallback lanes rescue below.
950
1119
  //
951
1120
  // Behavioral note: positive/negative COUNTS are all-time (same as the old
952
1121
  // per-ref readEvents call which had no `since` filter); hasSignal is bounded
953
1122
  // to feedbackSinceCutoff (same as the old inline `(e.ts ?? "") >= cutoff` guard).
954
1123
  const feedbackSummary = new Map();
955
1124
  {
956
- const feedbackCandidateSet = new Set([...processableRefs, ...noFeedbackPool].map((r) => r.ref));
1125
+ const feedbackCandidates = [...processableRefs, ...noFeedbackPool];
1126
+ const feedbackCandidateSet = new Set(feedbackCandidates.map((r) => r.ref));
1127
+ // Map each candidate's single durable event key back to its display ref.
1128
+ const feedbackRefByDurableKey = new Map(feedbackCandidates.flatMap((r) => improveStateReadRefs(r.ref, r.itemRef).map((key) => [key, r.ref])));
957
1129
  if (feedbackCandidateSet.size > 0) {
958
1130
  // Fetch ALL feedback events in one query (no ref filter, no since filter =
959
1131
  // single full table scan). Filtering per-ref in memory avoids N sequential
960
1132
  // state.db opens — the dominant FD-leak path on large stashes.
961
1133
  const { events: allFeedbackEvents } = readEvents({ type: "feedback" }, eventsCtx);
962
1134
  for (const e of allFeedbackEvents) {
963
- const ref = e.ref;
964
- if (!ref || !feedbackCandidateSet.has(ref))
1135
+ const ref = e.ref ? feedbackRefByDurableKey.get(e.ref) : undefined;
1136
+ if (!ref)
965
1137
  continue;
966
1138
  const entry = feedbackSummary.get(ref) ?? { hasSignal: false, positive: 0, negative: 0 };
967
1139
  const meta = e.metadata;
@@ -987,22 +1159,11 @@ export async function runImprovePreparationStage(args) {
987
1159
  }
988
1160
  }
989
1161
  }
990
- const signalFiltered = processableRefs.filter((candidate) => feedbackSummary.get(candidate.ref)?.hasSignal === true);
991
- // P0-A: also surface zero-feedback assets that have been retrieved many times.
992
- const RETRIEVAL_COUNT_THRESHOLD = options.minRetrievalCount ?? 5;
993
- const signalBearingSet = new Set(signalFiltered.map((r) => r.ref));
994
- // Zero-feedback candidates for P0-A: processableRefs without a recent signal,
995
- // plus the deferred noFeedbackPool. Dedupe by ref (the two sources are
996
- // disjoint by construction, but guard against overlap defensively).
997
- const noFeedbackSeen = new Set();
998
- const noFeedbackCandidates = [];
999
- for (const r of [...processableRefs.filter((r) => !signalBearingSet.has(r.ref)), ...noFeedbackPool]) {
1000
- if (noFeedbackSeen.has(r.ref))
1001
- continue;
1002
- noFeedbackSeen.add(r.ref);
1003
- noFeedbackCandidates.push(r);
1004
- }
1005
- let highRetrievalRefs = [];
1162
+ return feedbackSummary;
1163
+ }
1164
+ /** Retrieval counts + last-use timestamps for the candidate pools (candidate-gather). */
1165
+ function fetchRetrievalSignals(args) {
1166
+ const { options, primaryStashDir, signalFiltered, noFeedbackCandidates } = args;
1006
1167
  // Retrieval counts for the zero-feedback pool, hoisted so the Layer-2
1007
1168
  // proactive-maintenance selector below can reuse them without a second DB pass.
1008
1169
  // Also fetch lastUseMs here for the proactive-maintenance recency term (plan §WS-1
@@ -1012,67 +1173,64 @@ export async function runImprovePreparationStage(args) {
1012
1173
  let dbForRetrieval;
1013
1174
  try {
1014
1175
  dbForRetrieval = openExistingDatabase();
1015
- const showEventCount = countUsageEventsByType(dbForRetrieval, "show");
1016
- if (showEventCount === 0) {
1017
- warn("Warning: show events not yet in usage_events — zero-feedback fallback will match only search-retrieved assets.");
1018
- }
1019
- // Fetch retrieval counts for ALL candidates — not only the zero-feedback pool.
1020
- // Previously only noFeedbackCandidates were looked up, so feedback-bearing refs
1021
- // had retrievalFreq=0 in computeSalience(), collapsing their retrievalSalience
1022
- // to 0 regardless of actual use. Two assets of the same type — one
1023
- // heavily-retrieved, one never-touched — would receive identical rankScores.
1024
- // Fix (WS-1 blocker 3): union the feedback pool into the lookup.
1025
- const allCandidateRefs = [...new Set([...signalFiltered, ...noFeedbackCandidates].map((r) => r.ref))];
1026
- retrievalCounts = getRetrievalCounts(dbForRetrieval, allCandidateRefs);
1027
- lastUseMsForProactive = getLastUseMsByRef(dbForRetrieval, noFeedbackCandidates.map((r) => r.ref));
1028
- // High-retrieval signal-delta (simplified rule, 0.8.0): a no-feedback
1029
- // ref qualifies exactly once — when it has actually been retrieved
1030
- // (retrievalCount ≥ 1) AND retrievalCount ≥ threshold AND no prior reflect
1031
- // proposal exists for it. Once a reflect proposal is on record, subsequent
1032
- // re-eligibility requires explicit feedback (which flows through the normal
1033
- // signal-delta gate above). The explicit `> 0` guard keeps a threshold of 0
1034
- // from rescuing genuinely never-retrieved assets — the fallback is for
1035
- // *retrieved* assets, not silent ones. Tracking growth in retrieval count
1036
- // would require persisting the count in proposal metadata; deferred to a
1037
- // follow-up.
1038
- highRetrievalRefs = noFeedbackCandidates.filter((r) => {
1039
- const count = retrievalCounts.get(r.ref) ?? 0;
1040
- return count > 0 && count >= RETRIEVAL_COUNT_THRESHOLD && !lastReflectProposalTs.has(r.ref);
1176
+ // usage_events lives in state.db (Chunk-8 WI-8.3); entries stay in index.db,
1177
+ // so the retrieval-count reads take both handles.
1178
+ const dbForRetrievalIndex = dbForRetrieval;
1179
+ withStateDb((stateDb) => {
1180
+ const showEventCount = countUsageEventsByType(stateDb, "show");
1181
+ if (showEventCount === 0) {
1182
+ warn("Warning: show events not yet in usage_events — zero-feedback fallback will match only search-retrieved assets.");
1183
+ }
1184
+ // Fetch retrieval counts for ALL candidates — not only the zero-feedback pool.
1185
+ // Previously only noFeedbackCandidates were looked up, so feedback-bearing refs
1186
+ // had retrievalFreq=0 in computeSalience(), collapsing their retrievalSalience
1187
+ // to 0 regardless of actual use. Two assets of the same type — one
1188
+ // heavily-retrieved, one never-touched — would receive identical rankScores.
1189
+ // Fix (WS-1 blocker 3): union the feedback pool into the lookup.
1190
+ const allCandidateRefs = [...new Set([...signalFiltered, ...noFeedbackCandidates].map((r) => r.ref))];
1191
+ retrievalCounts = getRetrievalCounts(dbForRetrievalIndex, stateDb, allCandidateRefs, {
1192
+ sourceName: options.sourceName,
1193
+ stashDir: primaryStashDir,
1194
+ });
1041
1195
  });
1196
+ lastUseMsForProactive = getLastUseMsByRef(dbForRetrieval, noFeedbackCandidates.map((r) => r.ref), primaryStashDir);
1042
1197
  }
1043
1198
  catch (err) {
1044
1199
  rethrowIfTestIsolationError(err);
1045
- // best-effort: if DB unavailable, highRetrievalRefs stays empty
1200
+ // best-effort: if DB unavailable, retrievalCounts/lastUseMsForProactive stay empty
1046
1201
  }
1047
1202
  finally {
1048
1203
  if (dbForRetrieval)
1049
1204
  closeDatabase(dbForRetrieval);
1050
1205
  }
1051
- // ── Layer 2: PROACTIVE MAINTENANCE SELECTOR (third eligibility source) ─────
1052
- // The signal-delta gate and P0-A only surface assets with fresh feedback or a
1053
- // raw-retrieval spike. Neither revisits a stable, high-value asset on a
1054
- // schedule, so on a quiet stash useful assets drift stale and are never
1055
- // refreshed. When the `proactiveMaintenance` process is enabled (DEFAULT OFF)
1206
+ return { retrievalCounts, lastUseMsForProactive };
1207
+ }
1208
+ /** Layer 2 — the proactive-maintenance selector lane (candidate-gather). */
1209
+ function selectProactiveMaintenanceLane(args) {
1210
+ const { scope, improveProfile, resolvedPlan, eventsCtx, noFeedbackCandidates, lastReflectProposalTs, lastDistillProposalTs, retrievalCounts, lastUseMsForProactive, } = args;
1211
+ // ── Layer 2: PROACTIVE MAINTENANCE SELECTOR (second eligibility source) ────
1212
+ // The signal-delta gate only surfaces assets with fresh feedback. It never
1213
+ // revisits a stable, high-value asset on a schedule, so on a quiet stash
1214
+ // useful assets drift stale and are never refreshed. When the
1215
+ // `proactiveMaintenance` process is enabled (DEFAULT OFF)
1056
1216
  // and the run is whole-stash / type scope, this selector ranks the eligible
1057
1217
  // population by a composite maintenance priority, gates on staleness ("due"),
1058
1218
  // bounds to top-N, and folds the winners into the SAME candidate set the other
1059
- // two sources feed — so they flow through the existing #580 empty-diff /
1219
+ // sources feed — so they flow through the existing #580 empty-diff /
1060
1220
  // cosmetic suppression and additive-distill gates. It adds no new mutation
1061
1221
  // logic of its own. The due gate doubles as the rotation cooldown: a freshly
1062
1222
  // reflected asset is excluded until it ages back past `dueDays`, so successive
1063
1223
  // runs rotate through the due pool rather than re-selecting the same heads.
1064
1224
  let proactiveRefs = [];
1065
1225
  let proactiveMaintenanceSummary;
1066
- const proactiveEnabled = scope.mode !== "ref" && resolveProcessEnabled("proactiveMaintenance", improveProfile);
1226
+ const proactiveEnabled = scope.mode !== "ref" && resolvedPlan.processes.proactiveMaintenance.enabled;
1067
1227
  if (proactiveEnabled) {
1068
1228
  const pmCfg = improveProfile.processes?.proactiveMaintenance;
1069
1229
  const dueDays = pmCfg?.dueDays ?? DEFAULT_DUE_DAYS;
1070
1230
  const maxPerRun = pmCfg?.maxPerRun ?? pmCfg?.limit ?? DEFAULT_MAX_PER_RUN;
1071
1231
  // Candidate population: the zero-feedback / non-signal pool — exactly the
1072
- // assets the other two sources would NOT pick this run. Exclude any P0-A
1073
- // rescued this run so we never double-select the same ref.
1074
- const alreadySelected = new Set(highRetrievalRefs.map((r) => r.ref));
1075
- const pmCandidates = noFeedbackCandidates.filter((r) => !alreadySelected.has(r.ref));
1232
+ // assets the signal-delta gate would NOT pick this run.
1233
+ const pmCandidates = noFeedbackCandidates;
1076
1234
  const selection = selectProactiveMaintenanceRefs({
1077
1235
  candidates: pmCandidates,
1078
1236
  lastReflectTs: lastReflectProposalTs,
@@ -1116,6 +1274,11 @@ export async function runImprovePreparationStage(args) {
1116
1274
  `(${selection.neverReflected} never reflected, dueDays=${dueDays}, maxPerRun=${maxPerRun})`);
1117
1275
  }
1118
1276
  }
1277
+ return { proactiveRefs, proactiveMaintenanceSummary };
1278
+ }
1279
+ /** Layer 3 — the high-salience admission gate (#608/#644; candidate-gather). */
1280
+ function selectHighSalienceLane(args) {
1281
+ const { options, improveProfile, eventsCtx, noFeedbackCandidates, proactiveRefs, lastReflectProposalTs } = args;
1119
1282
  // ── Layer 3: HIGH-SALIENCE ADMISSION GATE (#608) ──────────────────────────
1120
1283
  // Zero-feedback refs whose encoding_salience (set at distill time by
1121
1284
  // scoreEncodingSalience) exceeds the configured salienceThreshold are admitted
@@ -1128,11 +1291,11 @@ export async function runImprovePreparationStage(args) {
1128
1291
  //
1129
1292
  // Cooldown: a ref qualifies at most once — when no prior reflect proposal
1130
1293
  // exists for it (`!lastReflectProposalTs.has`). Without this guard the lane
1131
- // re-selects the same high-salience refs on EVERY run (auto-accept emits a
1294
+ // re-selects the same high-salience refs on EVERY run (promotion emits a
1132
1295
  // `promoted` event, not `feedback`, so the ref never leaves
1133
1296
  // noFeedbackCandidates), burning LLM calls and churning the asset. This
1134
- // mirrors the P0-A high-retrieval gate's `!lastReflectProposalTs.has(r.ref)`
1135
- // guard above so all "rescue" lanes share the same once-per-asset semantics.
1297
+ // mirrors the same `!lastReflectProposalTs.has(r.ref)` once-per-asset
1298
+ // semantics the other "rescue" lanes share.
1136
1299
  //
1137
1300
  // Content-provenance gate (#644 follow-up): the row must ALSO carry a genuine
1138
1301
  // content-derived encoding score (`isContentEncodingRow`). Otherwise the lane
@@ -1144,11 +1307,11 @@ export async function runImprovePreparationStage(args) {
1144
1307
  // earn retrieval/feedback signal via the other lanes. This PRESERVES #608's
1145
1308
  // intent — distilled assets (the lane's real targets) keep their real content
1146
1309
  // score and still qualify — while cutting the type-stub waste. See §5 F1 of
1147
- // docs/design/improve-salience-working-reference.md and #608/#644.
1310
+ // #608/#644.
1148
1311
  const highSalienceRefs = [];
1149
1312
  const salienceCfg = (options.config ?? loadConfig()).improve?.salience;
1150
1313
  const salienceThreshold = salienceCfg?.salienceThreshold ?? 0.75;
1151
- const proactiveAndRetrievalSet = new Set([...highRetrievalRefs, ...proactiveRefs].map((r) => r.ref));
1314
+ const proactiveSelectedSet = new Set(proactiveRefs.map((r) => r.ref));
1152
1315
  try {
1153
1316
  withStateDb((dbForHighSalience) => {
1154
1317
  // Derive the cap from the resolved reflect limit (mirrors improve.ts's
@@ -1156,15 +1319,15 @@ export async function runImprovePreparationStage(args) {
1156
1319
  // collapse the lane to exactly 1 ref via the bare `?? 10` fallback.
1157
1320
  const effectiveLimit = options.limit ?? improveProfile?.processes?.reflect?.limit ?? improveProfile.limit ?? 10;
1158
1321
  const highSalienceCap = Math.max(1, Math.floor(effectiveLimit * 0.1));
1159
- const candidates = noFeedbackCandidates.filter((r) => !proactiveAndRetrievalSet.has(r.ref));
1322
+ const candidates = noFeedbackCandidates.filter((r) => !proactiveSelectedSet.has(r.ref));
1160
1323
  // Collect ALL qualifying candidates, then take the top-N BY SCORE — the
1161
1324
  // previous first-N-in-scan-order break meant a higher-salience candidate
1162
1325
  // found later in the scan lost its slot to an earlier lower-scoring one.
1163
1326
  const qualifying = [];
1164
1327
  for (const r of candidates) {
1165
- const row = getAssetSalience(dbForHighSalience, r.ref);
1328
+ const row = readAssetSalienceForImproveRef(dbForHighSalience, r.ref, r.itemRef);
1166
1329
  if (row &&
1167
- isContentEncodingRow(row, parseAssetRef(r.ref).type) &&
1330
+ isContentEncodingRow(row) &&
1168
1331
  row.encoding_salience >= salienceThreshold &&
1169
1332
  !lastReflectProposalTs.has(r.ref)) {
1170
1333
  qualifying.push({ ref: r, score: row.encoding_salience });
@@ -1184,108 +1347,39 @@ export async function runImprovePreparationStage(args) {
1184
1347
  info(`[improve] high-salience lane admitted ${highSalienceRefs.length} content-scored ref(s) ` +
1185
1348
  `(threshold=${salienceThreshold}, requires content-derived encoding_source)`);
1186
1349
  }
1187
- // Record an in-memory skip action for every zero-feedback ref that the
1188
- // partition loop deferred to P0-A but P0-A then declined (retrievalCount below
1189
- // threshold, or a prior reflect proposal already on record). These never make
1190
- // it into mergedRefs, so without this they would silently vanish from the run
1191
- // summary. No DB event is written here — these refs carry no signal at all, so
1192
- // there is nothing for the skip histogram to aggregate; the action log alone
1193
- // preserves the per-ref audit trail (mirrors the fully-skipped action above).
1194
- const rescuedSet = new Set([...highRetrievalRefs, ...proactiveRefs, ...highSalienceRefs].map((r) => r.ref));
1195
- for (const r of noFeedbackPool) {
1196
- if (rescuedSet.has(r.ref))
1197
- continue;
1198
- actions.push({
1199
- ref: r.ref,
1200
- mode: "distill-skipped",
1201
- result: { ok: true, reason: "no new signal since last proposal" },
1202
- });
1203
- }
1204
- // If the user explicitly scoped to a single ref, always act on it —
1205
- // skip the signal/retrieval filter entirely. The filter exists to avoid
1206
- // noisy "improve everything" runs; it should not gate an intentional
1207
- // per-ref invocation where the user's explicit choice is the signal.
1208
- //
1209
- // For type/all scope: only process refs with usage signals (recent feedback
1210
- // or sufficient retrievals). A stash with no signals has 0 eligible refs —
1211
- // usage is the gate. Run `akm feedback <ref> --positive` or retrieve assets
1212
- // to bring them into the eligible pool.
1213
- // Layer-2 proactive refs join the eligible set alongside feedback-signal and
1214
- // high-retrieval (P0-A) refs. The four sources are disjoint by construction
1215
- // (proactive draws from noFeedbackCandidates with the P0-A picks removed, and
1216
- // high-salience draws from the remainder), but dedupe defensively so a ref can
1217
- // never enter the loop twice. `requireFeedbackSignal` still suppresses all
1218
- // fallback sources for callers that want feedback-only runs.
1219
- const signalAndRetrievalRefs = dedupeRefs([
1220
- ...signalFiltered,
1221
- ...highRetrievalRefs,
1222
- ...proactiveRefs,
1223
- ...highSalienceRefs,
1224
- ]);
1225
- let mergedRefs = scope.mode === "ref" ? processableRefs : options.requireFeedbackSignal ? signalFiltered : signalAndRetrievalRefs;
1226
- // ── Attribution tagging: stamp each ref with the eligibility lane that
1227
- // selected it ──────────────────────────────────────────────────────────────
1228
- // Every reflect/distill proposal must record WHICH lane chose its source asset
1229
- // so downstream accept/reject/revert/retrieval outcomes can be sliced by lane
1230
- // (does the PROACTIVE lane produce value vs the reactive lanes?). We build the
1231
- // lane map here — the one place all four lanes are known — and stamp it onto
1232
- // each ImproveEligibleRef object. Because the ref objects are shared by
1233
- // reference across buckets, the stamp travels with the ref through the sort,
1234
- // disk-check, and loop stages down to the reflect/distill event emit sites and
1235
- // createProposal calls. See EligibilitySource for the lane vocabulary.
1236
- //
1237
- // Precedence (prefer the most specific reactive signal):
1238
- // scope > signal-delta > high-retrieval > proactive > high-salience
1239
- // A ref with real feedback is attributed to feedback even if it was also due
1240
- // for proactive maintenance or had high encoding salience. We apply lanes
1241
- // weakest-first so the strongest overwrites; the explicit --scope <ref> bypass
1242
- // wins outright (user intent).
1243
- const eligibilitySourceByRef = new Map();
1244
- for (const r of highSalienceRefs)
1245
- eligibilitySourceByRef.set(r.ref, "high-salience");
1246
- for (const r of proactiveRefs)
1247
- eligibilitySourceByRef.set(r.ref, "proactive");
1248
- for (const r of highRetrievalRefs)
1249
- eligibilitySourceByRef.set(r.ref, "high-retrieval");
1250
- for (const r of signalFiltered)
1251
- eligibilitySourceByRef.set(r.ref, "signal-delta");
1252
- if (scope.mode === "ref") {
1253
- // O-2 (#365): explicit --scope <ref> bypass — every ref in processableRefs
1254
- // arrived via the scopeRefBypass branch, so attribute the whole set to scope.
1255
- for (const r of processableRefs)
1256
- eligibilitySourceByRef.set(r.ref, "scope");
1257
- }
1258
- for (const r of mergedRefs) {
1259
- // "unknown" is a genuine fallback, never a silent alias for signal-delta:
1260
- // only refs we truly cannot attribute land here (none in practice, since
1261
- // mergedRefs is always a subset of the four lanes above).
1262
- r.eligibilitySource = eligibilitySourceByRef.get(r.ref) ?? "unknown";
1263
- }
1350
+ return highSalienceRefs;
1351
+ }
1352
+ /**
1353
+ * Pass: salience-score — the WS-2 outcome loop, the WS-1 salience vector
1354
+ * computation (#644 provenance preserved), persistence + rank-change report,
1355
+ * and the forgetting-safety injection. The valence-score call sites
1356
+ * (computeValenceScore) live VERBATIM inside the outcome/persist sub-passes.
1357
+ * Mutates the shared eligibilitySourceByRef map and ref objects in place —
1358
+ * attribution identity is load-bearing (see the candidate-gather comments).
1359
+ */
1360
+ function scoreSalience(args) {
1361
+ const { scope, options, primaryStashDir, eventsCtx, eligibilitySourceByRef, feedbackSummary, retrievalCounts, signalFiltered, proactiveRefs, highSalienceRefs, } = args;
1362
+ const mergedRefs = args.mergedRefs;
1363
+ // Chunk-5 flip F5e — resolve each candidate's durable item_ref ONCE for this
1364
+ // pass (the write/read key source for the outcome + salience state writers).
1365
+ const itemRefByRef = buildItemRefByRef(mergedRefs);
1264
1366
  // WS-1 — Unified salience vector (S1 seam).
1265
1367
  //
1266
- // The legacy sort combined three independent formulas (utility EMA, negative-only
1267
- // ratio / symmetric-valence magnitude, and the proactive-maintenance priority
1268
- // formula). WS-1 converges them into one `computeSalience()` call per ref, with
1368
+ // WS-1 converges utility, valence, and proactive-maintenance signals into one
1369
+ // `computeSalience()` call per ref, with
1269
1370
  // three independently-stored sub-scores and one documented rankScore projection.
1270
1371
  //
1271
- // Migration note: if a profile still has `symmetricValence` set, emit a one-time
1272
- // warning — its behaviour (symmetric |valence| attention) is now always-on as
1273
- // part of the salience vector, so the knob is a no-op and will be removed in 0.10.
1274
- if (improveProfile.symmetricValence === true) {
1275
- warn("[improve] Profile option 'symmetricValence' is deprecated (WS-1 salience vector). " +
1276
- "Symmetric valence is now always active; remove the option from your improve profile.");
1277
- }
1278
1372
  // Fetch last-use timestamps from the index DB for the full merged set so the
1279
1373
  // recency term in retrievalSalience is genuinely decayable (plan §WS-1 step 2).
1280
1374
  // This reuses the index DB opened earlier for retrieval counts; a separate
1281
1375
  // lightweight open is used here to avoid holding the connection longer than needed.
1282
1376
  let lastUseMsByRef = new Map();
1283
- // utilityMap is kept for backward-compatible observability (health report reads it).
1377
+ // Health and outcome reporting consume the utility projection.
1284
1378
  const utilityMap = buildUtilityMap(mergedRefs);
1285
1379
  let dbForSalience;
1286
1380
  try {
1287
1381
  dbForSalience = openExistingDatabase();
1288
- lastUseMsByRef = getLastUseMsByRef(dbForSalience, mergedRefs.map((r) => r.ref));
1382
+ lastUseMsByRef = getLastUseMsByRef(dbForSalience, mergedRefs.map((r) => r.ref), primaryStashDir);
1289
1383
  }
1290
1384
  catch (err) {
1291
1385
  rethrowIfTestIsolationError(err);
@@ -1295,6 +1389,49 @@ export async function runImprovePreparationStage(args) {
1295
1389
  if (dbForSalience)
1296
1390
  closeDatabase(dbForSalience);
1297
1391
  }
1392
+ const outcomeSalienceByRef = updateOutcomeScores({
1393
+ mergedRefs,
1394
+ itemRefByRef,
1395
+ feedbackSummary,
1396
+ retrievalCounts,
1397
+ lastUseMsByRef,
1398
+ utilityMap,
1399
+ primaryStashDir,
1400
+ eventsCtx,
1401
+ });
1402
+ const { salienceMap, nowForSalience } = computeSalienceVectors({
1403
+ mergedRefs,
1404
+ itemRefByRef,
1405
+ options,
1406
+ eventsCtx,
1407
+ retrievalCounts,
1408
+ lastUseMsByRef,
1409
+ utilityMap,
1410
+ outcomeSalienceByRef,
1411
+ });
1412
+ const pendingForgettingRefs = persistSalienceAndReportRanks({
1413
+ salienceMap,
1414
+ itemRefByRef,
1415
+ utilityMap,
1416
+ feedbackSummary,
1417
+ options,
1418
+ eventsCtx,
1419
+ nowForSalience,
1420
+ });
1421
+ const finalMergedRefs = applyForgettingSafety({
1422
+ pendingForgettingRefs,
1423
+ scope,
1424
+ mergedRefs,
1425
+ eligibilitySourceByRef,
1426
+ highSalienceRefs,
1427
+ proactiveRefs,
1428
+ signalFiltered,
1429
+ });
1430
+ return { mergedRefs: finalMergedRefs, utilityMap, lastUseMsByRef, salienceMap, nowForSalience };
1431
+ }
1432
+ /** WS-2 — update asset_outcome for the merged set; returns outcomeSalience by ref. */
1433
+ function updateOutcomeScores(args) {
1434
+ const { mergedRefs, itemRefByRef, feedbackSummary, retrievalCounts, lastUseMsByRef, utilityMap, primaryStashDir, eventsCtx, } = args;
1298
1435
  // ── WS-2 Outcome loop ─────────────────────────────────────────────────────
1299
1436
  //
1300
1437
  // Update asset_outcome for every ref in the merged set BEFORE computing the
@@ -1336,7 +1473,9 @@ export async function runImprovePreparationStage(args) {
1336
1473
  const valenceResult = computeValenceScore(fb);
1337
1474
  try {
1338
1475
  const result = updateAssetOutcome(outcomeDb, {
1339
- ref: r.ref,
1476
+ // Key by item_ref when resolved, else by the conceptId. Keep
1477
+ // rawOutcomeScores keyed by r.ref, its in-memory identity.
1478
+ ref: outcomeWriteKey(r.ref, itemRefByRef),
1340
1479
  currentRetrievalCount: retrievalCounts.get(r.ref) ?? 0,
1341
1480
  lastRetrievedAt: lastUseMsByRef.get(r.ref) ?? 0,
1342
1481
  acceptedChangeCount: acceptedCountByRef.get(r.ref) ?? 0,
@@ -1361,10 +1500,7 @@ export async function runImprovePreparationStage(args) {
1361
1500
  if (row.outcome_score > maxOutcomeScore)
1362
1501
  maxOutcomeScore = row.outcome_score;
1363
1502
  }
1364
- // Read-clip: legacy rows written before the OUTCOME_SCORE_MAX write-clip
1365
- // existed can sit above the ceiling (live max was 3.13). Without this
1366
- // clip they inflate the normalisation denominator and floor everyone
1367
- // else's outcomeSalience (#691 follow-up).
1503
+ // Keep the normalization denominator within the writer's score bounds.
1368
1504
  maxOutcomeScore = Math.min(maxOutcomeScore, OUTCOME_SCORE_MAX);
1369
1505
  // Proxy-adequacy tripwire (two-tailed): inverted (corr < −0.3) and
1370
1506
  // dead (|corr| < 0.1 at n ≥ 500) both emit health events.
@@ -1402,11 +1538,18 @@ export async function runImprovePreparationStage(args) {
1402
1538
  }
1403
1539
  // Also fetch outcome scores for refs NOT updated this run (stale or absent)
1404
1540
  // so the outcomeSalience read path works for all refs in the batch.
1541
+ // Chunk-5 flip F5e — query by each missing ref's WRITE key (item_ref,
1542
+ // else bare) and map the stored-key result back to the bare `r.ref`
1543
+ // identity that outcomeSalienceByRef is keyed on.
1405
1544
  const missingRefs = mergedRefs.map((r) => r.ref).filter((ref) => !rawOutcomeScores.has(ref));
1406
1545
  if (missingRefs.length > 0) {
1407
- const storedScores = getOutcomeScoresByRef(outcomeDb, missingRefs);
1408
- for (const [ref, score] of storedScores) {
1409
- outcomeSalienceByRef.set(ref, outcomeScoreToSalience(score, maxOutcomeScore));
1546
+ const refByWriteKey = new Map();
1547
+ for (const ref of missingRefs)
1548
+ refByWriteKey.set(outcomeWriteKey(ref, itemRefByRef), ref);
1549
+ const storedScores = getOutcomeScoresByRef(outcomeDb, [...refByWriteKey.keys()]);
1550
+ for (const [writeKey, score] of storedScores) {
1551
+ const bareRef = refByWriteKey.get(writeKey) ?? writeKey;
1552
+ outcomeSalienceByRef.set(bareRef, outcomeScoreToSalience(score, maxOutcomeScore));
1410
1553
  }
1411
1554
  }
1412
1555
  }, { path: eventsCtx?.dbPath, borrowed: eventsCtx?.db });
@@ -1415,6 +1558,11 @@ export async function runImprovePreparationStage(args) {
1415
1558
  rethrowIfTestIsolationError(err);
1416
1559
  // best-effort: outcome failures never block salience computation
1417
1560
  }
1561
+ return outcomeSalienceByRef;
1562
+ }
1563
+ /** WS-1 — compute the salience vector per ref (#644 provenance preserved). */
1564
+ function computeSalienceVectors(args) {
1565
+ const { mergedRefs, itemRefByRef, options, eventsCtx, retrievalCounts, lastUseMsByRef, utilityMap, outcomeSalienceByRef, } = args;
1418
1566
  // Compute the salience vector for every ref in the merged set.
1419
1567
  // retrievalCounts now covers the full candidate set (feedback-bearing + zero-feedback)
1420
1568
  // so feedback refs get their genuine retrieval frequency, not a 0-floor fallback.
@@ -1440,9 +1588,8 @@ export async function runImprovePreparationStage(args) {
1440
1588
  try {
1441
1589
  withStateDb((dbForStoredEncoding) => {
1442
1590
  for (const r of mergedRefs) {
1443
- const type = r.ref.includes(":") ? r.ref.slice(0, r.ref.indexOf(":")) : "";
1444
- const row = getAssetSalience(dbForStoredEncoding, r.ref);
1445
- if (row && isContentEncodingRow(row, type)) {
1591
+ const row = readAssetSalienceForImproveRef(dbForStoredEncoding, r.ref, itemRefByRef.get(r.ref));
1592
+ if (row && isContentEncodingRow(row)) {
1446
1593
  storedEncodingByRef.set(r.ref, row.encoding_salience);
1447
1594
  }
1448
1595
  }
@@ -1453,7 +1600,7 @@ export async function runImprovePreparationStage(args) {
1453
1600
  // best-effort: if DB unavailable, fall back to type-weight stub (prior behaviour)
1454
1601
  }
1455
1602
  for (const r of mergedRefs) {
1456
- const type = r.ref.includes(":") ? r.ref.slice(0, r.ref.indexOf(":")) : "";
1603
+ const type = assetTypeOf(r.ref);
1457
1604
  const sizeBytes = (() => {
1458
1605
  const fp = r.filePath;
1459
1606
  if (!fp)
@@ -1482,6 +1629,40 @@ export async function runImprovePreparationStage(args) {
1482
1629
  });
1483
1630
  salienceMap.set(r.ref, vector);
1484
1631
  }
1632
+ return { salienceMap, nowForSalience };
1633
+ }
1634
+ /** 1-indexed rank positions sorted by score desc (deterministic ref-asc tie-break). */
1635
+ function toRankPositions(scores) {
1636
+ const sorted = [...scores.entries()].sort(([refA, a], [refB, b]) => b !== a ? b - a : refA < refB ? -1 : refA > refB ? 1 : 0);
1637
+ return new Map(sorted.map(([ref], i) => [ref, i + 1]));
1638
+ }
1639
+ /**
1640
+ * Chunk-5 flip F5e — the durable salience write-key maps for one improve pass.
1641
+ * `wk(ref)` is a pool asset's write key (item_ref, else conceptId);
1642
+ * `normalizeStoredKey` maps each in-pool asset's durable spelling to its write
1643
+ * key (no stashSize double-count); and
1644
+ * `refByWriteKey` reverses a write key back to its filesystem-facing bare ref.
1645
+ */
1646
+ function buildSalienceWriteKeyMaps(itemRefByRef) {
1647
+ const wk = (ref) => salienceWriteKey(ref, itemRefByRef);
1648
+ const normalizeStoredKey = new Map();
1649
+ const refByWriteKey = new Map();
1650
+ for (const [ref, itemRef] of itemRefByRef) {
1651
+ const writeKey = wk(ref);
1652
+ refByWriteKey.set(writeKey, ref);
1653
+ for (const spelling of improveStateReadRefs(ref, itemRef)) {
1654
+ normalizeStoredKey.set(spelling, writeKey);
1655
+ }
1656
+ }
1657
+ return { wk, normalizeStoredKey, refByWriteKey };
1658
+ }
1659
+ /** Persist salience vectors + the WS-1 step-7 rank-change/forgetting report. */
1660
+ function persistSalienceAndReportRanks(args) {
1661
+ const { salienceMap, itemRefByRef, utilityMap, feedbackSummary, options, eventsCtx, nowForSalience } = args;
1662
+ // Chunk-5 flip F5e — the WRITE-key space. salienceMap stays keyed by each
1663
+ // candidate's own short `r.ref`; the state.db boundary keys by item_ref when
1664
+ // available and otherwise by conceptId.
1665
+ const { wk, normalizeStoredKey, refByWriteKey } = buildSalienceWriteKeyMaps(itemRefByRef);
1485
1666
  // Persist salience vectors to state.db (best-effort, non-blocking).
1486
1667
  // The canonical store enables WS-3 homeostatic demotion and WS-2 outcome reads.
1487
1668
  //
@@ -1504,7 +1685,7 @@ export async function runImprovePreparationStage(args) {
1504
1685
  // stash-wide), but it is the most faithful comparison possible at cutover and
1505
1686
  // allows the top-200→below-500 forgetting guard to fire if the formula change
1506
1687
  // dramatically reorders the candidate pool.
1507
- // See docs/archive/improve-reconciliation-plan.md §WS-1 step 7 — the stash-wide
1688
+ // WS-1 step 7 — the stash-wide
1508
1689
  // ordering was unreconstructable (no prior state.db snapshot), so this candidate-
1509
1690
  // pool partial reconstruction is the documented resolution for the first-run case.
1510
1691
  // Emit `improve_salience_first_run` to mark the cutover moment and include the
@@ -1534,7 +1715,21 @@ export async function runImprovePreparationStage(args) {
1534
1715
  // Step 7: stash-wide rank-change report BEFORE overwriting the table.
1535
1716
  //
1536
1717
  // Load ALL existing rows so rank positions are stash-relative, not pool-relative.
1537
- const existingAllScores = getAllRankScores(stateDb);
1718
+ // Source-scope by the `<bundle>//` prefix and fold each in-pool asset's
1719
+ // stored spelling onto its single write key so
1720
+ // the merge below never double-counts one asset across two spellings.
1721
+ const allStoredScores = getAllRankScores(stateDb);
1722
+ const existingAllScores = new Map();
1723
+ for (const [ref, score] of allStoredScores) {
1724
+ if (options.sourceName) {
1725
+ const boundary = ref.indexOf("//");
1726
+ const prefix = boundary >= 0 ? ref.slice(0, boundary) : undefined;
1727
+ const belongs = prefix === options.sourceName;
1728
+ if (!belongs)
1729
+ continue;
1730
+ }
1731
+ existingAllScores.set(normalizeStoredKey.get(ref) ?? ref, score);
1732
+ }
1538
1733
  if (existingAllScores.size === 0) {
1539
1734
  // Scenario A: first WS-1 run — table empty.
1540
1735
  //
@@ -1546,7 +1741,7 @@ export async function runImprovePreparationStage(args) {
1546
1741
  // Limitation: this covers only the current-run candidate pool, not the full
1547
1742
  // stash. The stash-wide ordering was never persisted (asset_salience is a new
1548
1743
  // WS-1 table), so this is the most faithful comparison possible at cutover.
1549
- // See docs/archive/improve-reconciliation-plan.md §WS-1 step 7.
1744
+ // WS-1 step 7.
1550
1745
  const reconstructedOldScores = new Map();
1551
1746
  for (const ref of salienceMap.keys()) {
1552
1747
  const utility = utilityMap.get(ref) ?? 0;
@@ -1555,12 +1750,8 @@ export async function runImprovePreparationStage(args) {
1555
1750
  reconstructedOldScores.set(ref, utility * UTILITY_WEIGHT + attention * FEEDBACK_WEIGHT);
1556
1751
  }
1557
1752
  // Assign 1-indexed rank positions sorted by score desc (tie-break: ref asc).
1558
- const toRanks = (scores) => {
1559
- const sorted = [...scores.entries()].sort(([refA, a], [refB, b]) => b !== a ? b - a : refA < refB ? -1 : refA > refB ? 1 : 0);
1560
- return new Map(sorted.map(([ref], i) => [ref, i + 1]));
1561
- };
1562
- const oldRanks = toRanks(reconstructedOldScores);
1563
- const newRanks = toRanks(new Map([...salienceMap.entries()].map(([ref, v]) => [ref, v.rankScore])));
1753
+ const oldRanks = toRankPositions(reconstructedOldScores);
1754
+ const newRanks = toRankPositions(new Map([...salienceMap.entries()].map(([ref, v]) => [ref, v.rankScore])));
1564
1755
  const firstRunReport = buildRankChangeReport(oldRanks, newRanks);
1565
1756
  if (firstRunReport.forgettingCandidates.length > 0) {
1566
1757
  warn(`[improve/salience] WS-1 first-run rank-change report: ${firstRunReport.forgettingCandidates.length} asset(s) fell from top-200 to below position 500 (cutover formula change). ` +
@@ -1575,7 +1766,7 @@ export async function runImprovePreparationStage(args) {
1575
1766
  ref: undefined,
1576
1767
  metadata: {
1577
1768
  candidateCount: salienceMap.size,
1578
- note: "first WS-1 salience run — partial reconstruction of old combinedEligibilityScore ordering for candidate pool (stash-wide ordering not available); see improve-reconciliation-plan.md §WS-1 step 7",
1769
+ note: "first WS-1 salience run — partial reconstruction of old combinedEligibilityScore ordering for candidate pool (stash-wide ordering not available); WS-1 step 7",
1579
1770
  forgettingCandidates: firstRunReport.forgettingCandidates.length,
1580
1771
  topDrops: firstRunReport.forgettingCandidates.slice(0, 10).map((e) => ({
1581
1772
  ref: e.ref,
@@ -1594,15 +1785,13 @@ export async function runImprovePreparationStage(args) {
1594
1785
  // picture of what the new ordering looks like after this run.
1595
1786
  const mergedNewScores = new Map(existingAllScores);
1596
1787
  for (const [ref, vector] of salienceMap) {
1597
- mergedNewScores.set(ref, vector.rankScore);
1788
+ // Chunk-5 flip F5e — key this run's fresh scores by the WRITE key so
1789
+ // they overwrite (never duplicate) the same asset's normalized stored row.
1790
+ mergedNewScores.set(wk(ref), vector.rankScore);
1598
1791
  }
1599
1792
  // Assign 1-indexed rank positions sorted by score desc (tie-break: ref asc).
1600
- const toRanks = (scores) => {
1601
- const sorted = [...scores.entries()].sort(([refA, a], [refB, b]) => b !== a ? b - a : refA < refB ? -1 : refA > refB ? 1 : 0);
1602
- return new Map(sorted.map(([ref], i) => [ref, i + 1]));
1603
- };
1604
- const oldRanks = toRanks(existingAllScores);
1605
- const newRanks = toRanks(mergedNewScores);
1793
+ const oldRanks = toRankPositions(existingAllScores);
1794
+ const newRanks = toRankPositions(mergedNewScores);
1606
1795
  const report = buildRankChangeReport(oldRanks, newRanks);
1607
1796
  if (report.forgettingCandidates.length > 0) {
1608
1797
  warn(`[improve/salience] WS-1 rank-change report: ${report.forgettingCandidates.length} asset(s) fell from top-200 to below position 500. ` +
@@ -1613,7 +1802,11 @@ export async function runImprovePreparationStage(args) {
1613
1802
  // Collect refs for protective consolidation pass (plan §WS-1 step 7).
1614
1803
  // These are force-included in the candidate pool (mergedRefs) after
1615
1804
  // this try block, bypassing cooldown/signal-delta gating.
1616
- pendingForgettingRefs = report.forgettingCandidates.map((e) => e.ref);
1805
+ // Chunk-5 flip F5e — map an in-pool candidate's write-key spelling
1806
+ // back to its bare `r.ref` so applyForgettingSafety re-stamps the
1807
+ // existing pool ref; a genuinely stash-wide (out-of-pool) candidate
1808
+ // keeps its stored spelling and is synthesised as a fresh stub.
1809
+ pendingForgettingRefs = report.forgettingCandidates.map((e) => refByWriteKey.get(e.ref) ?? e.ref);
1617
1810
  }
1618
1811
  appendEvent({
1619
1812
  eventType: "improve_salience_rank_change",
@@ -1631,7 +1824,8 @@ export async function runImprovePreparationStage(args) {
1631
1824
  }, eventsCtx);
1632
1825
  }
1633
1826
  for (const [ref, vector] of salienceMap) {
1634
- upsertAssetSalience(stateDb, ref, vector, nowForSalience);
1827
+ // Persist salience under item_ref when resolved, else the conceptId.
1828
+ upsertAssetSalience(stateDb, wk(ref), vector, nowForSalience);
1635
1829
  }
1636
1830
  }, { path: eventsCtx?.dbPath });
1637
1831
  }
@@ -1639,6 +1833,16 @@ export async function runImprovePreparationStage(args) {
1639
1833
  rethrowIfTestIsolationError(err);
1640
1834
  // best-effort: salience persistence failure never blocks ranking
1641
1835
  }
1836
+ return pendingForgettingRefs;
1837
+ }
1838
+ /**
1839
+ * The protective forgetting-safety injection (plan §WS-1 step 7). Returns the
1840
+ * (possibly extended) mergedRefs; re-stamps lane attribution in precedence
1841
+ * order on the SHARED ref objects and eligibilitySourceByRef map.
1842
+ */
1843
+ export function applyForgettingSafety(args) {
1844
+ const { pendingForgettingRefs, scope, eligibilitySourceByRef, highSalienceRefs, proactiveRefs, signalFiltered } = args;
1845
+ let mergedRefs = args.mergedRefs;
1642
1846
  // ── Protective consolidation pass (plan §WS-1 step 7) ─────────────────────
1643
1847
  // Forgetting candidates detected in scenario B are force-injected into
1644
1848
  // mergedRefs here, BEFORE the effectiveScore sort, bypassing cooldown and
@@ -1657,8 +1861,8 @@ export async function runImprovePreparationStage(args) {
1657
1861
  newForgettingRefs.push({ ref, reason: "scope-type", eligibilitySource: "forgetting-safety" });
1658
1862
  }
1659
1863
  // Always stamp the lane in the attribution map (overwrites weaker lanes;
1660
- // stronger reactive signals — scope/signal-delta/high-retrieval/proactive
1661
- // — are written after this block so they take precedence).
1864
+ // stronger reactive signals — scope/signal-delta/proactive — are written
1865
+ // after this block so they take precedence).
1662
1866
  eligibilitySourceByRef.set(ref, "forgetting-safety");
1663
1867
  }
1664
1868
  if (newForgettingRefs.length > 0) {
@@ -1666,25 +1870,23 @@ export async function runImprovePreparationStage(args) {
1666
1870
  }
1667
1871
  // Re-stamp attribution for any refs whose lane needs updating.
1668
1872
  // Precedence (weakest → strongest, each overwrites the previous):
1669
- // proactive < high-retrieval < forgetting-safety < signal-delta
1873
+ // proactive < forgetting-safety < signal-delta
1670
1874
  // Scope mode is already excluded by the outer guard (`scope.mode !== "ref"`).
1671
- // forgetting-safety sits above proactive and high-retrieval so that a ref
1672
- // flagged as a forgetting candidate is always visible to S5/WS-5 as such,
1673
- // even when it was also due for a proactive maintenance run. signal-delta
1674
- // overrides forgetting-safety because a ref with fresh feedback is reactive
1675
- // and doesn't need the protective pass label for measurement purposes.
1875
+ // forgetting-safety sits above proactive so that a ref flagged as a
1876
+ // forgetting candidate is always visible to S5/WS-5 as such, even when it
1877
+ // was also due for a proactive maintenance run. signal-delta overrides
1878
+ // forgetting-safety because a ref with fresh feedback is reactive and
1879
+ // doesn't need the protective pass label for measurement purposes.
1676
1880
  for (const r of highSalienceRefs)
1677
1881
  eligibilitySourceByRef.set(r.ref, "high-salience");
1678
1882
  for (const r of proactiveRefs)
1679
1883
  eligibilitySourceByRef.set(r.ref, "proactive");
1680
- for (const r of highRetrievalRefs)
1681
- eligibilitySourceByRef.set(r.ref, "high-retrieval");
1682
- // Apply forgetting-safety OVER proactive, high-retrieval, and high-salience
1683
- // (already stamped in the loop above via
1884
+ // Apply forgetting-safety OVER proactive and high-salience (already
1885
+ // stamped in the loop above via
1684
1886
  // `eligibilitySourceByRef.set(ref, "forgetting-safety")`). No-op here: the
1685
- // set() calls above for proactive/high-retrieval/high-salience overwrite the
1686
- // earlier forgetting-safety stamp — so we re-apply forgetting-safety now for
1687
- // those refs that are both forgetting candidates AND in another fallback lane.
1887
+ // set() calls above for proactive/high-salience overwrite the earlier
1888
+ // forgetting-safety stamp — so we re-apply forgetting-safety now for those
1889
+ // refs that are both forgetting candidates AND in another fallback lane.
1688
1890
  for (const ref of pendingForgettingRefs) {
1689
1891
  eligibilitySourceByRef.set(ref, "forgetting-safety");
1690
1892
  }
@@ -1697,6 +1899,158 @@ export async function runImprovePreparationStage(args) {
1697
1899
  r.eligibilitySource = eligibilitySourceByRef.get(r.ref) ?? "unknown";
1698
1900
  }
1699
1901
  }
1902
+ return mergedRefs;
1903
+ }
1904
+ /**
1905
+ * Pass: eligibility-filter — replay selection (#610), the no-op dampener sort,
1906
+ * coverage gaps, the disk-existence guard, the --limit slice, and the summary
1907
+ * info emits.
1908
+ */
1909
+ async function filterEligibility(args) {
1910
+ const { scope, options, plannedRefs, eventsCtx, salienceMap, eligibilitySourceByRef, distillOnlyRefs } = args;
1911
+ const { fullySkippedCount, preCooldownCount, signalAndRetrievalRefs, signalFiltered } = args.summary;
1912
+ const validationFailureRefs = args.validationFailureRefs;
1913
+ const replay = applyReplaySelection({
1914
+ scope,
1915
+ options,
1916
+ plannedRefs,
1917
+ eventsCtx,
1918
+ mergedRefs: args.mergedRefs,
1919
+ salienceMap,
1920
+ eligibilitySourceByRef,
1921
+ });
1922
+ const mergedRefs = replay.mergedRefs;
1923
+ const { replayRefSet, replayBudget } = replay;
1924
+ // Build no-op map for consolidation-selection dampener (plan §WS-1 step 8).
1925
+ // Reads consecutive_no_ops from the SAME pinned db handle used elsewhere in
1926
+ // this function. The effective score is used ONLY for processing/selection
1927
+ // order — the persisted rank_score in asset_salience is never mutated here.
1928
+ const noOpMap = new Map();
1929
+ try {
1930
+ const noOpDb = eventsCtx?.db ?? (eventsCtx?.dbPath ? openStateDatabase(eventsCtx.dbPath) : null);
1931
+ if (noOpDb) {
1932
+ const ownsNoOpDb = !eventsCtx?.db;
1933
+ try {
1934
+ for (const r of mergedRefs) {
1935
+ noOpMap.set(r.ref, readConsecutiveNoOpsForImproveRef(noOpDb, r.ref, r.itemRef));
1936
+ }
1937
+ }
1938
+ finally {
1939
+ if (ownsNoOpDb)
1940
+ noOpDb.close();
1941
+ }
1942
+ }
1943
+ }
1944
+ catch {
1945
+ // best-effort: dampener failure never blocks selection
1946
+ }
1947
+ // Sort by effective selection score (desc), with explicit ref-string tie-break
1948
+ // for determinism. The effective score applies the consolidation-selection
1949
+ // dampener: assets that have been repeatedly skipped (consecutive_no_ops >=
1950
+ // THRESHOLD) are penalised by FACTOR so they sort after peers with similar
1951
+ // rankScore. The persisted rank_score is left unchanged — this is the whole
1952
+ // point of the dampener (stable assets stay fully retrievable).
1953
+ //
1954
+ // WIRING NOTE (plan §WS-1 step 8 / "consolidation-selection" disambiguation):
1955
+ // "consolidation-selection" in the plan refers to THIS reflect/distill
1956
+ // eligibility ordering — i.e. which assets are chosen for the reflect/distill
1957
+ // LLM pass — NOT to akmConsolidate (the cluster-merge phase at ~line 1994,
1958
+ // which runs earlier and never reads noOpMap). The no-op counter originates
1959
+ // from no-change reflect / quality-rejected distill outcomes; the dampener
1960
+ // suppresses repeated LLM attempts on those same assets without touching their
1961
+ // persisted rank_score (so they remain fully retrievable).
1962
+ //
1963
+ // This is the only ranking path. The eligibilitySource lanes (signal-delta /
1964
+ // proactive / high-salience) survive as labels set above.
1965
+ const effectiveScore = (ref) => {
1966
+ const rankScore = salienceMap.get(ref)?.rankScore ?? 0;
1967
+ const noOps = noOpMap.get(ref) ?? 0;
1968
+ return noOps >= SALIENCE_NO_OP_DAMPEN_THRESHOLD ? rankScore * SALIENCE_NO_OP_DAMPEN_FACTOR : rankScore;
1969
+ };
1970
+ const sorted = [...mergedRefs].sort((a, b) => {
1971
+ const scoreA = effectiveScore(a.ref);
1972
+ const scoreB = effectiveScore(b.ref);
1973
+ if (scoreB !== scoreA)
1974
+ return scoreB - scoreA;
1975
+ // Stable tie-break: deterministic regardless of input ordering.
1976
+ return a.ref < b.ref ? -1 : a.ref > b.ref ? 1 : 0;
1977
+ });
1978
+ // Phase 0: surface coverage gaps from zero-result search queries
1979
+ let coverageGaps = [];
1980
+ try {
1981
+ const dbForGaps = openExistingDatabase();
1982
+ try {
1983
+ coverageGaps = getZeroResultSearches(dbForGaps);
1984
+ }
1985
+ finally {
1986
+ closeDatabase(dbForGaps);
1987
+ }
1988
+ }
1989
+ catch (err) {
1990
+ rethrowIfTestIsolationError(err);
1991
+ // best-effort
1992
+ }
1993
+ const diskCheck = await dropRefsMissingOnDisk({ sorted, options, eventsCtx });
1994
+ const assetMissingOnDisk = diskCheck.assetMissingOnDisk;
1995
+ const actionableRefs = diskCheck.actionableRefs;
1996
+ // Re-split actionableRefs (sorted) into reflect-path vs distill-only-path while
1997
+ // preserving sort order. distillOnlyRefs participate in the sort so --limit
1998
+ // picks them by score, not by arbitrary position.
1999
+ const distillOnlyRefSetForSort = new Set(distillOnlyRefs.map((r) => r.ref));
2000
+ const reflectAndDistillRefsAfterSort = [];
2001
+ const distillOnlyRefsAfterSort = [];
2002
+ for (const r of actionableRefs) {
2003
+ if (distillOnlyRefSetForSort.has(r.ref)) {
2004
+ distillOnlyRefsAfterSort.push(r);
2005
+ }
2006
+ else {
2007
+ reflectAndDistillRefsAfterSort.push(r);
2008
+ }
2009
+ }
2010
+ // ── Phase 5: --limit applies to the post-cooldown actionable set ──────────
2011
+ //
2012
+ // #610 ADDITIVITY: replay-lane refs are budgeted SEPARATELY from the --limit
2013
+ // fresh slice. Without this split, a high-rankScore replay ref could sort above
2014
+ // a fresh ref in the single combined slice and STEAL its slot (violating AC2).
2015
+ // We partition into the replay lane vs the rest, apply --limit to the
2016
+ // non-replay (fresh) refs only, then APPEND up to `replayBudget` replay refs
2017
+ // after the fresh slice. Sort order within each partition is preserved.
2018
+ //
2019
+ // Default replayBudget=0 reduces this to the exact pre-#610 expression: with no
2020
+ // replay refs, `nonReplayLoop === allLoopRefs`, so `baseLoop === old slice` and
2021
+ // `replayLoop.slice(0, 0) === []` — byte-identical.
2022
+ const allLoopRefs = [...reflectAndDistillRefsAfterSort, ...distillOnlyRefsAfterSort];
2023
+ const replayLoop = allLoopRefs.filter((r) => r.eligibilitySource === "replay");
2024
+ const nonReplayLoop = allLoopRefs.filter((r) => r.eligibilitySource !== "replay");
2025
+ const baseLoop = options.limit ? nonReplayLoop.slice(0, options.limit) : nonReplayLoop;
2026
+ const loopRefs = [...baseLoop, ...replayLoop.slice(0, replayBudget)];
2027
+ // Update the returned distillOnlyRefs to the sorted order so callers see the
2028
+ // ranked view (loop stage uses it as a Set so order is irrelevant, but the
2029
+ // shape change keeps downstream consumers consistent).
2030
+ const distillOnlyRefsResult = distillOnlyRefsAfterSort;
2031
+ const totalReflectBlocked = fullySkippedCount + distillOnlyRefs.length;
2032
+ if (totalReflectBlocked > 0) {
2033
+ info(`[improve] ${totalReflectBlocked} of ${preCooldownCount} indexed refs blocked by reflect signal-delta ` +
2034
+ `(${fullySkippedCount} fully skipped, ${distillOnlyRefs.length} routed to distill-only)`);
2035
+ }
2036
+ if (signalAndRetrievalRefs.length > 0) {
2037
+ info(`[improve] ${signalAndRetrievalRefs.length} refs with usage signals (${signalFiltered.length} feedback${replayRefSet.size > 0 ? `, ${replayRefSet.size} replay` : ""})`);
2038
+ }
2039
+ if (validationFailureRefs.size > 0) {
2040
+ info(`[improve] ${validationFailureRefs.size} with validation failures excluded`);
2041
+ }
2042
+ if (assetMissingOnDisk.length > 0) {
2043
+ info(`[improve] ${assetMissingOnDisk.length} candidates dropped — file not on disk`);
2044
+ }
2045
+ const deferredCount = actionableRefs.length - loopRefs.length;
2046
+ info(`[improve] ${actionableRefs.length} actionable; ${loopRefs.length} will be processed` +
2047
+ (options.limit && deferredCount > 0 ? ` (--limit ${options.limit} applied; ${deferredCount} deferred)` : ""));
2048
+ return { loopRefs, actionableRefs, distillOnlyRefs: distillOnlyRefsResult, coverageGaps };
2049
+ }
2050
+ /** The #610 bounded, additive replay-selection lane (eligibility-filter). */
2051
+ function applyReplaySelection(args) {
2052
+ const { scope, options, plannedRefs, eventsCtx, salienceMap, eligibilitySourceByRef } = args;
2053
+ let mergedRefs = args.mergedRefs;
1700
2054
  // ── REPLAY SELECTION layer (#610) ─────────────────────────────────────────
1701
2055
  // Bounded, ADDITIVE replay budget: up to `replayBudget` top-salience refs are
1702
2056
  // revisited even with zero reactive signal (no feedback, no retrieval) and
@@ -1717,7 +2071,31 @@ export async function runImprovePreparationStage(args) {
1717
2071
  try {
1718
2072
  withStateDb((replayDb) => {
1719
2073
  const alreadyInPool = new Set(mergedRefs.map((r) => r.ref));
1720
- const allRankScores = getAllRankScores(replayDb);
2074
+ const storedRankScores = getAllRankScores(replayDb);
2075
+ // Chunk-5 flip F5e — plan reverse map so a stored item_ref row re-keys
2076
+ // back onto its filesystem-facing bare ref (the replay stub must match
2077
+ // a planned entry for its filePath / disk check).
2078
+ const itemRefByPlanned = buildItemRefByRef(plannedRefs);
2079
+ const bareRefByItemRef = new Map();
2080
+ for (const [bareRef, itemRef] of itemRefByPlanned)
2081
+ if (itemRef)
2082
+ bareRefByItemRef.set(itemRef, bareRef);
2083
+ const allRankScores = options.sourceName
2084
+ ? (() => {
2085
+ // Source-scope by the bundle prefix, then re-key item_ref rows
2086
+ // onto the short conceptId used by the planned-ref pool.
2087
+ const m = new Map();
2088
+ for (const [ref, score] of storedRankScores) {
2089
+ const boundary = ref.indexOf("//");
2090
+ const prefix = boundary >= 0 ? ref.slice(0, boundary) : undefined;
2091
+ const belongs = prefix === options.sourceName;
2092
+ if (!belongs)
2093
+ continue;
2094
+ m.set(bareRefByItemRef.get(ref) ?? bareImproveRef(ref), score);
2095
+ }
2096
+ return m;
2097
+ })()
2098
+ : storedRankScores;
1721
2099
  // Candidate universe = every salience row NOT already in the pool, ordered by
1722
2100
  // rank_score desc with a deterministic ref-string tie-break (mirrors the main
1723
2101
  // sort). Converged refs (consecutive_no_ops >= dampener threshold) are fully
@@ -1727,7 +2105,7 @@ export async function runImprovePreparationStage(args) {
1727
2105
  for (const [ref, rankScore] of allRankScores) {
1728
2106
  if (alreadyInPool.has(ref))
1729
2107
  continue;
1730
- const noOps = getConsecutiveNoOps(replayDb, ref);
2108
+ const noOps = readConsecutiveNoOpsForImproveRef(replayDb, ref, itemRefByPlanned.get(ref));
1731
2109
  if (noOps >= SALIENCE_NO_OP_DAMPEN_THRESHOLD) {
1732
2110
  convergedSkipped++;
1733
2111
  continue;
@@ -1793,76 +2171,11 @@ export async function runImprovePreparationStage(args) {
1793
2171
  // best-effort: if DB unavailable, replayRefSet stays empty
1794
2172
  }
1795
2173
  }
1796
- // Build no-op map for consolidation-selection dampener (plan §WS-1 step 8).
1797
- // Reads consecutive_no_ops from the SAME pinned db handle used elsewhere in
1798
- // this function. The effective score is used ONLY for processing/selection
1799
- // order — the persisted rank_score in asset_salience is never mutated here.
1800
- const noOpMap = new Map();
1801
- try {
1802
- const noOpDb = eventsCtx?.db ?? (eventsCtx?.dbPath ? openStateDatabase(eventsCtx.dbPath) : null);
1803
- if (noOpDb) {
1804
- const ownsNoOpDb = !eventsCtx?.db;
1805
- try {
1806
- for (const r of mergedRefs) {
1807
- noOpMap.set(r.ref, getConsecutiveNoOps(noOpDb, r.ref));
1808
- }
1809
- }
1810
- finally {
1811
- if (ownsNoOpDb)
1812
- noOpDb.close();
1813
- }
1814
- }
1815
- }
1816
- catch {
1817
- // best-effort: dampener failure never blocks selection
1818
- }
1819
- // Sort by effective selection score (desc), with explicit ref-string tie-break
1820
- // for determinism. The effective score applies the consolidation-selection
1821
- // dampener: assets that have been repeatedly skipped (consecutive_no_ops >=
1822
- // THRESHOLD) are penalised by FACTOR so they sort after peers with similar
1823
- // rankScore. The persisted rank_score is left unchanged — this is the whole
1824
- // point of the dampener (stable assets stay fully retrievable).
1825
- //
1826
- // WIRING NOTE (plan §WS-1 step 8 / "consolidation-selection" disambiguation):
1827
- // "consolidation-selection" in the plan refers to THIS reflect/distill
1828
- // eligibility ordering — i.e. which assets are chosen for the reflect/distill
1829
- // LLM pass — NOT to akmConsolidate (the cluster-merge phase at ~line 1994,
1830
- // which runs earlier and never reads noOpMap). The no-op counter originates
1831
- // from no-change reflect / quality-rejected distill outcomes; the dampener
1832
- // suppresses repeated LLM attempts on those same assets without touching their
1833
- // persisted rank_score (so they remain fully retrievable).
1834
- //
1835
- // This is the ONLY ranking path — negativeOnlyRatio and the legacy
1836
- // symmetricValence branch are replaced. The three eligibilitySource lanes
1837
- // (signal-delta / high-retrieval / proactive) survive as labels (set above).
1838
- const effectiveScore = (ref) => {
1839
- const rankScore = salienceMap.get(ref)?.rankScore ?? 0;
1840
- const noOps = noOpMap.get(ref) ?? 0;
1841
- return noOps >= SALIENCE_NO_OP_DAMPEN_THRESHOLD ? rankScore * SALIENCE_NO_OP_DAMPEN_FACTOR : rankScore;
1842
- };
1843
- const sorted = [...mergedRefs].sort((a, b) => {
1844
- const scoreA = effectiveScore(a.ref);
1845
- const scoreB = effectiveScore(b.ref);
1846
- if (scoreB !== scoreA)
1847
- return scoreB - scoreA;
1848
- // Stable tie-break: deterministic regardless of input ordering.
1849
- return a.ref < b.ref ? -1 : a.ref > b.ref ? 1 : 0;
1850
- });
1851
- // Phase 0: surface coverage gaps from zero-result search queries
1852
- let coverageGaps = [];
1853
- try {
1854
- const dbForGaps = openExistingDatabase();
1855
- try {
1856
- coverageGaps = getZeroResultSearches(dbForGaps);
1857
- }
1858
- finally {
1859
- closeDatabase(dbForGaps);
1860
- }
1861
- }
1862
- catch (err) {
1863
- rethrowIfTestIsolationError(err);
1864
- // best-effort
1865
- }
2174
+ return { mergedRefs, replayRefSet, replayBudget };
2175
+ }
2176
+ /** The final disk-existence guard + its aggregated audit event (eligibility-filter). */
2177
+ async function dropRefsMissingOnDisk(args) {
2178
+ const { sorted, options, eventsCtx } = args;
1866
2179
  // actionableRefs is the post-cooldown, post-validation, post-signal, post-sort
1867
2180
  // set — i.e. the genuinely processable refs in priority order. Note: this is
1868
2181
  // a semantic shift from earlier code where actionableRefs was the pre-cooldown
@@ -1906,97 +2219,5 @@ export async function runImprovePreparationStage(args) {
1906
2219
  },
1907
2220
  }, eventsCtx);
1908
2221
  }
1909
- const actionableRefs = existsCheckedActionable;
1910
- // Re-split actionableRefs (sorted) into reflect-path vs distill-only-path while
1911
- // preserving sort order. distillOnlyRefs participate in the sort so --limit
1912
- // picks them by score, not by arbitrary position.
1913
- const distillOnlyRefSetForSort = new Set(distillOnlyRefs.map((r) => r.ref));
1914
- const reflectAndDistillRefsAfterSort = [];
1915
- const distillOnlyRefsAfterSort = [];
1916
- for (const r of actionableRefs) {
1917
- if (distillOnlyRefSetForSort.has(r.ref)) {
1918
- distillOnlyRefsAfterSort.push(r);
1919
- }
1920
- else {
1921
- reflectAndDistillRefsAfterSort.push(r);
1922
- }
1923
- }
1924
- // ── Phase 5: --limit applies to the post-cooldown actionable set ──────────
1925
- //
1926
- // #610 ADDITIVITY: replay-lane refs are budgeted SEPARATELY from the --limit
1927
- // fresh slice. Without this split, a high-rankScore replay ref could sort above
1928
- // a fresh ref in the single combined slice and STEAL its slot (violating AC2).
1929
- // We partition into the replay lane vs the rest, apply --limit to the
1930
- // non-replay (fresh) refs only, then APPEND up to `replayBudget` replay refs
1931
- // after the fresh slice. Sort order within each partition is preserved.
1932
- //
1933
- // Default replayBudget=0 reduces this to the exact pre-#610 expression: with no
1934
- // replay refs, `nonReplayLoop === allLoopRefs`, so `baseLoop === old slice` and
1935
- // `replayLoop.slice(0, 0) === []` — byte-identical.
1936
- const allLoopRefs = [...reflectAndDistillRefsAfterSort, ...distillOnlyRefsAfterSort];
1937
- const replayLoop = allLoopRefs.filter((r) => r.eligibilitySource === "replay");
1938
- const nonReplayLoop = allLoopRefs.filter((r) => r.eligibilitySource !== "replay");
1939
- const baseLoop = options.limit ? nonReplayLoop.slice(0, options.limit) : nonReplayLoop;
1940
- const loopRefs = [...baseLoop, ...replayLoop.slice(0, replayBudget)];
1941
- // Update the returned distillOnlyRefs to the sorted order so callers see the
1942
- // ranked view (loop stage uses it as a Set so order is irrelevant, but the
1943
- // shape change keeps downstream consumers consistent).
1944
- const distillOnlyRefsResult = distillOnlyRefsAfterSort;
1945
- const totalReflectBlocked = fullySkippedCount + distillOnlyRefs.length;
1946
- if (totalReflectBlocked > 0) {
1947
- info(`[improve] ${totalReflectBlocked} of ${preCooldownCount} indexed refs blocked by reflect signal-delta ` +
1948
- `(${fullySkippedCount} fully skipped, ${distillOnlyRefs.length} routed to distill-only)`);
1949
- }
1950
- if (signalAndRetrievalRefs.length > 0) {
1951
- info(`[improve] ${signalAndRetrievalRefs.length} refs with usage signals (${signalFiltered.length} feedback, ${highRetrievalRefs.length} high-retrieval${replayRefSet.size > 0 ? `, ${replayRefSet.size} replay` : ""})`);
1952
- }
1953
- if (validationFailureRefs.size > 0) {
1954
- info(`[improve] ${validationFailureRefs.size} with validation failures excluded`);
1955
- }
1956
- if (assetMissingOnDisk.length > 0) {
1957
- info(`[improve] ${assetMissingOnDisk.length} candidates dropped — file not on disk`);
1958
- }
1959
- const deferredCount = actionableRefs.length - loopRefs.length;
1960
- info(`[improve] ${actionableRefs.length} actionable; ${loopRefs.length} will be processed` +
1961
- (options.limit && deferredCount > 0 ? ` (--limit ${options.limit} applied; ${deferredCount} deferred)` : ""));
1962
- // WS-4: Per-phase threshold auto-tune for the extract phase.
1963
- // Persists result for the NEXT run's makeGateConfig to read.
1964
- const extractTuneDbPath = eventsCtx?.dbPath;
1965
- if (options.autoAccept !== undefined && extractTuneDbPath) {
1966
- try {
1967
- maybeAutoTuneThreshold(extractPass.extractGateCfg.phaseThreshold ?? options.autoAccept, options.config ?? loadConfig(), extractTuneDbPath, undefined, "extract");
1968
- }
1969
- catch (err) {
1970
- warn(`[improve] calibration auto-tune (extract) skipped: ${err instanceof Error ? err.message : String(err)}`);
1971
- }
1972
- }
1973
- return {
1974
- actions,
1975
- cleanupWarnings,
1976
- appliedCleanup,
1977
- memoryIndexHealth,
1978
- extract: extractResults,
1979
- actionableRefs,
1980
- signalBearingSet,
1981
- validationFailures,
1982
- schemaRepairs,
1983
- lintSummary,
1984
- loopRefs,
1985
- distillCooledRefs,
1986
- distillOnlyRefs: distillOnlyRefsResult,
1987
- coverageGaps,
1988
- recentErrors,
1989
- utilityMap,
1990
- gateAutoAcceptedCount,
1991
- gateAutoAcceptFailedCount,
1992
- consolidation: consolidationPass.consolidation,
1993
- consolidationRan: consolidationPass.consolidationRan,
1994
- ...(proactiveMaintenanceSummary ? { proactiveMaintenance: proactiveMaintenanceSummary } : {}),
1995
- };
2222
+ return { actionableRefs: existsCheckedActionable, assetMissingOnDisk };
1996
2223
  }
1997
- // TODO(refactor): 13 args including `actions`/`recentErrors` mutation channels. Restructure into immutable plan + mutable context objects — deferred to dedicated refactor with isolated testing.
1998
- /**
1999
- * Parameter object for {@link runImproveLoopStage} (WS10). Pure type reshape of
2000
- * the former inline arg struct — every field, name, and type is preserved so the
2001
- * function body and all runtime values are byte-identical. No control-flow change.
2002
- */