akm-cli 0.9.17-alpha.2 → 0.9.17-alpha.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (343) hide show
  1. package/CHANGELOG.md +756 -0
  2. package/dist/akm +94 -196
  3. package/dist/cli/shared.js +6 -2
  4. package/dist/cli.js +22 -9
  5. package/dist/commands/agent/agent-dispatch.js +1 -1
  6. package/dist/commands/command/command-execution.js +24 -62
  7. package/dist/commands/feedback-cli.js +0 -1
  8. package/dist/commands/health/accept-rate.js +2 -2
  9. package/dist/commands/health/checks.js +30 -75
  10. package/dist/commands/health/config-skew.js +38 -0
  11. package/dist/commands/health/egress.js +54 -0
  12. package/dist/commands/health/html-report.js +0 -38
  13. package/dist/commands/health/improve-metrics.js +123 -562
  14. package/dist/commands/health/plugin-staleness.js +53 -3
  15. package/dist/commands/health/renderers.js +12 -4
  16. package/dist/commands/health/report-view-model.js +11 -106
  17. package/dist/commands/health/types-improve.js +4 -19
  18. package/dist/commands/health/windows.js +64 -73
  19. package/dist/commands/health.js +122 -143
  20. package/dist/commands/improve/consolidate/chunking.js +25 -100
  21. package/dist/commands/improve/consolidate/sanitize.js +54 -149
  22. package/dist/commands/improve/consolidate.js +538 -1075
  23. package/dist/commands/improve/content-hash.js +16 -24
  24. package/dist/commands/improve/distill/content-repair.js +18 -100
  25. package/dist/commands/improve/distill-guards.js +20 -81
  26. package/dist/commands/improve/distill-promotion-policy.js +23 -243
  27. package/dist/commands/improve/distill.js +608 -1075
  28. package/dist/commands/improve/eligibility.js +126 -400
  29. package/dist/commands/improve/execution.js +3 -5
  30. package/dist/commands/improve/extract.js +487 -1046
  31. package/dist/commands/improve/feedback-valence.js +0 -25
  32. package/dist/commands/improve/improve-cli.js +29 -166
  33. package/dist/commands/improve/improve-result-file.js +10 -66
  34. package/dist/commands/improve/improve-strategies.js +12 -7
  35. package/dist/commands/improve/improve-usage-report.js +18 -64
  36. package/dist/commands/improve/improve.js +443 -1063
  37. package/dist/commands/improve/ledger.js +114 -0
  38. package/dist/commands/improve/locks.js +2 -8
  39. package/dist/commands/improve/loop-stages.js +459 -1172
  40. package/dist/commands/improve/memory/derived-ref.js +12 -77
  41. package/dist/commands/improve/memory/memory-belief.js +14 -118
  42. package/dist/commands/improve/memory/memory-improve.js +4 -3
  43. package/dist/commands/improve/outcome-loop.js +28 -156
  44. package/dist/commands/improve/planner.js +5 -10
  45. package/dist/commands/improve/preparation.js +851 -2339
  46. package/dist/commands/improve/proactive-maintenance.js +34 -101
  47. package/dist/commands/improve/reflect-noise.js +104 -280
  48. package/dist/commands/improve/reflect.js +621 -1367
  49. package/dist/commands/improve/salience.js +46 -232
  50. package/dist/commands/improve/session-asset.js +19 -100
  51. package/dist/commands/improve/stage.js +323 -0
  52. package/dist/commands/proposal/drain.js +251 -644
  53. package/dist/commands/proposal/proposal-cli.js +3 -18
  54. package/dist/commands/proposal/proposal-types.js +20 -41
  55. package/dist/commands/proposal/proposal.js +1 -2
  56. package/dist/commands/proposal/propose.js +134 -160
  57. package/dist/commands/proposal/repository.js +502 -1487
  58. package/dist/commands/proposal/validators/proposal-quality-validators.js +71 -174
  59. package/dist/commands/proposal/validators/proposal-validators.js +1 -1
  60. package/dist/commands/proposal/validators/proposals.js +13 -89
  61. package/dist/commands/read/curate.js +63 -413
  62. package/dist/commands/read/search-cli.js +16 -33
  63. package/dist/commands/read/search.js +17 -23
  64. package/dist/commands/read/show.js +2 -13
  65. package/dist/commands/sources/bundle-cli.js +25 -2
  66. package/dist/commands/sources/bundle-config-ops.js +7 -0
  67. package/dist/commands/sources/dangerous-env-audit.js +1 -2
  68. package/dist/commands/sources/info.js +2 -11
  69. package/dist/commands/sources/installed-stashes.js +197 -746
  70. package/dist/commands/sources/schema-repair.js +98 -129
  71. package/dist/commands/sources/source-add.js +62 -12
  72. package/dist/commands/sources/stash-cli.js +1 -1
  73. package/dist/commands/tasks/explain.js +10 -13
  74. package/dist/commands/tasks/tasks-cli.js +9 -8
  75. package/dist/commands/tasks/tasks.js +326 -930
  76. package/dist/commands/tasks/validate.js +42 -21
  77. package/dist/commands/workflow/plan.js +22 -29
  78. package/dist/commands/workflow-cli.js +4 -4
  79. package/dist/core/adapter/adapters/akm-adapter.js +0 -1
  80. package/dist/core/adapter/adapters/akm-lint.js +2 -3
  81. package/dist/core/adapter/adapters/akm-metadata.js +11 -12
  82. package/dist/core/adapter/adapters/akm-workflow-adapter.js +1 -1
  83. package/dist/core/adapter/execution-source.js +17 -29
  84. package/dist/core/asset/resolve-ref.js +1 -1
  85. package/dist/core/bundle-id.js +42 -5
  86. package/dist/core/bundle-rename.js +291 -0
  87. package/dist/core/config/config-io.js +1 -2
  88. package/dist/core/config/config-schema.js +1 -33
  89. package/dist/core/config/config-walker.js +1 -1
  90. package/dist/core/config/config.js +163 -68
  91. package/dist/core/config/legacy-source-shape-shim.js +38 -9
  92. package/dist/core/config/schema/embedding.js +20 -5
  93. package/dist/core/config/schema/engines.js +5 -0
  94. package/dist/core/config/schema/execution.js +1 -1
  95. package/dist/core/config/schema/experimental.js +1 -1
  96. package/dist/core/config/schema/improve-processes.js +21 -95
  97. package/dist/core/config/schema/improve.js +4 -42
  98. package/dist/core/config/schema/scheduler.js +12 -12
  99. package/dist/core/config/schema/search.js +6 -22
  100. package/dist/core/env-secret-ref.js +0 -1
  101. package/dist/core/errors.js +8 -9
  102. package/dist/core/file-lock.js +76 -173
  103. package/dist/core/logs-db.js +2 -2
  104. package/dist/core/paths.js +0 -27
  105. package/dist/core/redaction.js +109 -2
  106. package/dist/core/run-lock.js +2 -5
  107. package/dist/core/spawn-env.js +1 -1
  108. package/dist/core/state/migrations.js +108 -61
  109. package/dist/core/state-db-scope.js +2 -4
  110. package/dist/core/state-db.js +126 -692
  111. package/dist/core/type-presentation.js +1 -9
  112. package/dist/core/write-source.js +293 -1012
  113. package/dist/execution/input-contract.js +1 -1
  114. package/dist/execution/resolved-request.js +135 -689
  115. package/dist/execution/source.js +63 -257
  116. package/dist/execution/target-ref.js +1 -1
  117. package/dist/indexer/bundle-identity-guard.js +2 -2
  118. package/dist/indexer/db/graph-db.js +106 -46
  119. package/dist/indexer/ensure-index.js +44 -85
  120. package/dist/indexer/graph/graph-extraction.js +340 -562
  121. package/dist/indexer/graph/graph-related.js +130 -0
  122. package/dist/indexer/index-rebuild-lock.js +3 -11
  123. package/dist/indexer/index-writer-lock.js +8 -17
  124. package/dist/indexer/index-written-assets.js +139 -151
  125. package/dist/indexer/indexer.js +524 -846
  126. package/dist/indexer/materialize-embeddings.js +60 -397
  127. package/dist/indexer/passes/memory-inference.js +81 -90
  128. package/dist/indexer/passes/metadata.js +132 -200
  129. package/dist/indexer/read-preflight.js +0 -7
  130. package/dist/indexer/scan/doc-to-entry.js +1 -3
  131. package/dist/indexer/scan/drain-dir.js +1 -1
  132. package/dist/indexer/search/db-search.js +181 -590
  133. package/dist/indexer/search/fts-query.js +30 -41
  134. package/dist/indexer/search/ranking.js +28 -154
  135. package/dist/indexer/search/search-attribution.js +12 -32
  136. package/dist/indexer/search/search-fields.js +11 -15
  137. package/dist/indexer/search/search-hit-enrichers.js +54 -85
  138. package/dist/indexer/search/search-source.js +1 -4
  139. package/dist/indexer/usage/usage-events.js +2 -7
  140. package/dist/integrations/agent/engine-fallback.js +23 -40
  141. package/dist/integrations/agent/engine-resolution.js +93 -183
  142. package/dist/integrations/agent/execution.js +507 -0
  143. package/dist/integrations/agent/model-map.js +28 -156
  144. package/dist/integrations/agent/request-lowering.js +66 -141
  145. package/dist/integrations/agent/runner-dispatch.js +143 -321
  146. package/dist/integrations/agent/runner.js +54 -14
  147. package/dist/integrations/lockfile.js +53 -101
  148. package/dist/llm/embedders/deterministic.js +2 -3
  149. package/dist/llm/embedders/profile.js +71 -0
  150. package/dist/llm/embedders/remote.js +10 -15
  151. package/dist/llm/graph-extract.js +3 -12
  152. package/dist/llm/index-passes.js +3 -5
  153. package/dist/llm/memory-infer.js +1 -2
  154. package/dist/llm/metadata-enhance.js +1 -2
  155. package/dist/llm/structured-call.js +5 -24
  156. package/dist/output/generic-render.js +23 -11
  157. package/dist/output/html-render.js +13 -10
  158. package/dist/output/render-registry.js +3 -32
  159. package/dist/output/shapes/helpers.js +2 -34
  160. package/dist/output/shapes/passthrough.js +1 -9
  161. package/dist/{indexer/search/ranking-types.js → output/text/bundle-rename.js} +4 -1
  162. package/dist/output/text/command-format.js +60 -23
  163. package/dist/output/text/helpers.js +1 -1
  164. package/dist/output/text/migrate.js +5 -14
  165. package/dist/output/text/proposal-format.js +1 -2
  166. package/dist/output/text/workflow-format.js +0 -32
  167. package/dist/output/text.js +2 -0
  168. package/dist/registry/factory.js +4 -19
  169. package/dist/registry/network.js +66 -220
  170. package/dist/registry/providers/index.js +0 -2
  171. package/dist/registry/providers/skills-sh.js +3 -14
  172. package/dist/registry/providers/static-index.js +24 -26
  173. package/dist/registry/resolve.js +55 -131
  174. package/dist/scripts/akm-migrate-node.js +43937 -93313
  175. package/dist/scripts/akm-migrate.js +43697 -93071
  176. package/dist/setup/registry-stash-loader.js +4 -13
  177. package/dist/setup/semantic-assets.js +3 -44
  178. package/dist/setup/setup.js +1 -1
  179. package/dist/setup/steps/tasks.js +25 -15
  180. package/dist/sources/provider-factory.js +17 -18
  181. package/dist/sources/providers/filesystem.js +2 -3
  182. package/dist/sources/providers/git-install.js +7 -1
  183. package/dist/sources/providers/git-provider.js +0 -3
  184. package/dist/sources/providers/git-stash.js +0 -17
  185. package/dist/sources/providers/npm.js +2 -4
  186. package/dist/sources/providers/provider-utils.js +5 -10
  187. package/dist/sources/providers/website.js +0 -2
  188. package/dist/sources/snapshot-fetchers/website-ingest.js +1 -1
  189. package/dist/sources/website-url.js +2 -2
  190. package/dist/storage/database.js +9 -35
  191. package/dist/storage/repositories/improve-ledger-repository.js +168 -0
  192. package/dist/storage/repositories/index-connection.js +34 -70
  193. package/dist/storage/repositories/index-entries-repository.js +69 -111
  194. package/dist/storage/repositories/index-entry-mapper.js +1 -2
  195. package/dist/storage/repositories/index-entry-schema.js +83 -269
  196. package/dist/storage/repositories/index-fts-repository.js +86 -256
  197. package/dist/storage/repositories/index-llm-cache-repository.js +17 -0
  198. package/dist/storage/repositories/index-meta-repository.js +6 -4
  199. package/dist/storage/repositories/index-schema.js +192 -220
  200. package/dist/storage/repositories/index-utility-repository.js +8 -29
  201. package/dist/storage/repositories/index-vec-repository.js +133 -414
  202. package/dist/storage/repositories/outcome-repository.js +2 -1
  203. package/dist/storage/repositories/proposals-repository.js +35 -0
  204. package/dist/storage/repositories/registry-index-cache-repository.js +100 -0
  205. package/dist/storage/repositories/task-history-repository.js +26 -4
  206. package/dist/storage/repositories/workflow-runs-repository.js +53 -244
  207. package/dist/storage/sqlite-migrations.js +136 -0
  208. package/dist/storage/sqlite-pragmas.js +11 -9
  209. package/dist/storage/sqlite-transaction.js +170 -0
  210. package/dist/storage/state-db-integrity.js +34 -27
  211. package/dist/tasks/activation-config.js +134 -62
  212. package/dist/tasks/backends/cron.js +129 -277
  213. package/dist/tasks/backends/exec-utils.js +2 -5
  214. package/dist/tasks/backends/launchd.js +125 -745
  215. package/dist/tasks/backends/schtasks.js +101 -620
  216. package/dist/tasks/prepare/prepare-support.js +5 -15
  217. package/dist/tasks/prepare/prepare.js +0 -2
  218. package/dist/tasks/resolve-akm-bin.js +20 -79
  219. package/dist/tasks/run/attempt-lifecycle.js +0 -1
  220. package/dist/tasks/scheduler-binding.js +18 -238
  221. package/dist/tasks/scheduler-invocation.js +52 -52
  222. package/dist/tasks/scheduler-lock.js +53 -0
  223. package/dist/tasks/scheduler-sync.js +363 -679
  224. package/dist/tasks/source/parse-task-source.js +160 -10
  225. package/dist/tasks/source/task-source-v3-frozen.js +3 -4
  226. package/dist/tasks/source/task-to-v4.js +2 -2
  227. package/dist/workflows/authoring/authoring.js +3 -12
  228. package/dist/workflows/compile.js +211 -0
  229. package/dist/workflows/concurrency-policy.js +13 -74
  230. package/dist/workflows/exec/child-invocation.js +3 -17
  231. package/dist/workflows/exec/child-workflow.js +32 -141
  232. package/dist/workflows/exec/dispatch-redaction.js +13 -53
  233. package/dist/workflows/exec/environment.js +98 -0
  234. package/dist/workflows/exec/exec-unit.js +33 -140
  235. package/dist/workflows/exec/frozen-judge.js +7 -59
  236. package/dist/workflows/exec/native-executor.js +82 -341
  237. package/dist/workflows/exec/param-secrets.js +29 -47
  238. package/dist/workflows/exec/run-workflow.js +154 -387
  239. package/dist/workflows/exec/scheduler.js +9 -36
  240. package/dist/workflows/exec/step-work.js +127 -430
  241. package/dist/workflows/exec/unit-dispatch.js +11 -63
  242. package/dist/workflows/exec/unit-writer.js +8 -52
  243. package/dist/workflows/exec/worktree.js +39 -273
  244. package/dist/workflows/freeze/child-output-references.js +4 -15
  245. package/dist/workflows/freeze/environment.js +99 -92
  246. package/dist/workflows/freeze/freeze.js +172 -0
  247. package/dist/workflows/freeze/step-values.js +19 -21
  248. package/dist/workflows/freeze/targets/child-workflow.js +23 -92
  249. package/dist/workflows/freeze/targets/command.js +10 -33
  250. package/dist/workflows/freeze/targets/script.js +5 -12
  251. package/dist/workflows/freeze/targets/shell.js +3 -6
  252. package/dist/workflows/freeze/targets/task.js +25 -80
  253. package/dist/workflows/freeze/task-bindings.js +20 -67
  254. package/dist/workflows/{source-ir/github-yaml.js → github-yaml.js} +88 -206
  255. package/dist/workflows/ir/params.js +6 -51
  256. package/dist/workflows/ir/plan-hash.js +2 -34
  257. package/dist/workflows/parser.js +140 -43
  258. package/dist/{commands/improve/consolidate/types.js → workflows/plan.js} +2 -1
  259. package/dist/workflows/renderer.js +36 -69
  260. package/dist/workflows/resource-limits.js +12 -120
  261. package/dist/workflows/runtime/agent-identity.js +8 -40
  262. package/dist/workflows/runtime/run-outputs.js +3 -6
  263. package/dist/workflows/runtime/run-plan.js +316 -0
  264. package/dist/workflows/runtime/runs.js +48 -200
  265. package/dist/workflows/runtime/workflow-asset-loader.js +24 -57
  266. package/dist/workflows/{source-ir/semantics.js → source-semantics.js} +16 -20
  267. package/dist/workflows/validate-summary.js +2 -7
  268. package/docs/integration/bundling-akm.md +49 -42
  269. package/docs/migration/README.md +1 -0
  270. package/docs/migration/release-notes/0.9.17.md +41 -0
  271. package/docs/migration/v0.9.1-to-v0.9.2.md +19 -7
  272. package/docs/reference/cli.md +182 -125
  273. package/docs/reference/configuration.md +49 -56
  274. package/docs/reference/data-and-telemetry.md +19 -20
  275. package/docs/reference/tasks.md +86 -38
  276. package/docs/reference/workflow-schema.md +14 -18
  277. package/docs/reference/workflows.md +6 -9
  278. package/package.json +1 -1
  279. package/schemas/akm-config.json +87 -406
  280. package/dist/commands/health/advisories.js +0 -150
  281. package/dist/commands/health/metrics.js +0 -329
  282. package/dist/commands/health/surfaces.js +0 -102
  283. package/dist/commands/improve/anti-collapse.js +0 -83
  284. package/dist/commands/improve/collapse-detector.js +0 -432
  285. package/dist/commands/improve/consolidate/eligibility.js +0 -48
  286. package/dist/commands/improve/consolidate/merge.js +0 -146
  287. package/dist/commands/improve/distill/promote-memory.js +0 -329
  288. package/dist/commands/improve/distill/quality-gate.js +0 -500
  289. package/dist/commands/improve/memory/memory-contradiction-detect.js +0 -291
  290. package/dist/commands/improve/proposal-envelope.js +0 -31
  291. package/dist/commands/improve/run-context.js +0 -123
  292. package/dist/commands/improve/shared.js +0 -21
  293. package/dist/commands/improve/source-identity.js +0 -28
  294. package/dist/commands/improve/triage.js +0 -96
  295. package/dist/commands/proposal/drain-policies.js +0 -151
  296. package/dist/commands/sources/update-transaction.js +0 -220
  297. package/dist/core/action-contributors.js +0 -28
  298. package/dist/core/config/config-version-shim.js +0 -101
  299. package/dist/core/config/retired-experimental-keys-shim.js +0 -62
  300. package/dist/core/fs-txn.js +0 -405
  301. package/dist/core/lexical-score.js +0 -25
  302. package/dist/core/maintenance-barrier.js +0 -167
  303. package/dist/execution/executable-identity.js +0 -105
  304. package/dist/execution/guarded-source.js +0 -427
  305. package/dist/indexer/graph/graph-boost.js +0 -427
  306. package/dist/indexer/graph/graph-dedup.js +0 -95
  307. package/dist/indexer/search/name-match.js +0 -35
  308. package/dist/indexer/search/ranking-contributors.js +0 -515
  309. package/dist/indexer/walk/project-context.js +0 -192
  310. package/dist/integrations/agent/execution-cascade.js +0 -566
  311. package/dist/integrations/agent/execution-definitions.js +0 -202
  312. package/dist/integrations/agent/execution-lowering.js +0 -841
  313. package/dist/integrations/agent/execution-preparation.js +0 -98
  314. package/dist/integrations/agent/inline-execution.js +0 -74
  315. package/dist/registry/create-provider-registry.js +0 -29
  316. package/dist/registry/pinned-request-helper.js +0 -247
  317. package/dist/registry/pinned-transport.js +0 -717
  318. package/dist/sources/providers/index.js +0 -14
  319. package/dist/storage/engines/sqlite-migrations.js +0 -271
  320. package/dist/storage/repositories/canaries-repository.js +0 -107
  321. package/dist/storage/repositories/embedding-salvage-repository.js +0 -184
  322. package/dist/storage/repositories/registry-cache.js +0 -113
  323. package/dist/tasks/scheduler-sync-preview.js +0 -52
  324. package/dist/workflows/freeze/resolve-steps.js +0 -86
  325. package/dist/workflows/freeze/source-freeze.js +0 -64
  326. package/dist/workflows/ir/compile.js +0 -321
  327. package/dist/workflows/ir/environment-v4.js +0 -330
  328. package/dist/workflows/ir/freeze-v4.js +0 -153
  329. package/dist/workflows/ir/schema-v4.js +0 -745
  330. package/dist/workflows/ir/schema.js +0 -354
  331. package/dist/workflows/program/schema.js +0 -78
  332. package/dist/workflows/runtime/checkin.js +0 -57
  333. package/dist/workflows/runtime/plan-classifier.js +0 -196
  334. package/dist/workflows/runtime/unit-checkin.js +0 -45
  335. package/dist/workflows/runtime/unit-phases.js +0 -20
  336. package/dist/workflows/schema.js +0 -4
  337. package/dist/workflows/source-ir/compile.js +0 -200
  338. package/dist/workflows/source-ir/program.js +0 -50
  339. package/dist/workflows/source-ir/result.js +0 -26
  340. package/dist/workflows/source-ir/schema.js +0 -786
  341. package/dist/workflows/source-ir/triggers.js +0 -79
  342. package/dist/workflows/source-ir/uses.js +0 -40
  343. package/dist/workflows/validator.js +0 -60
@@ -1,13 +1,14 @@
1
1
  // This Source Code Form is subject to the terms of the Mozilla Public
2
2
  // License, v. 2.0. If a copy of the MPL was not distributed with this
3
3
  // file, You can obtain one at https://mozilla.org/MPL/2.0/.
4
+ /** The improve loop (reflect + distill per ref), the post-loop checks and the maintenance passes. */
4
5
  import fs from "node:fs";
5
6
  import path from "node:path";
6
7
  import { parseRefInput } from "../../core/asset/resolve-ref.js";
7
8
  import { daysToMs } from "../../core/common.js";
8
- import { DEFAULT_GRAPH_EXTRACTION_BATCH_SIZE, loadConfig } from "../../core/config/config.js";
9
+ import { DEFAULT_GRAPH_EXTRACTION_BATCH_SIZE, loadConfig, } from "../../core/config/config.js";
9
10
  import { UsageError } from "../../core/errors.js";
10
- import { appendEvent, readEvents } from "../../core/events.js";
11
+ import { appendEvent } from "../../core/events.js";
11
12
  import { openLogsDatabase, purgeOldTaskLogs } from "../../core/logs-db.js";
12
13
  import { getDbPath, getTaskLogDir } from "../../core/paths.js";
13
14
  import { withStateDb } from "../../core/state-db.js";
@@ -18,100 +19,81 @@ import { deriveWritableBundleIds } from "../../indexer/installations.js";
18
19
  import { collectPendingMemories, runMemoryInferencePass, } from "../../indexer/passes/memory-inference.js";
19
20
  import { resolveSourceEntries } from "../../indexer/search/search-source.js";
20
21
  import { isProcessEnabled } from "../../llm/feature-gate.js";
21
- import { withLlmStage } from "../../llm/usage-telemetry.js";
22
- import { purgeOldCycleMetrics } from "../../storage/repositories/canaries-repository.js";
23
22
  import { purgeOldEvents } from "../../storage/repositories/events-repository.js";
24
23
  import { purgeOldImproveRuns } from "../../storage/repositories/improve-runs-repository.js";
25
24
  import { closeDatabase, openIndexDatabase } from "../../storage/repositories/index-connection.js";
26
25
  import { getLiveRefSnapshot, isRefLiveInSnapshot, } from "../../storage/repositories/index-entries-repository.js";
27
26
  import { clearAssetOutcomeMissing, countAssetOutcomeMissing, deleteAssetOutcomeMissingBefore, listAssetOutcomeMissingState, stampAssetOutcomeMissing, } from "../../storage/repositories/outcome-repository.js";
28
27
  import { clearAssetSalienceMissing, countAssetSalienceMissing, deleteAssetSalienceMissingBefore, listAssetSalienceMissingState, stampAssetSalienceMissing, } from "../../storage/repositories/salience-repository.js";
29
- import { readFreelistInfo, vacuumStateDbIfReclaimable } from "../../storage/state-db-integrity.js";
28
+ import { readFreelistInfo, STATE_DB_VACUUMED_EVENT, vacuumIfReclaimable } from "../../storage/state-db-integrity.js";
30
29
  import { purgeOldTaskLogFiles } from "../../tasks/run/task-log.js";
31
- import { checkProposalGuard, expireStaleProposals, listProposals, purgeOrphanProposals } from "../proposal/repository.js";
30
+ import { expireStaleProposals, purgeOrphanProposals } from "../proposal/repository.js";
32
31
  import { checkDeadUrls } from "../url-checker.js";
33
- import { DEFAULT_RETENTION_DAYS as CYCLE_METRICS_RETENTION_DAYS, runCollapseDetector } from "./collapse-detector.js";
34
- import { defaultLookup, deriveLessonRef } from "./distill.js";
35
- import { wouldPromoteMemoryToKnowledge } from "./distill/promote-memory.js";
36
- import { deriveKnowledgeRef } from "./distill-promotion-policy.js";
37
- // Eligibility / candidate-selection predicates live in ./eligibility.
38
32
  import { findAssetFilePath, isDistillCandidateRef } from "./eligibility.js";
39
33
  import { shouldSkipRef } from "./improve-strategies.js";
40
- import { readOnlyEventsContext } from "./reflect.js";
34
+ import { recordLedgerAttempt, stateKey, stripBundle } from "./ledger.js";
35
+ import { pushRecentError } from "./preparation.js";
41
36
  import { recordNoOp, resetConsecutiveNoOps } from "./salience.js";
42
- import { errMessage } from "./shared.js";
43
- import { bareImproveRef, durableImproveRef } from "./source-identity.js";
44
- // ── improve loop / post-loop / maintenance stages ───────────────────
45
- // The cycle stages run by akmImprove, extracted from improve.ts.
46
- /** O-5 / #378: rolling per-originator error-window cap. */
47
- const RECENT_ERRORS_CAP = 3;
48
- /** O-5 / #378: push a per-originator error into the rolling window. */
49
- function pushRecentError(recentErrors, originator, msg) {
50
- if (!recentErrors[originator])
51
- recentErrors[originator] = [];
52
- recentErrors[originator].push(msg);
53
- if (recentErrors[originator].length > RECENT_ERRORS_CAP)
54
- recentErrors[originator].shift();
55
- }
56
- /**
57
- * Build the per-run loop environment from the run context: the derived guards
58
- * and the pending-proposal preload.
59
- */
37
+ import { attributeStage, errMessage } from "./stage.js";
60
38
  export function prepareImproveLoopEnv(args) {
61
- const { ctx, scope, options, reflectFn, distillFn, loopRefs, signalBearingSet, distillCooledRefs, distillOnlyRefs, recentErrors, rejectedProposalsByRef, startMs, budgetMs, improveProfile, resolvedPlan, } = args;
62
- // WI-9.10: the legacy dual context's optional eventsCtx/budgetSignal are
63
- // now RunContext's required eventsCtx / optional signal — renamed local
64
- // aliases so the rest of this function (and the ImproveLoopEnv object
65
- // literal below) is unchanged. `primaryStashDir` deliberately comes from
66
- // the state's honest optional field, NOT `ctx.stashDir`: the rare
67
- // unresolvable-primary path must keep skipping the `if (primaryStashDir)`
68
- // guards below (see the field's doc in ./improve-run-types).
69
- const primaryStashDir = args.primaryStashDir;
70
- const eventsCtx = ctx.eventsCtx;
71
- const budgetSignal = ctx.signal;
72
- // O-1 (#364): compute remaining budget at call time so each sub-call
73
- // receives only its fair share of the wall-clock budget.
74
- const remainingBudgetMs = () => Math.max(0, budgetMs - (Date.now() - startMs));
75
- // Build a Set for O(1) membership test — these refs skip the reflect call (Bug D2).
76
- const distillOnlyRefSet = new Set(distillOnlyRefs.map((r) => r.ref));
77
- // requirePlannedRefs guard: when the distill profile sets this flag, skip
78
- // distill for distill-only refs if the reflect phase produced no planned refs.
79
- // Prevents the distill loop from generating hundreds of distill-skipped events
80
- // on quiet passes (all refs on reflect cooldown, no new signal to distill).
81
- const requirePlannedRefs = improveProfile?.processes?.distill?.requirePlannedRefs === true;
82
- const hasReflectEligibleRefs = loopRefs.some((r) => !distillOnlyRefSet.has(r.ref));
83
- const skipDistillDueToRequirePlannedRefs = requirePlannedRefs && !hasReflectEligibleRefs;
84
- // Pre-load all pending proposals once instead of querying per asset in the loop.
85
- const dedupeStashDirForProposals = primaryStashDir ?? options.stashDir;
86
- const pendingProposalRefSet = new Set(dedupeStashDirForProposals
87
- ? listProposals(dedupeStashDirForProposals, { status: "pending" }).map((p) => p.ref)
88
- : []);
39
+ const distillOnlyRefSet = new Set(args.distillOnlyRefs.map((r) => r.ref));
40
+ const requirePlannedRefs = args.improveProfile?.processes?.distill?.requirePlannedRefs === true;
89
41
  return {
90
- scope,
91
- options,
92
- primaryStashDir,
93
- reflectFn,
94
- distillFn,
95
- signalBearingSet,
96
- distillCooledRefs,
42
+ scope: args.scope,
43
+ options: args.options,
44
+ primaryStashDir: args.primaryStashDir,
45
+ reflectFn: args.reflectFn,
46
+ distillFn: args.distillFn,
47
+ signalBearingSet: args.signalBearingSet,
48
+ distillCooledRefs: args.distillCooledRefs,
97
49
  distillOnlyRefSet,
98
- recentErrors,
99
- rejectedProposalsByRef,
100
- eventsCtx,
101
- improveProfile,
102
- resolvedPlan,
103
- budgetSignal,
104
- skipDistillDueToRequirePlannedRefs,
105
- pendingProposalRefSet,
106
- remainingBudgetMs,
50
+ recentErrors: args.recentErrors,
51
+ eventsCtx: args.eventsCtx,
52
+ improveProfile: args.improveProfile,
53
+ resolvedPlan: args.resolvedPlan,
54
+ budgetSignal: args.budgetSignal,
55
+ skipDistillDueToRequirePlannedRefs: requirePlannedRefs && args.loopRefs.every((r) => distillOnlyRefSet.has(r.ref)),
56
+ remainingBudgetMs: () => Math.max(0, args.budgetMs - (Date.now() - args.startMs)),
107
57
  };
108
58
  }
109
59
  /**
110
- * One improve-loop iteration for a single planned ref: the reflect pass, then
111
- * the distill pass, with the per-ref error classification (B7) around both.
112
- * `continue` in the old inline loop body is an early `return` inside the
113
- * passes; the orchestrator folds the returned tally and owns the run counters.
60
+ * Record a loop attempt in the improve ledger. Proposals and quality
61
+ * rejections record themselves; the loop records what they cannot see.
114
62
  */
63
+ function recordLoopAttempt(planned, env, source, outcome, detail) {
64
+ const stashDir = env.primaryStashDir ?? env.options.stashDir;
65
+ if (!stashDir || env.options.dryRun)
66
+ return;
67
+ recordLedgerAttempt({ eventsCtx: env.eventsCtx }, {
68
+ stashDir,
69
+ ref: stateKey(planned.ref, planned.itemRef),
70
+ source,
71
+ outcome,
72
+ ...(detail !== undefined ? { detail } : {}),
73
+ });
74
+ }
75
+ /** Plasticity counter: repeated no-ops dampen an asset's selection score; a change lifts it. */
76
+ function recordPlasticity(env, planned, outcome) {
77
+ const db = env.eventsCtx?.db;
78
+ if (!db || !outcome)
79
+ return;
80
+ const key = stateKey(planned.ref, planned.itemRef);
81
+ try {
82
+ if (outcome === "noop")
83
+ recordNoOp(db, key);
84
+ else
85
+ resetConsecutiveNoOps(db, key);
86
+ }
87
+ catch {
88
+ // best-effort
89
+ }
90
+ }
91
+ function recordSkip(tally, ref, reason, event) {
92
+ tally.actions.push({ ref, mode: "distill-skipped", result: { ok: true, reason } });
93
+ if (event)
94
+ appendEvent({ eventType: "improve_skipped", ref, metadata: { reason: event.reason } }, event.env.eventsCtx);
95
+ }
96
+ /** One loop iteration: reflect, then distill. A distill UsageError is a validation failure. */
115
97
  export async function processImproveLoopRef(planned, env) {
116
98
  const tally = {
117
99
  actions: [],
@@ -120,20 +102,15 @@ export async function processImproveLoopRef(planned, env) {
120
102
  memoryRefsForInference: [],
121
103
  };
122
104
  try {
123
- // Bug D2: distillOnlyRefs skip the reflect call but still run the distill path.
124
- // Bug D1: in-loop distill-cooldown check removed — distill-cooled candidates
125
- // have their synthetic actions emitted in runImprovePreparationStage.
126
105
  const isDistillOnly = env.distillOnlyRefSet.has(planned.ref);
127
- const parsedPlannedRef = parseRefInput(planned.ref);
128
- await runLoopReflectPass(planned, isDistillOnly, env, tally);
129
- // isDistillOnly refs: no reflect action emitted — proceed directly to the distill pass.
130
- await runLoopDistillPass(planned, parsedPlannedRef, isDistillOnly, env, tally);
106
+ const parsed = parseRefInput(planned.ref);
107
+ if (!isDistillOnly)
108
+ await runLoopReflectPass(planned, env, tally);
109
+ await runLoopDistillPass(planned, parsed.type, isDistillOnly, env, tally);
131
110
  }
132
111
  catch (err) {
133
- // B7: UsageError thrown by akmDistill on validation_failed should be recorded
134
- // as mode:"distill" with outcome:"validation_failed", NOT as a generic error.
135
- // The distill_invoked event was already emitted inside akmDistill before the throw.
136
112
  if (err instanceof UsageError) {
113
+ recordLoopAttempt(planned, env, "distill", "failed", err.message);
137
114
  tally.actions.push({
138
115
  ref: planned.ref,
139
116
  mode: "distill",
@@ -141,583 +118,177 @@ export async function processImproveLoopRef(planned, env) {
141
118
  });
142
119
  }
143
120
  else {
144
- tally.actions.push({
145
- ref: planned.ref,
146
- mode: "error",
147
- result: { ok: false, error: errMessage(err) },
148
- });
121
+ tally.actions.push({ ref: planned.ref, mode: "error", result: { ok: false, error: errMessage(err) } });
149
122
  }
150
123
  }
151
124
  return tally;
152
125
  }
153
- /**
154
- * Reflect half of one loop iteration: type/profile gates, the reflect call with
155
- * recent-error avoidPatterns, outcome classification (cooldown / guard-reject /
156
- * type-refused / noise-gate), and plasticity counters. Records onto the per-ref
157
- * tally only.
158
- */
159
- async function runLoopReflectPass(planned, isDistillOnly, env, tally) {
160
- const { options, primaryStashDir, reflectFn, eventsCtx, improveProfile, resolvedPlan, budgetSignal } = env;
161
- // B6: derived memories are machine-generated; skip reflect to avoid noisy proposals.
162
- // shouldDistillMemoryRef already returns false for .derived refs, so the distill
163
- // path is also a no-op for them — we just avoid unnecessary agent spawns.
164
- // D2: distillOnlyRefs also skip the reflect call (reflect-cooled, distill path only).
165
- if (!isDistillOnly && !planned.ref.endsWith(".derived")) {
166
- // Type guard: skip reflect for unsupported types (script, env, task, etc.)
167
- // and raw wiki directories, driven by the active improve profile.
168
- const reflectSkip = shouldSkipRef(planned.ref, "reflect", improveProfile);
169
- if (reflectSkip.skip) {
170
- tally.actions.push({
171
- ref: planned.ref,
172
- mode: "reflect-skipped",
173
- result: { ok: true, reason: reflectSkip.reason },
174
- });
175
- }
176
- else {
177
- // O-5 / #378: only inject reflect-originator errors into the reflect call.
178
- // Cross-task errors (e.g. schema-repair) must NOT contaminate reflect prompts.
179
- const reflectErrors = env.recentErrors.reflect ?? [];
180
- if (reflectErrors.length > 0)
181
- tally.reflectsWithErrorContext++;
182
- // O-1 (#364): pass remaining budget as timeoutMs so the agent spawn is
183
- // bounded by the wall-clock deadline rather than the default per-profile timeout.
184
- const reflectBudgetMs = env.remainingBudgetMs();
185
- const reflectEngine = resolvedPlan.processes.reflect.runner?.engine;
186
- const reflectTarget = options.sourceName && primaryStashDir ? { source: options.sourceName, root: primaryStashDir } : undefined;
187
- // Re-enter canonical named-engine lowering with the config snapshot and
188
- // process profile frozen into the invocation plan. The loop never injects
189
- // a RunnerSpec seam or observes later caller-config mutations.
190
- const reflectCallArgs = {
191
- ref: planned.ref,
192
- // Carry the resolved item_ref so reflect uses the same durable key for
193
- // events and state reads.
194
- ...(planned.itemRef ? { itemRef: planned.itemRef } : {}),
195
- task: options.task,
196
- // Active strategy supplies non-engine process tuning.
197
- ...(improveProfile ? { improveProfile } : {}),
198
- config: resolvedPlan.config,
199
- ...(primaryStashDir ? { stashDir: primaryStashDir } : {}),
200
- ...(reflectTarget ? { target: reflectTarget } : {}),
201
- ...(reflectErrors.length > 0 ? { avoidPatterns: [...reflectErrors] } : {}),
202
- eventSource: "improve",
203
- // #639 — resolve the low-value filter from the ACTIVE improve profile
204
- // (default off when unset), so the running strategy decides.
205
- lowValueFilter: improveProfile.processes?.reflect?.lowValueFilter?.enabled === true,
206
- ...(reflectBudgetMs > 0 ? { timeoutMs: reflectBudgetMs } : {}),
207
- signal: budgetSignal,
208
- // R25: reflect's event emits reuse the run's long-lived state.db handle.
209
- eventsCtx: env.eventsCtx,
210
- // Attribution: carry the eligibility lane so reflect stamps it on
211
- // the reflect_invoked event and the persisted proposal.
212
- ...(planned.eligibilitySource ? { eligibilitySource: planned.eligibilitySource } : {}),
213
- };
214
- // R9 (tier2-0917): the fingerprint/rejection-backoff guard `createProposal`
215
- // runs AFTER reflect's ~39s generation + judge is computable from inputs
216
- // available before dispatch. Check it here first — on a hit, skip the LLM
217
- // call entirely and synthesize the same "cooldown" envelope reflect.ts's
218
- // createProposal-skip branch returns, so everything below (mode
219
- // classification, improve_reflect_outcome, plasticity) is unchanged. This
220
- // pre-check is an optimisation only — createProposal's post-generation
221
- // check stays authoritative (see checkProposalGuard's doc comment).
222
- const guardStash = primaryStashDir ?? options.stashDir;
223
- const guardSkip = guardStash
224
- ? checkProposalGuard({
225
- stash: guardStash,
226
- ref: planned.ref,
227
- source: "reflect",
228
- ...(reflectTarget ? { target: reflectTarget } : {}),
229
- ...(reflectEngine ? { modelId: reflectEngine } : {}),
230
- })
231
- : undefined;
232
- let reflectResult;
233
- if (guardSkip) {
234
- // Mirror reflect.ts's buildReflectEventEmitters().emitInvoked(): the
235
- // signal-delta cursor (buildLatestProposalTsMap) reads `reflect_invoked`
236
- // events regardless of outcome, so it must still advance for this ref
237
- // even though reflectFn was never called.
238
- appendEvent({
239
- eventType: "reflect_invoked",
240
- ref: planned.itemRef ?? durableImproveRef(planned.ref),
241
- metadata: {
242
- ...(options.task ? { task: options.task } : {}),
243
- ...(reflectEngine ? { engine: reflectEngine } : {}),
244
- ...(planned.eligibilitySource ? { eligibilitySource: planned.eligibilitySource } : {}),
245
- },
246
- }, eventsCtx);
247
- // Mirror reflect.ts's buildReflectEventEmitters().emitFailed(): every
248
- // reflect_invoked must be paired with a reflect_completed so observers
249
- // building closed-loop telemetry see balanced invoke/complete pairs.
250
- // reflectFn is never called on this path, so reflect.ts's own
251
- // emitFailed (fired from its post-generation cooldown branch) never
252
- // runs either — this is the pre-generation guard's own pairing.
253
- appendEvent({
254
- eventType: "reflect_completed",
255
- ref: planned.itemRef ?? durableImproveRef(planned.ref),
256
- metadata: {
257
- source: "reflect",
258
- ok: false,
259
- reason: "cooldown",
260
- subreason: "pre_generation_guard",
261
- proposalSkipReason: guardSkip.reason,
262
- ...(guardSkip.existingProposalId ? { existingProposalId: guardSkip.existingProposalId } : {}),
263
- },
264
- }, eventsCtx);
265
- reflectResult = {
266
- schemaVersion: 2,
267
- ok: false,
268
- reason: "cooldown",
269
- error: `Proposal skipped (${guardSkip.reason}): ${guardSkip.message}`,
270
- ref: planned.ref,
271
- ...(reflectEngine ? { engine: reflectEngine } : {}),
272
- exitCode: null,
273
- };
274
- }
275
- else {
276
- reflectResult = await withLlmStage("reflect", () => reflectFn(reflectCallArgs), {
277
- engine: reflectEngine,
278
- process: "reflect",
279
- });
280
- }
281
- const isCooldown = !reflectResult.ok && reflectResult.reason === "cooldown";
282
- // Content-policy guard hits (reflect size-rail rejections) are NOT
283
- // LLM faults — the agent responded fine, the downstream guard
284
- // blocked the output. Route them to a distinct `reflect-guard-rejected`
285
- // mode so health metrics can split deterministic guard hits out of
286
- // true LLM failures. See
287
- // `/tmp/akm-health-investigations/metrics-taxonomy-review.md` §1a.
288
- const isGuardReject = !reflectResult.ok && reflectResult.reason === "content_policy_reject";
289
- // Type-guard rejection (reflect refused a script/env/task ref) is
290
- // also NOT an LLM failure — the LLM is never invoked. Route to the
291
- // existing `reflect-skipped` bucket so it does not inflate the
292
- // failure-rate numerator. ~9% of `reflect-failed` events in the
293
- // user's stack were this case; see review §1a row "Reflect refused
294
- // asset type".
295
- const isTypeRefused = !reflectResult.ok && reflectResult.reason === "unsupported_type";
296
- // Noise-gate suppression (#580): the candidate edit was an empty
297
- // diff or a cosmetic-only reformat of the current asset. Like
298
- // `unsupported_type`, this is a deterministic skip — not an LLM
299
- // fault — so it routes to the `reflect-skipped` bucket and stays
300
- // out of recentErrors/avoidPatterns.
301
- const isNoChange = !reflectResult.ok && reflectResult.reason === "no_change";
302
- // Quality-gate rejection (R3): the judge rejected an otherwise
303
- // well-parsed proposal. Stays in the `reflect-failed` bucket for
304
- // metrics continuity, but — like the deterministic skips above —
305
- // must not be injected into recentErrors/avoidPatterns: the judge's
306
- // rejection text is not a reusable "avoid this pattern" lesson, and
307
- // feeding it back in poisoned later prompts in the same run.
308
- const isQualityRejected = !reflectResult.ok && reflectResult.reason === "quality_rejected";
309
- tally.actions.push({
310
- ref: planned.ref,
311
- mode: reflectResult.ok
312
- ? "reflect"
313
- : isCooldown
314
- ? "reflect-cooldown"
315
- : isGuardReject
316
- ? "reflect-guard-rejected"
317
- : isTypeRefused || isNoChange
318
- ? "reflect-skipped"
319
- : "reflect-failed",
320
- result: reflectResult,
321
- });
322
- // Cooldown skips, guard rejects, type-refused skips, noise-gate
323
- // skips, and quality-gate rejections are not failures — do not
324
- // pollute recentErrors with them (those get injected as
325
- // `avoidPatterns` into the next reflect prompt). Guard rejects ARE
326
- // worth showing the LLM as a learn-signal so the next iteration sees
327
- // "your last expansion was too large"; type-refused, no-change, and
328
- // quality-rejected are deterministic/judge-side and add no learning
329
- // signal.
330
- if (!reflectResult.ok && !isCooldown && !isTypeRefused && !isNoChange && !isQualityRejected) {
331
- const errMsg = reflectResult.error ?? reflectResult.reason ?? "unknown reflect error";
332
- tally.recentErrorPushes.push({ originator: "reflect", message: errMsg });
333
- }
334
- // improve_reflect_outcome — per-asset metric for tuning the reflect path.
335
- appendEvent({
336
- eventType: "improve_reflect_outcome",
337
- ref: planned.ref,
338
- metadata: {
339
- ok: reflectResult.ok,
340
- durationMs: reflectResult.ok ? reflectResult.durationMs : undefined,
341
- engine: reflectResult.engine,
342
- reason: reflectResult.ok ? undefined : reflectResult.reason,
343
- },
344
- }, eventsCtx);
345
- // Plasticity counter (plan §WS-1 step 8): record no-ops so the
346
- // WS-1 selection comparator (effectiveScore, ~line 3073) can dampen
347
- // repeatedly-silent assets during consolidation-selection.
348
- // A no_change reflect means the LLM was invoked but found nothing to
349
- // improve — the asset is stable. Track it. A successful reflect means
350
- // the asset changed; reset the counter so the dampener lifts.
351
- // Use the same item_ref-or-conceptId salience key as preparation/distill.
352
- const plasticityKey = planned.itemRef ?? durableImproveRef(planned.ref);
353
- if (isNoChange && eventsCtx?.db) {
354
- try {
355
- recordNoOp(eventsCtx.db, plasticityKey);
356
- }
357
- catch {
358
- // best-effort: plasticity counter failure never blocks the run
359
- }
360
- }
361
- else if (reflectResult.ok && eventsCtx?.db) {
362
- try {
363
- resetConsecutiveNoOps(eventsCtx.db, plasticityKey);
364
- }
365
- catch {
366
- // best-effort
367
- }
368
- }
369
- } // end else (reflect type/profile check)
370
- }
371
- else if (!isDistillOnly && planned.ref.endsWith(".derived")) {
372
- // B6: .derived refs skip reflect; record synthetic skip action.
373
- tally.actions.push({
374
- ref: planned.ref,
375
- mode: "distill-skipped",
376
- result: { ok: true, reason: "derived-memory-reflect-skipped" },
377
- });
378
- appendEvent({
379
- eventType: "improve_skipped",
380
- ref: planned.ref,
381
- metadata: { reason: "derived_memory_reflect_skipped" },
382
- }, eventsCtx);
383
- }
384
- }
385
- /**
386
- * Distill half of one loop iteration: the profile / requirePlannedRefs /
387
- * candidate-type / weak-signal / cooldown gates, then the pending-proposal and
388
- * reject-grace dedup checks, then {@link invokeDistillAndRecord}. Each gate
389
- * that was a `continue` in the old inline loop body is an early `return` here.
390
- */
391
- async function runLoopDistillPass(planned, parsedPlannedRef, isDistillOnly, env, tally) {
392
- const { options, primaryStashDir, eventsCtx, improveProfile, resolvedPlan } = env;
393
- const hasRecentFeedbackSignal = env.signalBearingSet.has(planned.ref);
394
- const explicitRefScope = env.scope.mode === "ref";
395
- // Profile gate: apply the full type-filter / raw-wiki / disabled rules to
396
- // distill so callers who configure `profile.processes.distill.allowedTypes`
397
- // or land on raw-wiki refs get a recorded skip action instead of silently
398
- // proceeding.
399
- const distillSkip = shouldSkipRef(planned.ref, "distill", improveProfile);
400
- if (distillSkip.skip) {
401
- tally.actions.push({
402
- ref: planned.ref,
403
- mode: "distill-skipped",
404
- result: { ok: true, reason: distillSkip.reason },
405
- });
126
+ async function runLoopReflectPass(planned, env, tally) {
127
+ const { options, primaryStashDir, improveProfile, resolvedPlan } = env;
128
+ // Derived memories are machine-generated: never reflected.
129
+ if (planned.ref.endsWith(".derived")) {
130
+ recordSkip(tally, planned.ref, "derived-memory-reflect-skipped", { env, reason: "derived_memory_reflect_skipped" });
406
131
  return;
407
132
  }
408
- // requirePlannedRefs guard: skip distill for distill-only refs when no
409
- // reflect-eligible refs were planned this run, preventing mass skip events.
410
- if (env.skipDistillDueToRequirePlannedRefs && isDistillOnly) {
411
- tally.actions.push({
412
- ref: planned.ref,
413
- mode: "distill-skipped",
414
- result: { ok: true, reason: "require_planned_refs" },
415
- });
133
+ const reflectSkip = shouldSkipRef(planned.ref, "reflect", improveProfile);
134
+ if (reflectSkip.skip) {
135
+ tally.actions.push({ ref: planned.ref, mode: "reflect-skipped", result: { ok: true, reason: reflectSkip.reason } });
416
136
  return;
417
137
  }
418
- // See `isDistillCandidateRef` — excludes `lesson:*` (and anything else in
419
- // DISTILL_REFUSED_INPUT_TYPES) so distill never gets queued for an input
420
- // it will refuse.
421
- const shouldAttemptDistill = isDistillCandidateRef(planned.ref, options.stashDir);
422
- const skipMemoryDistillForWeakSignal = !isDistillOnly && parsedPlannedRef.type === "memory" && !hasRecentFeedbackSignal && !explicitRefScope;
423
- // distillCooledRefs guard: pre-filter emitted synthetic actions for distill-candidate
424
- // refs; non-candidate refs in the set are blocked here.
425
- // O-2 (#365): bypass the distill cooldown when the user explicitly targeted
426
- // this ref via --scope — their intent overrides unattended-run policies.
427
- if (shouldAttemptDistill &&
428
- !skipMemoryDistillForWeakSignal &&
429
- (!env.distillCooledRefs.has(planned.ref) || explicitRefScope)) {
430
- // TODO(refactor): single call site needs both lesson+knowledge refs for proposal dedup. If a third target ref type is added, extract deriveAllTargetRefs(inputRef): string[].
431
- const lessonRef = deriveLessonRef(planned.ref);
432
- const knowledgeRef = deriveKnowledgeRef(planned.ref);
433
- const dedupeStashDir = primaryStashDir ?? options.stashDir;
434
- if (dedupeStashDir) {
435
- // B2: check both lesson ref and knowledge ref since auto-promoted memories
436
- // create knowledge: proposals, not lesson: proposals.
437
- const hasExistingPending = env.pendingProposalRefSet.has(lessonRef) || env.pendingProposalRefSet.has(knowledgeRef);
438
- if (hasExistingPending) {
439
- tally.actions.push({
440
- ref: planned.ref,
441
- mode: "distill-skipped",
442
- result: { ok: true, reason: "pending proposal exists" },
443
- });
444
- appendEvent({
445
- eventType: "improve_skipped",
446
- ref: planned.ref,
447
- metadata: { reason: "pending_proposal_exists" },
448
- }, eventsCtx);
449
- return;
450
- }
451
- // D-2 (#370): reject-aware cooldown for distill. When the reviewer
452
- // recently rejected a distilled lesson or knowledge proposal for this
453
- // asset, skip re-distillation for a 1-day grace window. Prevents the
454
- // same rejected proposal from being regenerated immediately. The
455
- // window is fixed (the 0.8.0 redesign moved per-ref cooldowns to
456
- // signal-delta gates and dropped --distill-cooldown-days; a short
457
- // reject grace is preserved here so a fresh rejection isn't
458
- // overridden by the same run).
459
- // References: ExpeL arXiv:2308.10144, STaR arXiv:2203.14465.
460
- const DISTILL_REJECT_COOLDOWN_MS = daysToMs(1);
461
- const recentlyRejectedLesson = !explicitRefScope && // O-2: bypass when --scope <ref> is explicit
462
- (env.rejectedProposalsByRef.has(lessonRef) || env.rejectedProposalsByRef.has(knowledgeRef));
463
- if (recentlyRejectedLesson) {
464
- const rejectedEntry = env.rejectedProposalsByRef.get(lessonRef) ?? env.rejectedProposalsByRef.get(knowledgeRef);
465
- const rejectedAgeMs = rejectedEntry ? Date.now() - new Date(rejectedEntry.ts).getTime() : 0;
466
- if (rejectedAgeMs < DISTILL_REJECT_COOLDOWN_MS) {
467
- tally.actions.push({
468
- ref: planned.ref,
469
- mode: "distill-skipped",
470
- result: { ok: true, reason: "distill reject grace window" },
471
- });
472
- appendEvent({
473
- eventType: "improve_skipped",
474
- ref: planned.ref,
475
- metadata: {
476
- reason: "distill_reject_grace_window",
477
- },
478
- }, eventsCtx);
479
- return;
480
- }
481
- }
482
- // R9 extension (r2-6, tier2-0917; PRECHECK, tier3-0917): the
483
- // fingerprint/rejection-backoff guard `createProposal` runs AFTER
484
- // distill's ~generation + judge is computable from inputs available
485
- // before dispatch — mirror the reflect pre-check above so a guard hit
486
- // skips the LLM call entirely. Distill's real `createProposal` call
487
- // always targets the derived lesson/knowledge ref (`effectiveLessonRef`
488
- // in distill.ts), never the input ref.
489
- // Which ref that is: for every non-memory distill-candidate type,
490
- // `targetKind` defaults to "lesson" (distill.ts ~L882, `invokeDistill
491
- // AndRecord` above only ever sets `proposalKind: "auto"` for memory
492
- // refs) and is never overridden to "knowledge", so lessonRef is the
493
- // ONLY real target. For memory refs (`proposalKind: "auto"`), the
494
- // target is decided at dispatch by `planMemoryKnowledgePromotion`
495
- // (knowledgeRef via promotion, lessonRef as fallback) — that decision
496
- // IS cheap and LLM-free (a deterministic score over the asset content
497
- // + its feedback history, plus one lookup for an existing knowledge
498
- // file), so it is pre-checked exactly via `wouldPromoteMemoryToKnowledge`,
499
- // a thin wrapper that delegates to `planMemoryKnowledgePromotion`
500
- // itself so this can never drift from distill's real decision. A
501
- // guard hit on the ref distill would NOT have targeted must never
502
- // suppress a legitimate dispatch.
503
- // §23.6 fingerprint model-id term: distill resolves models, not
504
- // engines (unlike reflect), so this must match `distillRunner?.
505
- // connection.model` in distill.ts, not the engine name.
506
- const distillModelId = resolvedPlan.processes.distill.runner?.connection.model;
507
- let realTargetRef = lessonRef;
508
- if (parsedPlannedRef.type === "memory") {
509
- // distill.ts's real dispatch (akmDistill) always derives
510
- // durableInputRef from options.ref alone (durableImproveRef(inputRef),
511
- // never itemRef) and reads/scores content via that ref
512
- // (loadAndScoreInputSalience's `lookup(durableInputRef)`); mirror
513
- // that here so the pre-check can never read/score a different file
514
- // than the real dispatch would. itemRef is preferred only for the
515
- // feedback-events query, matching readDistillFeedback's
516
- // `ref: options.itemRef ?? durableInputRef`.
517
- const durableInputRef = durableImproveRef(planned.ref);
518
- const feedbackRef = planned.itemRef ?? durableInputRef;
519
- const lookup = (ref) => defaultLookup(ref, dedupeStashDir);
520
- const filePath = await lookup(durableInputRef);
521
- const assetContent = filePath && fs.existsSync(filePath) ? fs.readFileSync(filePath, "utf8") : null;
522
- // PRECHECK (tier3-0917-r3, r3-4): reuse the loop's long-lived
523
- // eventsCtx.db handle when one is open, instead of opening a fresh
524
- // read-only state.db connection per memory ref (R25). Degrades to
525
- // the previous readOnly-open when no live handle is present (e.g.
526
- // this function invoked without a run-scoped eventsCtx), via the
527
- // same readOnlyEventsContext helper reflect.ts's read call sites use.
528
- const { events: feedbackEvents } = readEvents({ ref: feedbackRef, type: "feedback" }, readOnlyEventsContext(eventsCtx));
529
- const promotesToKnowledge = await wouldPromoteMemoryToKnowledge({
530
- inputRef: planned.ref,
531
- durableInputRef,
532
- assetContent,
533
- feedbackEvents,
534
- config: options.config ?? loadConfig(),
535
- stash: dedupeStashDir,
536
- lookup,
537
- });
538
- if (promotesToKnowledge)
539
- realTargetRef = knowledgeRef;
540
- }
541
- const guardSkip = checkProposalGuard({
542
- stash: dedupeStashDir,
543
- ref: realTargetRef,
544
- source: "distill",
545
- ...(distillModelId ? { modelId: distillModelId } : {}),
138
+ // Only reflect's own recent errors reach its prompt.
139
+ const reflectErrors = env.recentErrors.reflect ?? [];
140
+ if (reflectErrors.length > 0)
141
+ tally.reflectsWithErrorContext++;
142
+ const budgetMs = env.remainingBudgetMs();
143
+ const reflectArgs = {
144
+ ref: planned.ref,
145
+ ...(planned.itemRef ? { itemRef: planned.itemRef } : {}),
146
+ task: options.task,
147
+ ...(improveProfile ? { improveProfile } : {}),
148
+ config: resolvedPlan.config,
149
+ ...(primaryStashDir ? { stashDir: primaryStashDir } : {}),
150
+ ...(options.sourceName && primaryStashDir ? { target: { source: options.sourceName, root: primaryStashDir } } : {}),
151
+ ...(reflectErrors.length > 0 ? { avoidPatterns: [...reflectErrors] } : {}),
152
+ eventSource: "improve",
153
+ lowValueFilter: improveProfile.processes?.reflect?.lowValueFilter?.enabled === true,
154
+ ...(budgetMs > 0 ? { timeoutMs: budgetMs } : {}),
155
+ signal: env.budgetSignal,
156
+ eventsCtx: env.eventsCtx,
157
+ ...(planned.eligibilitySource ? { eligibilitySource: planned.eligibilitySource } : {}),
158
+ };
159
+ const result = await attributeStage(resolvedPlan, "reflect", () => env.reflectFn(reflectArgs));
160
+ const reason = result.ok ? undefined : result.reason;
161
+ // A refused type or an unchanged asset is a deterministic skip, not an LLM
162
+ // fault; a guard rejection (size rail) gets its own bucket for health.
163
+ const skipped = reason === "unsupported_type" || reason === "no_change";
164
+ tally.actions.push({
165
+ ref: planned.ref,
166
+ mode: result.ok
167
+ ? "reflect"
168
+ : reason === "content_policy_reject"
169
+ ? "reflect-guard-rejected"
170
+ : skipped
171
+ ? "reflect-skipped"
172
+ : "reflect-failed",
173
+ result,
174
+ });
175
+ // A quality rejection recorded itself, and the judge's text is no lesson for
176
+ // the next prompt; skips revisit on the `unchanged` cadence.
177
+ if (!result.ok && reason !== "quality_rejected") {
178
+ recordLoopAttempt(planned, env, "reflect", skipped ? "unchanged" : "failed", reason);
179
+ if (!skipped) {
180
+ tally.recentErrorPushes.push({
181
+ originator: "reflect",
182
+ message: result.error ?? reason ?? "unknown reflect error",
546
183
  });
547
- if (guardSkip) {
548
- tally.actions.push({
549
- ref: planned.ref,
550
- mode: "distill-skipped",
551
- result: { ok: true, reason: guardSkip.reason },
552
- });
553
- // Mirror distill.ts's own proposal-skip branch (the post-generation
554
- // guard `createProposal` hits): emit `distill_invoked` with a
555
- // `skipped` outcome so the signal-delta cursor
556
- // (buildLatestProposalTsMap, eligibility.ts) advances for this ref
557
- // even though distillFn was never called.
558
- appendEvent({
559
- eventType: "distill_invoked",
560
- // Use item_ref when resolved, otherwise the input conceptId —
561
- // matches distill.ts's own distill_invoked key.
562
- ref: planned.itemRef ?? durableImproveRef(planned.ref),
563
- metadata: {
564
- outcome: "skipped",
565
- proposalRef: realTargetRef,
566
- message: guardSkip.message,
567
- skipReason: guardSkip.reason,
568
- ...(planned.eligibilitySource ? { eligibilitySource: planned.eligibilitySource } : {}),
569
- },
570
- }, eventsCtx);
571
- return;
572
- }
573
184
  }
574
- await invokeDistillAndRecord(planned, parsedPlannedRef, env, tally);
575
185
  }
576
- else if (skipMemoryDistillForWeakSignal) {
577
- tally.actions.push({
578
- ref: planned.ref,
579
- mode: "distill-skipped",
580
- result: { ok: true, reason: "memory requires recent feedback signal" },
186
+ appendEvent({
187
+ eventType: "improve_reflect_outcome",
188
+ ref: planned.ref,
189
+ metadata: {
190
+ ok: result.ok,
191
+ durationMs: result.ok ? result.durationMs : undefined,
192
+ engine: result.engine,
193
+ reason,
194
+ },
195
+ }, env.eventsCtx);
196
+ recordPlasticity(env, planned, reason === "no_change" ? "noop" : result.ok ? "changed" : undefined);
197
+ }
198
+ async function runLoopDistillPass(planned, refType, isDistillOnly, env, tally) {
199
+ const { options, primaryStashDir, improveProfile, resolvedPlan } = env;
200
+ const distillSkip = shouldSkipRef(planned.ref, "distill", improveProfile);
201
+ if (distillSkip.skip)
202
+ return recordSkip(tally, planned.ref, distillSkip.reason);
203
+ if (env.skipDistillDueToRequirePlannedRefs && isDistillOnly) {
204
+ return recordSkip(tally, planned.ref, "require_planned_refs");
205
+ }
206
+ const explicitRefScope = env.scope.mode === "ref";
207
+ const weakMemorySignal = !isDistillOnly && refType === "memory" && !env.signalBearingSet.has(planned.ref) && !explicitRefScope;
208
+ if (weakMemorySignal) {
209
+ return recordSkip(tally, planned.ref, "memory requires recent feedback signal", {
210
+ env,
211
+ reason: "memory_distill_requires_feedback",
581
212
  });
582
- appendEvent({
583
- eventType: "improve_skipped",
584
- ref: planned.ref,
585
- metadata: { reason: "memory_distill_requires_feedback" },
586
- }, eventsCtx);
587
213
  }
588
- }
589
- /**
590
- * The distill invocation for one ref that passed every gate: the `distillFn`
591
- * call, memory-inference queueing, and plasticity counters.
592
- */
593
- async function invokeDistillAndRecord(planned, parsedPlannedRef, env, tally) {
594
- const { options, primaryStashDir, distillFn, eventsCtx, improveProfile, resolvedPlan, budgetSignal } = env;
595
- const distillResult = await withLlmStage("distill", () => distillFn({
214
+ // The ledger holds cooled refs; an explicit `--scope` ref overrides it.
215
+ if (!isDistillCandidateRef(planned.ref, options.stashDir))
216
+ return;
217
+ if (env.distillCooledRefs.has(planned.ref) && !explicitRefScope)
218
+ return;
219
+ const result = await attributeStage(resolvedPlan, "distill", () => env.distillFn({
596
220
  ref: planned.ref,
597
- // Carry the resolved item_ref so distill matches preparation's state key.
598
221
  ...(planned.itemRef ? { itemRef: planned.itemRef } : {}),
599
- ...(parsedPlannedRef.type === "memory" ? { proposalKind: "auto" } : {}),
222
+ ...(refType === "memory" ? { proposalKind: "auto" } : {}),
600
223
  ...(primaryStashDir ? { stashDir: primaryStashDir } : {}),
601
- // Active profile so distill's per-process reads honor `--profile`.
602
224
  ...(improveProfile ? { improveProfile } : {}),
603
225
  config: options.config,
604
226
  llmRunner: resolvedPlan.processes.distill.runner,
605
- signal: budgetSignal,
606
- // R25: distill's event emits reuse the run's long-lived state.db handle.
607
- eventsCtx,
608
- // Attribution: carry the eligibility lane so distill stamps it on the
609
- // distill_invoked event and the persisted proposal.
227
+ signal: env.budgetSignal,
228
+ eventsCtx: env.eventsCtx,
610
229
  ...(planned.eligibilitySource ? { eligibilitySource: planned.eligibilitySource } : {}),
611
- }), { engine: resolvedPlan.processes.distill.runner?.engine, process: "distill" });
612
- tally.actions.push({ ref: planned.ref, mode: "distill", result: distillResult });
613
- if (parsedPlannedRef.type === "memory") {
614
- const promotedToKnowledge = distillResult.outcome === "queued" && distillResult.proposalKind === "knowledge";
615
- if (!promotedToKnowledge)
616
- tally.memoryRefsForInference.push(planned.ref);
230
+ }));
231
+ tally.actions.push({ ref: planned.ref, mode: "distill", result });
232
+ // `queued` and the quality outcomes recorded themselves; a transport failure
233
+ // or a disabled process is not an attempt, so the ref stays eligible.
234
+ if (result.outcome === "skipped") {
235
+ recordLoopAttempt(planned, env, "distill", "unchanged", result.skipReason ?? result.message);
617
236
  }
618
- // Plasticity counter (plan §WS-1 step 8) for the distill path.
619
- // quality_rejected: the LLM ran but produced output that didn't pass the
620
- // quality gate — the asset is not yielding useful distill output.
621
- // queued: a proposal was produced; reset the no-op counter.
622
- if (eventsCtx?.db) {
623
- // Use the same item_ref-or-conceptId key as the distill/preparation writers.
624
- const plasticityKey = planned.itemRef ?? durableImproveRef(planned.ref);
625
- try {
626
- if (distillResult.outcome === "quality_rejected" || distillResult.outcome === "skipped") {
627
- recordNoOp(eventsCtx.db, plasticityKey);
628
- }
629
- else if (distillResult.outcome === "queued") {
630
- resetConsecutiveNoOps(eventsCtx.db, plasticityKey);
631
- }
632
- }
633
- catch {
634
- // best-effort: plasticity counter failure never blocks the run
635
- }
636
- }
637
- }
638
- /**
639
- * Wall-clock budget exhausted mid-loop (O-1 / #364): emit the improve_skipped
640
- * events for the current and remaining refs (B11) and return the terminal
641
- * error action for the orchestrator to record before breaking out of the loop.
642
- */
643
- function recordBudgetExhausted(args) {
644
- const { planned, loopRefs, completedCount, startMs, eventsCtx } = args;
645
- const remaining = loopRefs.length - completedCount;
646
- info(`[improve] budget exhausted after ${Math.round((Date.now() - startMs) / 60000)}min — ${remaining} assets skipped`);
647
- appendEvent({
648
- eventType: "improve_skipped",
649
- ref: planned.ref,
650
- metadata: {
651
- reason: "budget_exhausted",
652
- remaining,
653
- },
654
- }, eventsCtx);
655
- // B11: Emit improve_skipped for all remaining assets that will not be processed.
656
- for (const remainingRef of loopRefs.slice(completedCount + 1)) {
657
- appendEvent({
658
- eventType: "improve_skipped",
659
- ref: remainingRef.ref,
660
- metadata: { reason: "budget_exhausted_batch", remaining: loopRefs.length - completedCount - 1 },
661
- }, eventsCtx);
237
+ if (refType === "memory" && !(result.outcome === "queued" && result.proposalKind === "knowledge")) {
238
+ tally.memoryRefsForInference.push(planned.ref);
662
239
  }
663
- return {
664
- ref: planned.ref,
665
- mode: "error",
666
- result: { ok: false, error: "timeout: improve wall-clock budget exhausted" },
667
- };
240
+ recordPlasticity(env, planned, result.outcome === "quality_rejected" || result.outcome === "skipped"
241
+ ? "noop"
242
+ : result.outcome === "queued"
243
+ ? "changed"
244
+ : undefined);
668
245
  }
669
246
  export async function runImproveLoopStage(args) {
670
- const { ctx, loopRefs, actions, recentErrors, startMs, budgetMs } = args;
671
- const eventsCtx = ctx.eventsCtx;
247
+ const { loopRefs, actions, recentErrors, startMs, budgetMs, eventsCtx } = args;
672
248
  const env = prepareImproveLoopEnv(args);
673
- let completedCount = 0;
674
249
  let reflectsWithErrorContext = 0;
675
250
  const memoryRefsForInference = new Set();
676
- for (const planned of loopRefs) {
251
+ for (const [index, planned] of loopRefs.entries()) {
677
252
  if (Date.now() - startMs >= budgetMs) {
678
- actions.push(recordBudgetExhausted({ planned, loopRefs, completedCount, startMs, eventsCtx }));
253
+ const remaining = loopRefs.length - index;
254
+ info(`[improve] budget exhausted after ${Math.round((Date.now() - startMs) / 60000)}min — ${remaining} assets skipped`);
255
+ appendEvent({ eventType: "improve_skipped", ref: planned.ref, metadata: { reason: "budget_exhausted", remaining } }, eventsCtx);
256
+ for (const rest of loopRefs.slice(index + 1)) {
257
+ appendEvent({
258
+ eventType: "improve_skipped",
259
+ ref: rest.ref,
260
+ metadata: { reason: "budget_exhausted_batch", remaining: remaining - 1 },
261
+ }, eventsCtx);
262
+ }
263
+ actions.push({
264
+ ref: planned.ref,
265
+ mode: "error",
266
+ result: { ok: false, error: "timeout: improve wall-clock budget exhausted" },
267
+ });
679
268
  break;
680
269
  }
681
270
  const tally = await processImproveLoopRef(planned, env);
682
- // Fold the per-ref tally into run-level state — the passes never touch it.
683
271
  actions.push(...tally.actions);
684
272
  for (const push of tally.recentErrorPushes)
685
273
  pushRecentError(recentErrors, push.originator, push.message);
686
274
  reflectsWithErrorContext += tally.reflectsWithErrorContext;
687
275
  for (const ref of tally.memoryRefsForInference)
688
276
  memoryRefsForInference.add(ref);
689
- completedCount++;
690
- info(`[improve] ${completedCount}/${loopRefs.length} ${planned.ref}`);
277
+ info(`[improve] ${index + 1}/${loopRefs.length} ${planned.ref}`);
691
278
  }
692
279
  return { reflectsWithErrorContext, memoryRefsForInference };
693
280
  }
694
281
  export async function runImprovePostLoopStage(args) {
695
- const { scope, options, primaryStashDir, actionableRefs, appliedCleanup, cleanupWarnings, memoryRefsForInference, eventsCtx, budgetSignal, improveProfile, resolvedPlan, consolidationRan, } = args;
696
- const allWarnings = [...cleanupWarnings, ...(appliedCleanup?.warnings ?? [])];
282
+ const { scope, primaryStashDir, actionableRefs } = args;
283
+ const allWarnings = [...args.cleanupWarnings, ...(args.appliedCleanup?.warnings ?? [])];
697
284
  info("[improve] post-loop maintenance starting");
698
- const maintenanceResult = await runImproveMaintenancePasses({
699
- options,
700
- primaryStashDir,
701
- actionableRefs,
702
- memoryRefsForInference,
703
- allWarnings,
704
- // O-1 (#364): forward the budget signal to memory inference + graph extraction.
705
- budgetSignal,
706
- eventsCtx,
707
- improveProfile,
708
- resolvedPlan,
709
- });
285
+ const maintenance = await runImproveMaintenancePasses({ ...args, allWarnings });
710
286
  let deadUrls;
711
287
  let deadUrlCoverage;
712
288
  if (scope.mode === "all" && primaryStashDir && actionableRefs.length > 0) {
713
289
  try {
714
- // Every actionable knowledge ref is scanned for URLs — there used to be
715
- // a `.slice(0, 10)` here, capping the scan to the first ten refs while
716
- // `deadUrlCoverage.total` counted only those, so a real bundle reported
717
- // checked === total while most refs were never looked at (#892). URL
718
- // extraction (a regex over already-loaded text) is cheap; it is the
719
- // network requests that are expensive, and those are bounded by
720
- // `checkDeadUrls`'s concurrency limit, not by trimming what gets scanned.
290
+ // Every actionable knowledge ref is scanned; checkDeadUrls bounds the
291
+ // network concurrency (#892).
721
292
  const knowledgeEntries = actionableRefs
722
293
  .filter((r) => {
723
294
  try {
@@ -728,9 +299,6 @@ export async function runImprovePostLoopStage(args) {
728
299
  }
729
300
  })
730
301
  .map((r) => {
731
- // The URL scan needs the document body; filePath is pre-resolved on
732
- // eligible refs at planning time (#591). Best-effort — an unreadable
733
- // or unresolved file contributes no URLs, same as before.
734
302
  let body = "";
735
303
  if (r.filePath) {
736
304
  try {
@@ -754,114 +322,63 @@ export async function runImprovePostLoopStage(args) {
754
322
  // best-effort
755
323
  }
756
324
  }
757
- // ── R5: collapse/churn detector ────────────────────────────────────────────
758
- // One snapshot per QUALIFYING cycle: consolidate processed work. Deterministic,
759
- // observe-only, fail-open (the orchestrator catches everything) — and inert
760
- // on the ~9-in-10 default-profile runs that touch no merges.
761
- let cycleMetrics;
762
- if (!options.dryRun && consolidationRan) {
763
- cycleMetrics = runCollapseDetector({
764
- runId: options.runId ?? "improve-adhoc",
765
- ...(improveProfile ? { improveProfile } : {}),
766
- pass: "consolidate",
767
- mergeFloorViolations: args.consolidationMergeFloorViolations ?? 0,
768
- config: options.config ?? loadConfig(),
769
- ...(eventsCtx ? { eventsCtx } : {}),
770
- });
771
- }
772
325
  return {
773
326
  allWarnings,
774
327
  deadUrls,
775
328
  ...(deadUrlCoverage ? { deadUrlCoverage } : {}),
776
- ...(cycleMetrics ? { cycleMetrics } : {}),
777
- ...(maintenanceResult.memoryInference ? { memoryInference: maintenanceResult.memoryInference } : {}),
778
- ...(maintenanceResult.graphExtraction ? { graphExtraction: maintenanceResult.graphExtraction } : {}),
779
- ...(maintenanceResult.actions && maintenanceResult.actions.length > 0
780
- ? { maintenanceActions: maintenanceResult.actions }
781
- : {}),
782
- memoryInferenceDurationMs: maintenanceResult.memoryInferenceDurationMs,
783
- graphExtractionDurationMs: maintenanceResult.graphExtractionDurationMs,
784
- orphansPurged: maintenanceResult.orphansPurged,
785
- proposalsExpired: maintenanceResult.proposalsExpired,
329
+ ...(maintenance.memoryInference ? { memoryInference: maintenance.memoryInference } : {}),
330
+ ...(maintenance.graphExtraction ? { graphExtraction: maintenance.graphExtraction } : {}),
331
+ ...(maintenance.actions && maintenance.actions.length > 0 ? { maintenanceActions: maintenance.actions } : {}),
332
+ memoryInferenceDurationMs: maintenance.memoryInferenceDurationMs,
333
+ graphExtractionDurationMs: maintenance.graphExtractionDurationMs,
334
+ orphansPurged: maintenance.orphansPurged,
335
+ proposalsExpired: maintenance.proposalsExpired,
786
336
  };
787
337
  }
788
- // Exported for tests (#584/#585 DB-locking regression coverage); production
789
- // callers reach it only through akmImprove → runImprovePostLoopStage.
338
+ /**
339
+ * Memory inference → index what it wrote → graph extraction → proposal hygiene
340
+ * (orphan purge, expiration) → orphan-state GC → retention purges. Warnings go
341
+ * to `allWarnings`.
342
+ */
790
343
  export async function runImproveMaintenancePasses(args) {
791
- const { options, primaryStashDir, memoryRefsForInference, allWarnings, budgetSignal, eventsCtx } = args;
792
- if (!primaryStashDir)
793
- return { memoryInferenceDurationMs: 0, graphExtractionDurationMs: 0 };
794
- if (budgetSignal?.aborted)
344
+ const { options, primaryStashDir, allWarnings, budgetSignal, eventsCtx } = args;
345
+ if (!primaryStashDir || budgetSignal?.aborted)
795
346
  return { memoryInferenceDurationMs: 0, graphExtractionDurationMs: 0 };
796
347
  const config = options.config ?? loadConfig();
797
- const sources = resolveSourceEntries(options.stashDir, config);
798
- const memoryInferenceFn = options.memoryInferenceFn ?? runMemoryInferencePass;
799
- const graphExtractionFn = options.graphExtractionFn ?? runGraphExtractionPass;
800
- const openIndexDb = () => openIndexDatabase(getDbPath(), config.embedding?.dimension ? { embeddingDim: config.embedding.dimension } : undefined);
801
- const dbCell = {};
802
348
  const ctx = {
803
349
  config,
804
- sources,
350
+ sources: resolveSourceEntries(options.stashDir, config),
805
351
  primaryStashDir,
806
352
  eventsCtx,
807
353
  budgetSignal,
808
354
  improveProfile: args.improveProfile,
809
355
  resolvedPlan: args.resolvedPlan,
810
- memoryInferenceFn,
811
- graphExtractionFn,
356
+ memoryInferenceFn: options.memoryInferenceFn ?? runMemoryInferencePass,
357
+ graphExtractionFn: options.graphExtractionFn ?? runGraphExtractionPass,
812
358
  };
813
- const collected = await runMaintenancePassesUnderLease(ctx, dbCell, {
814
- actionableRefs: args.actionableRefs,
815
- memoryRefsForInference,
816
- allWarnings,
817
- openIndexDb,
818
- });
819
- return {
820
- ...(collected.memoryInference ? { memoryInference: collected.memoryInference } : {}),
821
- ...(collected.graphExtraction ? { graphExtraction: collected.graphExtraction } : {}),
822
- ...(collected.actions.length > 0 ? { actions: collected.actions } : {}),
823
- memoryInferenceDurationMs: collected.memoryInferenceDurationMs,
824
- graphExtractionDurationMs: collected.graphExtractionDurationMs,
825
- orphansPurged: collected.orphansPurged,
826
- proposalsExpired: collected.proposalsExpired,
827
- };
828
- }
829
- /**
830
- * The maintenance sequence (formerly the ~389-line anonymous
831
- * `withIndexWriterLease` callback, before #872 removed the index-rebuild
832
- * lease): memory inference → reindex-after-inference → graph extraction →
833
- * proposal hygiene (orphan purge, expiration) → retention purges. Each pass
834
- * returns its results and warnings; this orchestrator folds warnings into the
835
- * caller's `allWarnings` sink at the same points the inline code pushed them.
836
- */
837
- async function runMaintenancePassesUnderLease(ctx, dbCell, args) {
838
- const { allWarnings } = args;
359
+ const openIndexDb = () => openIndexDatabase(getDbPath());
360
+ const dbCell = {};
839
361
  const actions = [];
840
362
  try {
841
- dbCell.current = args.openIndexDb();
363
+ dbCell.current = openIndexDb();
842
364
  const inference = await runMemoryInferenceMaintenancePass(ctx, dbCell, args.memoryRefsForInference);
843
365
  if (inference.action)
844
366
  actions.push(inference.action);
845
367
  allWarnings.push(...inference.warnings);
846
- const memoryInference = inference.memoryInference;
847
- // R78 (tier1-0917): index exactly the files memory inference wrote (derived children
848
- // + rewritten parents) instead of a full reindex — typically one written
849
- // fact per run, which used to pay a full-corpus reindex regardless.
850
- if (memoryInference && memoryInference.writtenPaths.length > 0) {
851
- info(`[improve] indexing ${memoryInference.writtenPaths.length} file(s) written by memory inference`);
368
+ const written = inference.memoryInference?.writtenPaths ?? [];
369
+ if (written.length > 0) {
370
+ // Index exactly the files inference wrote. indexWrittenAssets opens its
371
+ // own write handle, so ours closes first and reopens after.
372
+ info(`[improve] indexing ${written.length} file(s) written by memory inference`);
852
373
  try {
853
- // #584: indexWrittenAssets opens its own write handle on the same
854
- // index.db WAL file, so the maintenance handle must be closed first
855
- // and a fresh one reopened after, even on failure.
856
- if (dbCell.current) {
374
+ if (dbCell.current)
857
375
  closeDatabase(dbCell.current);
858
- dbCell.current = undefined;
859
- }
376
+ dbCell.current = undefined;
860
377
  try {
861
- await indexWrittenAssets(ctx.primaryStashDir, memoryInference.writtenPaths);
378
+ await indexWrittenAssets(primaryStashDir, written);
862
379
  }
863
380
  finally {
864
- dbCell.current = args.openIndexDb();
381
+ dbCell.current = openIndexDb();
865
382
  }
866
383
  info("[improve] indexing after memory inference complete");
867
384
  }
@@ -869,32 +386,22 @@ async function runMaintenancePassesUnderLease(ctx, dbCell, args) {
869
386
  allWarnings.push(`indexing after memory inference failed: ${errMessage(err)}`);
870
387
  }
871
388
  }
872
- const graph = await runGraphExtractionMaintenancePass(ctx, dbCell, {
873
- actionableRefs: args.actionableRefs,
874
- memoryRefsForInference: args.memoryRefsForInference,
875
- });
389
+ const graph = await runGraphExtractionMaintenancePass(ctx, dbCell, args);
876
390
  if (graph.action)
877
391
  actions.push(graph.action);
878
392
  allWarnings.push(...graph.warnings);
879
- const orphan = runOrphanProposalPurgePass(ctx);
880
- allWarnings.push(...orphan.warnings);
881
- // #733: orphan-state GC — stamps/clears/(optionally) collects
882
- // asset_salience/asset_outcome rows whose ref no longer resolves in
883
- // index.db. Needs the SAME already-open index.db handle (dbCell.current)
884
- // the passes above share.
885
- const stateGc = runOrphanStateGcPass(ctx, dbCell);
886
- allWarnings.push(...stateGc.warnings);
887
- const expiration = runProposalExpirationPass(ctx);
888
- allWarnings.push(...expiration.warnings);
393
+ const hygiene = runProposalHygienePass(ctx);
394
+ allWarnings.push(...hygiene.warnings);
395
+ allWarnings.push(...runOrphanStateGcPass(ctx, dbCell).warnings);
889
396
  allWarnings.push(...runRetentionPurgePass(ctx).warnings);
890
397
  return {
891
- memoryInference,
892
- graphExtraction: graph.graphExtraction,
893
- actions,
398
+ ...(inference.memoryInference ? { memoryInference: inference.memoryInference } : {}),
399
+ ...(graph.graphExtraction ? { graphExtraction: graph.graphExtraction } : {}),
400
+ ...(actions.length > 0 ? { actions } : {}),
894
401
  memoryInferenceDurationMs: inference.durationMs,
895
402
  graphExtractionDurationMs: graph.durationMs,
896
- orphansPurged: orphan.orphansPurged,
897
- proposalsExpired: expiration.proposalsExpired,
403
+ orphansPurged: hygiene.orphansPurged,
404
+ proposalsExpired: hygiene.proposalsExpired,
898
405
  };
899
406
  }
900
407
  finally {
@@ -902,542 +409,322 @@ async function runMaintenancePassesUnderLease(ctx, dbCell, args) {
902
409
  closeDatabase(dbCell.current);
903
410
  }
904
411
  }
412
+ /** Time one LLM maintenance pass; a throw becomes a `<label> failed: …` warning. */
413
+ async function timedLlmPass(label, run) {
414
+ const start = Date.now();
415
+ try {
416
+ const result = await run();
417
+ return { result, durationMs: Date.now() - start, warnings: [] };
418
+ }
419
+ catch (err) {
420
+ return { durationMs: Date.now() - start, warnings: [`${label} failed: ${errMessage(err)}`] };
421
+ }
422
+ }
905
423
  /**
906
- * Memory inference candidate-discovery (post-Item 9 fix from
907
- * memories/akm-improve-critical-review-2026-05-20). Previously this pass
908
- * was gated on memoryRefsForInference.size > 0 AND passed those refs as a
909
- * candidateRefs filter. But memoryRefsForInference is populated from refs
910
- * distilled THIS RUN — by the time that happens, those parents are
911
- * already split (`inferenceProcessed: true`) and `isPendingMemory` excludes
912
- * them. The genuinely-pending parents in the stash never entered the
913
- * filter. Result: 0/0/0 for 25 consecutive runs.
914
- *
915
- * Fix: always run the pass when the feature is enabled; let the pass's
916
- * own `collectPendingMemories` + `isPendingMemory` predicate find
917
- * candidates from the filesystem-of-truth. The this-run set is still
918
- * logged as a hint but no longer used as a filter.
424
+ * Memory inference over every pending parent in the stash. The pass discovers
425
+ * its own candidates; the refs distilled this run are only logged as a hint.
919
426
  */
920
427
  export async function runMemoryInferenceMaintenancePass(ctx, dbCell, memoryRefsForInference) {
921
- const { config, sources, primaryStashDir, budgetSignal, improveProfile, resolvedPlan, memoryInferenceFn } = ctx;
922
- const warnings = [];
923
- let memoryInference;
924
- let durationMs = 0;
925
- let action;
926
- const memoryInferenceDisabledByProfile = improveProfile?.processes?.memoryInference?.enabled === false;
927
- const minPendingCount = improveProfile?.processes?.memoryInference?.minPendingCount;
928
- const pendingBelowMinCount = (() => {
929
- if (!primaryStashDir || minPendingCount === undefined || minPendingCount <= 0)
930
- return false;
428
+ const { config, sources, primaryStashDir, resolvedPlan } = ctx;
429
+ const settings = ctx.improveProfile?.processes?.memoryInference;
430
+ if (settings?.enabled === false) {
431
+ info("[improve] memory inference skipped (disabled by improve profile)");
432
+ return { durationMs: 0, warnings: [] };
433
+ }
434
+ const minPendingCount = settings?.minPendingCount;
435
+ if (primaryStashDir && minPendingCount !== undefined && minPendingCount > 0) {
931
436
  const pending = collectPendingMemories(primaryStashDir).length;
932
437
  if (pending < minPendingCount) {
933
438
  info(`[improve] memory inference skipped (${pending} pending < minPendingCount ${minPendingCount})`);
934
- return true;
935
- }
936
- return false;
937
- })();
938
- if (memoryInferenceDisabledByProfile) {
939
- info("[improve] memory inference skipped (disabled by improve profile)");
940
- }
941
- else if (pendingBelowMinCount) {
942
- // skipped — message already emitted above
943
- }
944
- else {
945
- const hintRefs = memoryRefsForInference.size;
946
- info(hintRefs > 0
947
- ? `[improve] memory inference starting (${hintRefs} hint refs touched this run; pass discovers all pending)`
948
- : "[improve] memory inference starting (discovering pending parents)");
949
- const inferenceStart = Date.now();
950
- try {
951
- // O-1 (#364): pass budget signal so a hung inference call is cancelled.
952
- memoryInference = await withLlmStage("memory-inference", () => memoryInferenceFn({
953
- config,
954
- ...(resolvedPlan
955
- ? {
956
- llmRunner: resolvedPlan.processes.memoryInference.runner,
957
- }
958
- : {}),
959
- sources,
960
- signal: budgetSignal,
961
- db: dbCell.current,
962
- reEnrich: false,
963
- onProgress: (event) => {
964
- const current = event.currentRef ? ` ${event.currentRef}` : "";
965
- info(`[improve] memory inference ${event.processed}/${event.total}${current} (written ${event.writtenFacts}, skipped ${event.skippedNoFacts})`);
966
- },
967
- }), { engine: resolvedPlan?.processes.memoryInference.runner?.engine, process: "memoryInference" });
968
- durationMs = Date.now() - inferenceStart;
969
- // Synthetic sentinel ref (ref-grammar decision D-R3): a colon-free
970
- // `<domain>/_<marker>` label on the event row, never parsed as an asset
971
- // ref. The domain is the asset stash-subdir for asset-scoped sentinels
972
- // (`memories/…`) and the subsystem name for maintenance/artifact sentinels
973
- // (`graph/…`, `events/…`, `proposals/…`, `health/…`, …). Readers match the
974
- // event by `eventType`, never by this string.
975
- action = { ref: "memories/_inference", mode: "memory-inference", result: memoryInference };
976
- info(`[improve] memory inference complete (${memoryInference.writtenFacts} facts written from ${memoryInference.splitParents} parents)`);
977
- }
978
- catch (err) {
979
- durationMs = Date.now() - inferenceStart;
980
- warnings.push(`memory inference failed: ${errMessage(err)}`);
439
+ return { durationMs: 0, warnings: [] };
981
440
  }
982
441
  }
983
- return { memoryInference, durationMs, action, warnings };
442
+ const hintRefs = memoryRefsForInference.size;
443
+ info(hintRefs > 0
444
+ ? `[improve] memory inference starting (${hintRefs} hint refs touched this run; pass discovers all pending)`
445
+ : "[improve] memory inference starting (discovering pending parents)");
446
+ const pass = await timedLlmPass("memory inference", () => attributeStage(resolvedPlan, "memoryInference", () => ctx.memoryInferenceFn({
447
+ config,
448
+ ...(resolvedPlan ? { llmRunner: resolvedPlan.processes.memoryInference.runner } : {}),
449
+ sources,
450
+ signal: ctx.budgetSignal,
451
+ db: dbCell.current,
452
+ reEnrich: false,
453
+ onProgress: (event) => {
454
+ const current = event.currentRef ? ` ${event.currentRef}` : "";
455
+ info(`[improve] memory inference ${event.processed}/${event.total}${current} (written ${event.writtenFacts}, skipped ${event.skippedNoFacts})`);
456
+ },
457
+ })));
458
+ const memoryInference = pass.result;
459
+ if (!memoryInference)
460
+ return { durationMs: pass.durationMs, warnings: pass.warnings };
461
+ info(`[improve] memory inference complete (${memoryInference.writtenFacts} facts written from ${memoryInference.splitParents} parents)`);
462
+ return {
463
+ memoryInference,
464
+ durationMs: pass.durationMs,
465
+ // Sentinel refs (`<domain>/_<marker>`) label maintenance events; never parsed as assets.
466
+ action: { ref: "memories/_inference", mode: "memory-inference", result: memoryInference },
467
+ warnings: pass.warnings,
468
+ };
984
469
  }
985
470
  /**
986
- * Graph-extraction maintenance pass.
987
- *
988
- * INVARIANT: graph extraction normally runs only on files touched by
989
- * actionable refs (candidatePaths). Full-corpus scans are opt-in via
990
- * profile.processes.graphExtraction.fullScan = true (used by the
991
- * `graph-refresh` built-in profile and its weekly scheduled task).
992
- * The empty-Set fallback is intentional when no refs were touched —
993
- * the extractor's filter rejects every file and returns empty, keeping
994
- * the pass invoked so the action is recorded and tests stay exercised.
471
+ * Graph extraction over the files this run touched, or the whole corpus when
472
+ * the profile sets `graphExtraction.fullScan` (the `graph-refresh` strategy).
473
+ * With nothing touched the pass still runs and extracts nothing.
995
474
  */
996
475
  export async function runGraphExtractionMaintenancePass(ctx, dbCell, args) {
997
- const { config, sources, primaryStashDir, budgetSignal, improveProfile, resolvedPlan, graphExtractionFn } = ctx;
998
- const warnings = [];
999
- let graphExtraction;
1000
- let durationMs = 0;
1001
- let action;
476
+ const { config, sources, primaryStashDir, resolvedPlan } = ctx;
477
+ const settings = ctx.improveProfile?.processes?.graphExtraction;
1002
478
  const graphEnabled = resolvedPlan ? true : isProcessEnabled("index", "graph_extraction", config);
1003
- const graphExtractionDisabledByProfile = improveProfile?.processes?.graphExtraction?.enabled === false;
1004
- const graphExtractionFullScan = improveProfile?.processes?.graphExtraction?.fullScan === true;
1005
- // #624 P2: optional incremental high-signal-first cap. Unset = process all
1006
- // eligible (byte-identical to today; no ranking/slice).
1007
- const graphExtractionTopN = improveProfile?.processes?.graphExtraction?.topN;
1008
- const graphExtractionIncludeTypes = improveProfile?.processes?.graphExtraction?.includeTypes ?? [
1009
- ...DEFAULT_GRAPH_EXTRACTION_INCLUDE_TYPES,
1010
- ];
1011
- const graphExtractionBatchSize = improveProfile?.processes?.graphExtraction?.batchSize ?? DEFAULT_GRAPH_EXTRACTION_BATCH_SIZE;
1012
- const graphExtractionMaxChunksPerAsset = improveProfile?.processes?.graphExtraction?.maxChunksPerAsset;
1013
- // Build the set of refs actually touched this run.
1014
- const touchedRefs = new Set();
1015
- for (const r of args.actionableRefs)
1016
- touchedRefs.add(r.ref);
1017
- for (const r of args.memoryRefsForInference)
1018
- touchedRefs.add(r);
1019
- if (graphExtractionDisabledByProfile) {
479
+ if (settings?.enabled === false) {
1020
480
  info("[improve] graph extraction skipped (disabled by improve profile)");
481
+ return { durationMs: 0, warnings: [] };
1021
482
  }
1022
- else if (sources.length > 0 && graphEnabled) {
1023
- info(`[improve] graph extraction starting${graphExtractionFullScan ? " (full-corpus scan)" : ""}`);
1024
- const extractionStart = Date.now();
1025
- try {
1026
- // Resolve touched refs to absolute file paths. Skipped for fullScan
1027
- // (candidatePaths stays undefined → extractor processes all files).
1028
- let candidatePaths;
1029
- if (!graphExtractionFullScan) {
1030
- candidatePaths = new Set();
1031
- if (primaryStashDir && touchedRefs.size > 0) {
1032
- const writableBundleIds = deriveWritableBundleIds(resolveSourceEntries(primaryStashDir));
1033
- const resolved = await Promise.all([...touchedRefs].map((ref) => findAssetFilePath(ref, primaryStashDir, writableBundleIds).catch(() => null)));
1034
- for (const p of resolved) {
1035
- if (typeof p === "string" && p.length > 0)
1036
- candidatePaths.add(p);
1037
- }
1038
- }
483
+ if (sources.length === 0)
484
+ return { durationMs: 0, warnings: [] };
485
+ if (!graphEnabled) {
486
+ info("[improve] graph extraction skipped (features.index.graph_extraction is disabled)");
487
+ return { durationMs: 0, warnings: [] };
488
+ }
489
+ const fullScan = settings?.fullScan === true;
490
+ info(`[improve] graph extraction starting${fullScan ? " (full-corpus scan)" : ""}`);
491
+ const pass = await timedLlmPass("graph extraction", async () => {
492
+ let candidatePaths;
493
+ if (!fullScan) {
494
+ candidatePaths = new Set();
495
+ const touched = new Set([...args.actionableRefs.map((r) => r.ref), ...args.memoryRefsForInference]);
496
+ if (primaryStashDir && touched.size > 0) {
497
+ const writableBundleIds = deriveWritableBundleIds(resolveSourceEntries(primaryStashDir));
498
+ const resolved = await Promise.all([...touched].map((ref) => findAssetFilePath(ref, primaryStashDir, writableBundleIds).catch(() => null)));
499
+ for (const p of resolved)
500
+ if (typeof p === "string" && p.length > 0)
501
+ candidatePaths.add(p);
1039
502
  }
1040
- const progressHandler = (event) => {
503
+ }
504
+ return attributeStage(resolvedPlan, "graphExtraction", () => ctx.graphExtractionFn({
505
+ config,
506
+ ...(resolvedPlan ? { llmRunner: resolvedPlan.processes.graphExtraction.runner } : {}),
507
+ sources,
508
+ signal: ctx.budgetSignal,
509
+ db: dbCell.current,
510
+ reEnrich: false,
511
+ onProgress: (event) => {
1041
512
  const current = event.currentPath ? ` ${path.basename(event.currentPath)}` : "";
1042
513
  info(`[improve] graph extraction ${event.processed}/${event.total}${current} (extracted ${event.extracted}, entities ${event.totalEntities}, relations ${event.totalRelations})`);
1043
- };
1044
- // O-1 (#364): pass budget signal so a hung graph extraction call is cancelled.
1045
- graphExtraction = await withLlmStage("graph-extraction", () => graphExtractionFn({
1046
- config,
1047
- ...(resolvedPlan
1048
- ? {
1049
- llmRunner: resolvedPlan.processes.graphExtraction.runner,
1050
- }
1051
- : {}),
1052
- sources,
1053
- signal: budgetSignal,
1054
- db: dbCell.current,
1055
- reEnrich: false,
1056
- onProgress: progressHandler,
1057
- options: {
1058
- candidatePaths,
1059
- includeTypes: graphExtractionIncludeTypes,
1060
- batchSize: graphExtractionBatchSize,
1061
- ...(graphExtractionTopN != null ? { topN: graphExtractionTopN } : {}),
1062
- ...(graphExtractionMaxChunksPerAsset != null
1063
- ? { maxChunksPerAsset: graphExtractionMaxChunksPerAsset }
1064
- : {}),
1065
- },
1066
- }), { engine: resolvedPlan?.processes.graphExtraction.runner?.engine, process: "graphExtraction" });
1067
- durationMs = Date.now() - extractionStart;
1068
- // Synthetic sentinel ref (D-R3): `graph` has no asset stash-subdir, so the
1069
- // colon-free `graph/_artifact` names the subsystem, per the sentinel
1070
- // convention documented at the memory-inference writer above.
1071
- action = { ref: "graph/_artifact", mode: "graph-extraction", result: graphExtraction };
1072
- info(`[improve] graph extraction complete (${graphExtraction.quality.extractedFiles} files, ${graphExtraction.quality.entityCount} entities, ${graphExtraction.quality.relationCount} relations)`);
1073
- }
1074
- catch (err) {
1075
- durationMs = Date.now() - extractionStart;
1076
- warnings.push(`graph extraction failed: ${errMessage(err)}`);
1077
- }
1078
- }
1079
- else if (sources.length > 0 && !graphEnabled) {
1080
- info("[improve] graph extraction skipped (features.index.graph_extraction is disabled)");
1081
- }
1082
- return { graphExtraction, durationMs, action, warnings };
514
+ },
515
+ options: {
516
+ candidatePaths,
517
+ includeTypes: settings?.includeTypes ?? [...DEFAULT_GRAPH_EXTRACTION_INCLUDE_TYPES],
518
+ batchSize: settings?.batchSize ?? DEFAULT_GRAPH_EXTRACTION_BATCH_SIZE,
519
+ ...(settings?.topN != null ? { topN: settings.topN } : {}),
520
+ ...(settings?.maxChunksPerAsset != null ? { maxChunksPerAsset: settings.maxChunksPerAsset } : {}),
521
+ },
522
+ }));
523
+ });
524
+ const graphExtraction = pass.result;
525
+ if (!graphExtraction)
526
+ return { durationMs: pass.durationMs, warnings: pass.warnings };
527
+ info(`[improve] graph extraction complete (${graphExtraction.quality.extractedFiles} files, ${graphExtraction.quality.entityCount} entities, ${graphExtraction.quality.relationCount} relations)`);
528
+ return {
529
+ graphExtraction,
530
+ durationMs: pass.durationMs,
531
+ action: { ref: "graph/_artifact", mode: "graph-extraction", result: graphExtraction },
532
+ warnings: pass.warnings,
533
+ };
1083
534
  }
1084
535
  /**
1085
- * Orphan proposal purge — reject pending reflect proposals whose target
1086
- * asset no longer exists on disk. Runs after graph extraction so newly
1087
- * promoted assets from accept flows during this run are already present.
536
+ * Reject pending proposals whose target no longer exists, then expire pending
537
+ * proposals past the retention window; each emits a roll-up event.
1088
538
  */
1089
- function runOrphanProposalPurgePass(ctx) {
1090
- const { primaryStashDir, sources, eventsCtx } = ctx;
539
+ function runProposalHygienePass(ctx) {
540
+ const { primaryStashDir, eventsCtx } = ctx;
1091
541
  const warnings = [];
1092
542
  let orphansPurged = 0;
543
+ let proposalsExpired = 0;
1093
544
  try {
1094
- const purgeResult = purgeOrphanProposals(primaryStashDir, sources.map((s) => s.path));
1095
- orphansPurged = purgeResult.rejected;
1096
- if (purgeResult.rejected > 0) {
1097
- info(`[improve] orphan purge: ${purgeResult.rejected}/${purgeResult.checked} orphaned proposals rejected (${purgeResult.durationMs}ms)`);
545
+ const purge = purgeOrphanProposals(primaryStashDir, ctx.sources.map((s) => s.path));
546
+ orphansPurged = purge.rejected;
547
+ if (purge.rejected > 0) {
548
+ info(`[improve] orphan purge: ${purge.rejected}/${purge.checked} orphaned proposals rejected (${purge.durationMs}ms)`);
1098
549
  }
1099
550
  appendEvent({
1100
551
  eventType: "proposal_orphan_purge",
1101
552
  ref: "proposals/_orphan-purge",
1102
553
  metadata: {
1103
- checked: purgeResult.checked,
1104
- rejected: purgeResult.rejected,
1105
- durationMs: purgeResult.durationMs,
1106
- byType: purgeResult.byType,
1107
- orphans: purgeResult.orphans.map((o) => o.ref),
554
+ checked: purge.checked,
555
+ rejected: purge.rejected,
556
+ durationMs: purge.durationMs,
557
+ byType: purge.byType,
558
+ orphans: purge.orphans.map((o) => o.ref),
1108
559
  },
1109
560
  }, eventsCtx);
1110
561
  }
1111
562
  catch (err) {
1112
563
  warnings.push(`orphan purge failed: ${errMessage(err)}`);
1113
564
  }
1114
- return { orphansPurged, warnings };
1115
- }
1116
- /**
1117
- * Phase 6B (Advantage D6b): expire pending proposals that have aged past
1118
- * the retention window. Runs AFTER orphan purge so we never double-archive
1119
- * a proposal that orphan-purge already moved. `expireStaleProposals` emits
1120
- * its own per-proposal `proposal_expired` events; we additionally emit a
1121
- * single roll-up event here for parity with the orphan-purge surface.
1122
- */
1123
- function runProposalExpirationPass(ctx) {
1124
- const { primaryStashDir, config, eventsCtx } = ctx;
1125
- const warnings = [];
1126
- let proposalsExpired = 0;
1127
565
  try {
1128
- const expireResult = expireStaleProposals(primaryStashDir, config);
1129
- proposalsExpired = expireResult.expired;
1130
- if (expireResult.expired > 0) {
1131
- info(`[improve] expiration: ${expireResult.expired}/${expireResult.checked} pending proposals expired ` +
1132
- `(retention=${expireResult.retentionDays}d, ${expireResult.durationMs}ms)`);
566
+ const expiry = expireStaleProposals(primaryStashDir, ctx.config);
567
+ proposalsExpired = expiry.expired;
568
+ if (expiry.expired > 0) {
569
+ info(`[improve] expiration: ${expiry.expired}/${expiry.checked} pending proposals expired ` +
570
+ `(retention=${expiry.retentionDays}d, ${expiry.durationMs}ms)`);
1133
571
  }
1134
572
  appendEvent({
1135
573
  eventType: "proposal_expiration_pass",
1136
574
  ref: "proposals/_expiration",
1137
575
  metadata: {
1138
- checked: expireResult.checked,
1139
- expired: expireResult.expired,
1140
- durationMs: expireResult.durationMs,
1141
- retentionDays: expireResult.retentionDays,
1142
- expiredProposals: expireResult.expiredProposals,
576
+ checked: expiry.checked,
577
+ expired: expiry.expired,
578
+ durationMs: expiry.durationMs,
579
+ retentionDays: expiry.retentionDays,
580
+ expiredProposals: expiry.expiredProposals,
1143
581
  },
1144
582
  }, eventsCtx);
1145
583
  }
1146
584
  catch (err) {
1147
585
  warnings.push(`proposal expiration failed: ${errMessage(err)}`);
1148
586
  }
1149
- return { proposalsExpired, warnings };
587
+ return { orphansPurged, proposalsExpired, warnings };
1150
588
  }
1151
589
  /**
1152
- * Fix #2 (observability 0.8.0): trim the events table in state.db so it
1153
- * doesn't grow unbounded. `akm health` writes a `health_probe` row on every
1154
- * invocation, and every command surface emits at least one event besides —
1155
- * without this trim, state.db is a permanent append-only log. Config key
1156
- * `improve.eventRetentionDays` (default 90, set 0 to disable) controls the
1157
- * window. The purge runs against state.db (a different SQLite file from
1158
- * the index handle the other passes use).
590
+ * Trim the observability data that grows append-only — state.db events and
591
+ * improve_runs, logs.db task_logs, and the per-run task log files — to
592
+ * `improve.eventRetentionDays` (default 90; 0 disables), then VACUUM state.db
593
+ * when enough pages are free. Each store fails on its own. state.db work
594
+ * borrows the run's long-lived handle: a second writer on the same WAL file
595
+ * locks (#585).
1159
596
  */
1160
597
  export function runRetentionPurgePass(ctx) {
1161
598
  const { config, eventsCtx } = ctx;
1162
599
  const warnings = [];
1163
600
  const retentionDays = typeof config.improve?.eventRetentionDays === "number" ? config.improve.eventRetentionDays : 90;
1164
- if (retentionDays > 0) {
1165
- // #585: reuse the long-lived eventsCtx.db connection when akmImprove
1166
- // opened one — opening a second state.db write connection while
1167
- // eventsDb is still live made two simultaneous writers contend on the
1168
- // same WAL file ("database is locked"). Only the eventsCtx.dbPath
1169
- // fallback path (state.db failed to open up-front) opens — and then
1170
- // owns and closes — its own handle. C2 still holds: the fallback uses
1171
- // the boundary-pinned path, never a live `process.env` re-read.
1172
- try {
1173
- withStateDb((stateDb) => {
1174
- const purgedCount = purgeOldEvents(stateDb, retentionDays);
1175
- if (purgedCount > 0) {
1176
- info(`[improve] events purge: ${purgedCount} event(s) older than ${retentionDays}d removed from state.db`);
1177
- }
1178
- appendEvent({
1179
- eventType: "events_purged",
1180
- ref: "events/_purge",
1181
- metadata: { purgedCount, retentionDays },
1182
- }, eventsCtx);
1183
- // improve_runs uses the same retention window as events — both are
1184
- // observability/audit data, both grow append-only, both have a
1185
- // dedicated purge helper. Mirroring the events purge here means a
1186
- // single retention knob (improve.eventRetentionDays) governs both.
1187
- const improveRunsPurged = purgeOldImproveRuns(stateDb, retentionDays);
1188
- if (improveRunsPurged > 0) {
1189
- info(`[improve] improve_runs purge: ${improveRunsPurged} run(s) older than ${retentionDays}d removed from state.db`);
1190
- }
1191
- appendEvent({
1192
- eventType: "improve_runs_purged",
1193
- ref: "improve_runs/_purge",
1194
- metadata: { purgedCount: improveRunsPurged, retentionDays },
1195
- }, eventsCtx);
1196
- // R5: improve_cycle_metrics has its OWN retention window
1197
- // (default 365d — a slow collapse needs a longer trend than
1198
- // the 90d events window). canary_queries rows are never purged.
1199
- const cycleRetention = config.improve?.collapseDetector?.retentionDays ?? CYCLE_METRICS_RETENTION_DAYS;
1200
- const cycleMetricsPurged = purgeOldCycleMetrics(stateDb, cycleRetention);
1201
- if (cycleMetricsPurged > 0) {
1202
- info(`[improve] cycle-metrics purge: ${cycleMetricsPurged} row(s) older than ${cycleRetention}d removed from state.db`);
1203
- appendEvent({
1204
- // Dedicated type (mirrors improve_runs_purged) so consumers
1205
- // never have to disambiguate purge targets via the ref string.
1206
- eventType: "improve_cycle_metrics_purged",
1207
- ref: "improve_cycle_metrics/_purge",
1208
- metadata: { purgedCount: cycleMetricsPurged, retentionDays: cycleRetention },
1209
- }, eventsCtx);
1210
- }
1211
- // R0 step 3: opportunistic post-purge VACUUM. Reads the freelist
1212
- // off this same connection (no second state.db handle) and only
1213
- // runs when reclaimable space crosses STATE_DB_FREELIST_WARN_RATIO.
1214
- const vacuumOutcome = vacuumStateDbIfReclaimable(stateDb, readFreelistInfo(stateDb), eventsCtx);
1215
- if (vacuumOutcome.ran) {
1216
- info(`[improve] state.db vacuum: ${vacuumOutcome.pagesBefore} -> ${vacuumOutcome.pagesAfter} pages`);
1217
- }
1218
- }, { path: eventsCtx?.dbPath, borrowed: eventsCtx?.db });
1219
- }
1220
- catch (err) {
1221
- warnings.push(`events purge failed: ${errMessage(err)}`);
1222
- }
1223
- // task_logs in logs.db (#579) shares the same retention window as
1224
- // events/improve_runs — all three are observability data governed by
1225
- // the single improve.eventRetentionDays knob. Separate try/finally
1226
- // because logs.db is a different file: a locked/missing logs.db must
1227
- // not block the state.db purges above.
1228
- let logsDb;
1229
- try {
1230
- logsDb = openLogsDatabase();
1231
- const taskLogsPurged = purgeOldTaskLogs(logsDb, retentionDays);
1232
- if (taskLogsPurged > 0) {
1233
- info(`[improve] task_logs purge: ${taskLogsPurged} log line(s) older than ${retentionDays}d removed from logs.db`);
1234
- }
1235
- appendEvent({
1236
- eventType: "task_logs_purged",
1237
- ref: "task_logs/_purge",
1238
- metadata: { purgedCount: taskLogsPurged, retentionDays },
1239
- }, eventsCtx);
1240
- }
1241
- catch (err) {
1242
- warnings.push(`task_logs purge failed: ${errMessage(err)}`);
1243
- }
1244
- finally {
1245
- if (logsDb) {
1246
- try {
1247
- logsDb.close();
1248
- }
1249
- catch {
1250
- // best-effort
1251
- }
1252
- }
1253
- }
1254
- // Per-run flat log files under getTaskLogDir() (#951): logs.db above is
1255
- // the durable record and already retention-purged, so the transitional
1256
- // `<taskId>/<timestamp>.log` tail files can be deleted on the same
1257
- // window without losing anything. A separate try/catch — a filesystem
1258
- // problem here must not block the DB purges above.
601
+ if (retentionDays <= 0)
602
+ return { warnings };
603
+ const report = (eventType, ref, purgedCount, what) => {
604
+ if (purgedCount > 0)
605
+ info(`[improve] ${eventType}: ${purgedCount} ${what} older than ${retentionDays}d removed`);
606
+ appendEvent({ eventType, ref, metadata: { purgedCount, retentionDays } }, eventsCtx);
607
+ };
608
+ try {
609
+ withStateDb((stateDb) => {
610
+ report("events_purged", "events/_purge", purgeOldEvents(stateDb, retentionDays), "event(s)");
611
+ report("improve_runs_purged", "improve_runs/_purge", purgeOldImproveRuns(stateDb, retentionDays), "run(s)");
612
+ const vacuum = vacuumIfReclaimable(stateDb, readFreelistInfo(stateDb), { eventType: STATE_DB_VACUUMED_EVENT }, eventsCtx);
613
+ if (vacuum.ran)
614
+ info(`[improve] state.db vacuum: ${vacuum.pagesBefore} -> ${vacuum.pagesAfter} pages`);
615
+ }, { path: eventsCtx?.dbPath, borrowed: eventsCtx?.db });
616
+ }
617
+ catch (err) {
618
+ warnings.push(`events purge failed: ${errMessage(err)}`);
619
+ }
620
+ let logsDb;
621
+ try {
622
+ logsDb = openLogsDatabase();
623
+ report("task_logs_purged", "task_logs/_purge", purgeOldTaskLogs(logsDb, retentionDays), "log line(s)");
624
+ }
625
+ catch (err) {
626
+ warnings.push(`task_logs purge failed: ${errMessage(err)}`);
627
+ }
628
+ finally {
1259
629
  try {
1260
- const taskLogFilesPurged = purgeOldTaskLogFiles(undefined, retentionDays);
1261
- if (taskLogFilesPurged > 0) {
1262
- info(`[improve] task log files purge: ${taskLogFilesPurged} file(s) older than ${retentionDays}d removed from ${getTaskLogDir()}`);
1263
- }
1264
- appendEvent({
1265
- eventType: "task_log_files_purged",
1266
- ref: "task_log_files/_purge",
1267
- metadata: { purgedCount: taskLogFilesPurged, retentionDays },
1268
- }, eventsCtx);
630
+ logsDb?.close();
1269
631
  }
1270
- catch (err) {
1271
- warnings.push(`task log files purge failed: ${errMessage(err)}`);
632
+ catch {
633
+ // best-effort
1272
634
  }
1273
635
  }
636
+ try {
637
+ report("task_log_files_purged", "task_log_files/_purge", purgeOldTaskLogFiles(undefined, retentionDays), `file(s) under ${getTaskLogDir()}`);
638
+ }
639
+ catch (err) {
640
+ warnings.push(`task log files purge failed: ${errMessage(err)}`);
641
+ }
1274
642
  return { warnings };
1275
643
  }
1276
- // ── #733 — orphan-state GC pass (Workstream C) ──────────────────────────────
1277
- //
1278
- // Deliberately lean: one maintenance pass, one additive migration (021), one
1279
- // event type (asset_state_gc), one config gate (improve.stateGc.collect,
1280
- // default false). See docs/architecture/specs/0.9.0-close-out-plan.md
1281
- // Workstream C for the full design rationale. No quarantine archive, no
1282
- // circuit breaker, no health-advisory plumbing, no new tables.
1283
644
  /**
1284
- * Grace window (ms) before an unresolved `asset_salience` / `asset_outcome`
1285
- * row becomes delete-eligible — only when `improve.stateGc.collect` is true.
1286
- * A named constant, not a config knob (owner ruling — see the close-out
1287
- * plan's Workstream C). Mirrors `TXN_SWEEP_GRACE_MS` (src/core/fs-txn.ts:298).
645
+ * Grace window before an unresolved salience/outcome row may be deleted (only
646
+ * when `improve.stateGc.collect` is true).
1288
647
  */
1289
648
  export const STATE_GC_GRACE_MS = daysToMs(7);
1290
649
  /**
1291
- * Resolve one state-table's stored `asset_ref` against the live index.
1292
- *
1293
- * "ref not present in entries.item_ref" is the authoritative-deletion
1294
- * predicate (see the pass doc comment below), so this checks the same two
1295
- * spellings `getEntryByRef` (index-entries-repository.ts) resolves — an exact
1296
- * bundle-qualified item_ref, or a bare conceptId matched by suffix across all
1297
- * bundles — but against a prebuilt {@link LiveRefSnapshot}
1298
- * (`getLiveRefSnapshot`) instead of a database round trip per row: with up to
1299
- * a few thousand pending rows per run, one probe per row was the dominant
1300
- * cost R78 (tier1-0917).
1301
- *
1302
- * On top of that, falls back to the BARE conceptId form (`bareImproveRef` —
1303
- * the same primitive `preparation.ts`'s `normalizeStoredKey` map is built
1304
- * from via `improveStateReadRefs`) when the stored ref carries a bundle
1305
- * prefix that no longer matches exactly. This is the legacy-spelling
1306
- * normalization trap: a naive `asset_ref NOT IN (SELECT item_ref FROM
1307
- * entries)` would treat a live asset whose row predates bundle-qualification
1308
- * (or whose bundle prefix is stale) as an orphan and delete it. Preferring
1309
- * "never delete a live row" over "never miss a genuinely dead one" mirrors
1310
- * `getEntryByRef`'s own bare-conceptId suffix-match trade-off.
650
+ * A stored state ref is live when it, or its bare conceptId, resolves in the
651
+ * index. The bare fallback keeps a live asset whose row predates
652
+ * bundle-qualification from being collected.
1311
653
  */
1312
654
  function isStateRefLive(snapshot, storedRef) {
1313
655
  if (isRefLiveInSnapshot(snapshot, storedRef))
1314
656
  return true;
1315
- const bare = bareImproveRef(storedRef);
657
+ const bare = stripBundle(storedRef);
1316
658
  return bare !== storedRef && isRefLiveInSnapshot(snapshot, bare);
1317
659
  }
1318
660
  /**
1319
- * Sweep ONE state table: stamp refs that just went unresolved, clear refs
1320
- * that resolved again, and — only when `collect` is true — delete rows whose
1321
- * `missing_since` is older than {@link STATE_GC_GRACE_MS}. `pending` is a
1322
- * point-in-time snapshot taken AFTER stamp/clear/delete (re-queried, not
1323
- * accumulated), so it reflects the current backlog rather than this run's
1324
- * delta — "every run emits the counts either way, so live data accumulates
1325
- * proof" (close-out plan, Workstream C).
1326
- */
1327
- function gcOneStateTable(args) {
1328
- const { refRows, liveRefs, now, collect, stamp, clear, deleteOlderThan, countPending } = args;
1329
- const toStamp = [];
1330
- const toClear = [];
1331
- for (const row of refRows) {
1332
- const live = isStateRefLive(liveRefs, row.asset_ref);
1333
- if (!live && row.missing_since == null)
1334
- toStamp.push(row.asset_ref);
1335
- else if (live && row.missing_since != null)
1336
- toClear.push(row.asset_ref);
1337
- }
1338
- if (toStamp.length > 0)
1339
- stamp(toStamp, now);
1340
- if (toClear.length > 0)
1341
- clear(toClear);
1342
- const collected = collect ? deleteOlderThan(now - STATE_GC_GRACE_MS) : 0;
1343
- const pending = countPending();
1344
- return { pending, collected };
1345
- }
1346
- /**
1347
- * Orphan-state GC — #733 (Workstream C). For each of the two per-asset state
1348
- * tables (`asset_salience`, `asset_outcome`), stamps `missing_since` on refs
1349
- * that no longer resolve against `entries.item_ref` in index.db, clears the
1350
- * stamp on refs that resolve again, and — only when `improve.stateGc.collect`
1351
- * is true — deletes rows whose stamp is older than {@link STATE_GC_GRACE_MS}.
1352
- *
1353
- * "ref not present in entries.item_ref" IS the authoritative-deletion
1354
- * predicate: "absent ≠ deleted" is inherited from the indexer, not
1355
- * re-implemented here — an incomplete or failed source scan preserves that
1356
- * source's last-known-good `entries` rows (indexer.ts ~1195-1199), and
1357
- * mass-wipe is already gated upstream (`preserveExistingIndex` +
1358
- * `fullDelete && scanComplete`). A temporarily unreachable source therefore
1359
- * never surfaces candidates; no separate scan-status tracking is needed.
1360
- *
1361
- * Runs under the SAME index-writer lease / borrowed-state.db-connection
1362
- * discipline as the neighboring maintenance passes: `dbCell.current` supplies
1363
- * the already-open index.db handle (#584), and state.db access goes through
1364
- * `withStateDb(..., { borrowed: eventsCtx?.db })` so this never opens a
1365
- * second live writer alongside a long-lived `eventsCtx.db` connection — see
1366
- * the #585 comment on {@link runRetentionPurgePass}'s events-purge call for
1367
- * why that matters ("database is locked").
1368
- *
1369
- * Exported for direct test coverage (tests/integration/commands/improve/
1370
- * state-gc.test.ts), mirroring the `runMemoryInferenceMaintenancePass` /
1371
- * `runGraphExtractionMaintenancePass` / `runRetentionPurgePass` precedent;
1372
- * production callers reach it only through `runMaintenancePassesUnderLease`.
661
+ * Orphan-state GC (#733) over `asset_salience` and `asset_outcome`: stamp
662
+ * `missing_since` on refs that no longer resolve in index.db, clear it on refs
663
+ * that resolve again, and — only with `improve.stateGc.collect` — delete rows
664
+ * stamped longer ago than {@link STATE_GC_GRACE_MS}. An unreachable source keeps
665
+ * its last-known index rows, so it never surfaces candidates. `pending` is the
666
+ * backlog after the sweep; the event is emitted only when there is something to
667
+ * report.
1373
668
  */
1374
669
  export function runOrphanStateGcPass(ctx, dbCell) {
1375
- const { eventsCtx, config } = ctx;
1376
- const warnings = [];
670
+ const { eventsCtx } = ctx;
1377
671
  const indexDb = dbCell.current;
1378
- if (!indexDb) {
1379
- warnings.push("orphan state GC skipped: no index.db handle available");
1380
- return { pending: 0, collected: 0, warnings };
1381
- }
1382
- const collect = config.improve?.stateGc?.collect === true;
672
+ if (!indexDb)
673
+ return { pending: 0, collected: 0, warnings: ["orphan state GC skipped: no index.db handle available"] };
674
+ const collect = ctx.config.improve?.stateGc?.collect === true;
1383
675
  const now = Date.now();
1384
676
  let pending = 0;
1385
677
  let collected = 0;
1386
678
  try {
1387
- // R78 (tier1-0917): one query for every live item_ref, shared by both tables' sweeps
1388
- // below — replaces a `getEntryByRef` round trip per pending row. Inside
1389
- // the try so a schema mismatch (e.g. a DB version upgrade that dropped
1390
- // `entries`) degrades to the "orphan state GC failed" warning below
1391
- // instead of escaping this pass and failing the whole maintenance run.
1392
679
  const liveRefs = getLiveRefSnapshot(indexDb);
680
+ const sweep = (table) => {
681
+ const toStamp = [];
682
+ const toClear = [];
683
+ for (const row of table.rows) {
684
+ const live = isStateRefLive(liveRefs, row.asset_ref);
685
+ if (!live && row.missing_since == null)
686
+ toStamp.push(row.asset_ref);
687
+ else if (live && row.missing_since != null)
688
+ toClear.push(row.asset_ref);
689
+ }
690
+ if (toStamp.length > 0)
691
+ table.stamp(toStamp, now);
692
+ if (toClear.length > 0)
693
+ table.clear(toClear);
694
+ const removed = collect ? table.deleteOlderThan(now - STATE_GC_GRACE_MS) : 0;
695
+ return { pending: table.countPending(), collected: removed };
696
+ };
1393
697
  withStateDb((stateDb) => {
1394
- const salienceResult = gcOneStateTable({
1395
- refRows: listAssetSalienceMissingState(stateDb),
1396
- liveRefs,
1397
- now,
1398
- collect,
1399
- stamp: (refs, ts) => stampAssetSalienceMissing(stateDb, refs, ts),
1400
- clear: (refs) => clearAssetSalienceMissing(stateDb, refs),
1401
- deleteOlderThan: (cutoffMs) => deleteAssetSalienceMissingBefore(stateDb, cutoffMs),
1402
- countPending: () => countAssetSalienceMissing(stateDb),
1403
- });
1404
- const outcomeResult = gcOneStateTable({
1405
- refRows: listAssetOutcomeMissingState(stateDb),
1406
- liveRefs,
1407
- now,
1408
- collect,
1409
- stamp: (refs, ts) => stampAssetOutcomeMissing(stateDb, refs, ts),
1410
- clear: (refs) => clearAssetOutcomeMissing(stateDb, refs),
1411
- deleteOlderThan: (cutoffMs) => deleteAssetOutcomeMissingBefore(stateDb, cutoffMs),
1412
- countPending: () => countAssetOutcomeMissing(stateDb),
1413
- });
1414
- // Table-name-shaped keys ("salience"/"outcome", not the guarded
1415
- // "asset_salience"/"asset_outcome" table names) — this file sits
1416
- // outside src/storage/repositories/**, where the state-table-sql
1417
- // lint rule (#672) forbids naming those tables even in a log string
1418
- // or an object key, not just in raw SQL.
1419
- const byTable = { salience: salienceResult, outcome: outcomeResult };
1420
- pending = salienceResult.pending + outcomeResult.pending;
1421
- collected = salienceResult.collected + outcomeResult.collected;
698
+ // Keys avoid the state table names: the state-table-sql lint rule (#672)
699
+ // forbids them outside the repositories.
700
+ const byTable = {
701
+ salience: sweep({
702
+ rows: listAssetSalienceMissingState(stateDb),
703
+ stamp: (refs, ts) => stampAssetSalienceMissing(stateDb, refs, ts),
704
+ clear: (refs) => clearAssetSalienceMissing(stateDb, refs),
705
+ deleteOlderThan: (cutoff) => deleteAssetSalienceMissingBefore(stateDb, cutoff),
706
+ countPending: () => countAssetSalienceMissing(stateDb),
707
+ }),
708
+ outcome: sweep({
709
+ rows: listAssetOutcomeMissingState(stateDb),
710
+ stamp: (refs, ts) => stampAssetOutcomeMissing(stateDb, refs, ts),
711
+ clear: (refs) => clearAssetOutcomeMissing(stateDb, refs),
712
+ deleteOlderThan: (cutoff) => deleteAssetOutcomeMissingBefore(stateDb, cutoff),
713
+ countPending: () => countAssetOutcomeMissing(stateDb),
714
+ }),
715
+ };
716
+ pending = byTable.salience.pending + byTable.outcome.pending;
717
+ collected = byTable.salience.collected + byTable.outcome.collected;
1422
718
  if (pending > 0 || collected > 0) {
1423
719
  info(`[improve] orphan state GC: ${pending} pending, ${collected} collected ` +
1424
- `(salience ${salienceResult.pending}/${salienceResult.collected}, ` +
1425
- `outcome ${outcomeResult.pending}/${outcomeResult.collected})`);
1426
- // #733 — asset_state_gc reports the current per-table backlog
1427
- // snapshot (`pending`) plus this run's deletions (`collected`);
1428
- // emitted only when there is something to report (mirrors the
1429
- // rekey script's no-op-stays-silent precedent) so a perpetually
1430
- // clean stash never accumulates events.
1431
- appendEvent({
1432
- eventType: "asset_state_gc",
1433
- ref: "asset_state/_gc",
1434
- metadata: { pending, collected, byTable },
1435
- }, eventsCtx);
720
+ `(salience ${byTable.salience.pending}/${byTable.salience.collected}, ` +
721
+ `outcome ${byTable.outcome.pending}/${byTable.outcome.collected})`);
722
+ appendEvent({ eventType: "asset_state_gc", ref: "asset_state/_gc", metadata: { pending, collected, byTable } }, eventsCtx);
1436
723
  }
1437
724
  }, { path: eventsCtx?.dbPath, borrowed: eventsCtx?.db });
1438
725
  }
1439
726
  catch (err) {
1440
- warnings.push(`orphan state GC failed: ${errMessage(err)}`);
727
+ return { pending, collected, warnings: [`orphan state GC failed: ${errMessage(err)}`] };
1441
728
  }
1442
- return { pending, collected, warnings };
729
+ return { pending, collected, warnings: [] };
1443
730
  }