akm-cli 0.9.17-alpha.2 → 0.9.17-alpha.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (343) hide show
  1. package/CHANGELOG.md +756 -0
  2. package/dist/akm +94 -196
  3. package/dist/cli/shared.js +6 -2
  4. package/dist/cli.js +22 -9
  5. package/dist/commands/agent/agent-dispatch.js +1 -1
  6. package/dist/commands/command/command-execution.js +24 -62
  7. package/dist/commands/feedback-cli.js +0 -1
  8. package/dist/commands/health/accept-rate.js +2 -2
  9. package/dist/commands/health/checks.js +30 -75
  10. package/dist/commands/health/config-skew.js +38 -0
  11. package/dist/commands/health/egress.js +54 -0
  12. package/dist/commands/health/html-report.js +0 -38
  13. package/dist/commands/health/improve-metrics.js +123 -562
  14. package/dist/commands/health/plugin-staleness.js +53 -3
  15. package/dist/commands/health/renderers.js +12 -4
  16. package/dist/commands/health/report-view-model.js +11 -106
  17. package/dist/commands/health/types-improve.js +4 -19
  18. package/dist/commands/health/windows.js +64 -73
  19. package/dist/commands/health.js +122 -143
  20. package/dist/commands/improve/consolidate/chunking.js +25 -100
  21. package/dist/commands/improve/consolidate/sanitize.js +54 -149
  22. package/dist/commands/improve/consolidate.js +538 -1075
  23. package/dist/commands/improve/content-hash.js +16 -24
  24. package/dist/commands/improve/distill/content-repair.js +18 -100
  25. package/dist/commands/improve/distill-guards.js +20 -81
  26. package/dist/commands/improve/distill-promotion-policy.js +23 -243
  27. package/dist/commands/improve/distill.js +608 -1075
  28. package/dist/commands/improve/eligibility.js +126 -400
  29. package/dist/commands/improve/execution.js +3 -5
  30. package/dist/commands/improve/extract.js +487 -1046
  31. package/dist/commands/improve/feedback-valence.js +0 -25
  32. package/dist/commands/improve/improve-cli.js +29 -166
  33. package/dist/commands/improve/improve-result-file.js +10 -66
  34. package/dist/commands/improve/improve-strategies.js +12 -7
  35. package/dist/commands/improve/improve-usage-report.js +18 -64
  36. package/dist/commands/improve/improve.js +443 -1063
  37. package/dist/commands/improve/ledger.js +114 -0
  38. package/dist/commands/improve/locks.js +2 -8
  39. package/dist/commands/improve/loop-stages.js +459 -1172
  40. package/dist/commands/improve/memory/derived-ref.js +12 -77
  41. package/dist/commands/improve/memory/memory-belief.js +14 -118
  42. package/dist/commands/improve/memory/memory-improve.js +4 -3
  43. package/dist/commands/improve/outcome-loop.js +28 -156
  44. package/dist/commands/improve/planner.js +5 -10
  45. package/dist/commands/improve/preparation.js +851 -2339
  46. package/dist/commands/improve/proactive-maintenance.js +34 -101
  47. package/dist/commands/improve/reflect-noise.js +104 -280
  48. package/dist/commands/improve/reflect.js +621 -1367
  49. package/dist/commands/improve/salience.js +46 -232
  50. package/dist/commands/improve/session-asset.js +19 -100
  51. package/dist/commands/improve/stage.js +323 -0
  52. package/dist/commands/proposal/drain.js +251 -644
  53. package/dist/commands/proposal/proposal-cli.js +3 -18
  54. package/dist/commands/proposal/proposal-types.js +20 -41
  55. package/dist/commands/proposal/proposal.js +1 -2
  56. package/dist/commands/proposal/propose.js +134 -160
  57. package/dist/commands/proposal/repository.js +502 -1487
  58. package/dist/commands/proposal/validators/proposal-quality-validators.js +71 -174
  59. package/dist/commands/proposal/validators/proposal-validators.js +1 -1
  60. package/dist/commands/proposal/validators/proposals.js +13 -89
  61. package/dist/commands/read/curate.js +63 -413
  62. package/dist/commands/read/search-cli.js +16 -33
  63. package/dist/commands/read/search.js +17 -23
  64. package/dist/commands/read/show.js +2 -13
  65. package/dist/commands/sources/bundle-cli.js +25 -2
  66. package/dist/commands/sources/bundle-config-ops.js +7 -0
  67. package/dist/commands/sources/dangerous-env-audit.js +1 -2
  68. package/dist/commands/sources/info.js +2 -11
  69. package/dist/commands/sources/installed-stashes.js +197 -746
  70. package/dist/commands/sources/schema-repair.js +98 -129
  71. package/dist/commands/sources/source-add.js +62 -12
  72. package/dist/commands/sources/stash-cli.js +1 -1
  73. package/dist/commands/tasks/explain.js +10 -13
  74. package/dist/commands/tasks/tasks-cli.js +9 -8
  75. package/dist/commands/tasks/tasks.js +326 -930
  76. package/dist/commands/tasks/validate.js +42 -21
  77. package/dist/commands/workflow/plan.js +22 -29
  78. package/dist/commands/workflow-cli.js +4 -4
  79. package/dist/core/adapter/adapters/akm-adapter.js +0 -1
  80. package/dist/core/adapter/adapters/akm-lint.js +2 -3
  81. package/dist/core/adapter/adapters/akm-metadata.js +11 -12
  82. package/dist/core/adapter/adapters/akm-workflow-adapter.js +1 -1
  83. package/dist/core/adapter/execution-source.js +17 -29
  84. package/dist/core/asset/resolve-ref.js +1 -1
  85. package/dist/core/bundle-id.js +42 -5
  86. package/dist/core/bundle-rename.js +291 -0
  87. package/dist/core/config/config-io.js +1 -2
  88. package/dist/core/config/config-schema.js +1 -33
  89. package/dist/core/config/config-walker.js +1 -1
  90. package/dist/core/config/config.js +163 -68
  91. package/dist/core/config/legacy-source-shape-shim.js +38 -9
  92. package/dist/core/config/schema/embedding.js +20 -5
  93. package/dist/core/config/schema/engines.js +5 -0
  94. package/dist/core/config/schema/execution.js +1 -1
  95. package/dist/core/config/schema/experimental.js +1 -1
  96. package/dist/core/config/schema/improve-processes.js +21 -95
  97. package/dist/core/config/schema/improve.js +4 -42
  98. package/dist/core/config/schema/scheduler.js +12 -12
  99. package/dist/core/config/schema/search.js +6 -22
  100. package/dist/core/env-secret-ref.js +0 -1
  101. package/dist/core/errors.js +8 -9
  102. package/dist/core/file-lock.js +76 -173
  103. package/dist/core/logs-db.js +2 -2
  104. package/dist/core/paths.js +0 -27
  105. package/dist/core/redaction.js +109 -2
  106. package/dist/core/run-lock.js +2 -5
  107. package/dist/core/spawn-env.js +1 -1
  108. package/dist/core/state/migrations.js +108 -61
  109. package/dist/core/state-db-scope.js +2 -4
  110. package/dist/core/state-db.js +126 -692
  111. package/dist/core/type-presentation.js +1 -9
  112. package/dist/core/write-source.js +293 -1012
  113. package/dist/execution/input-contract.js +1 -1
  114. package/dist/execution/resolved-request.js +135 -689
  115. package/dist/execution/source.js +63 -257
  116. package/dist/execution/target-ref.js +1 -1
  117. package/dist/indexer/bundle-identity-guard.js +2 -2
  118. package/dist/indexer/db/graph-db.js +106 -46
  119. package/dist/indexer/ensure-index.js +44 -85
  120. package/dist/indexer/graph/graph-extraction.js +340 -562
  121. package/dist/indexer/graph/graph-related.js +130 -0
  122. package/dist/indexer/index-rebuild-lock.js +3 -11
  123. package/dist/indexer/index-writer-lock.js +8 -17
  124. package/dist/indexer/index-written-assets.js +139 -151
  125. package/dist/indexer/indexer.js +524 -846
  126. package/dist/indexer/materialize-embeddings.js +60 -397
  127. package/dist/indexer/passes/memory-inference.js +81 -90
  128. package/dist/indexer/passes/metadata.js +132 -200
  129. package/dist/indexer/read-preflight.js +0 -7
  130. package/dist/indexer/scan/doc-to-entry.js +1 -3
  131. package/dist/indexer/scan/drain-dir.js +1 -1
  132. package/dist/indexer/search/db-search.js +181 -590
  133. package/dist/indexer/search/fts-query.js +30 -41
  134. package/dist/indexer/search/ranking.js +28 -154
  135. package/dist/indexer/search/search-attribution.js +12 -32
  136. package/dist/indexer/search/search-fields.js +11 -15
  137. package/dist/indexer/search/search-hit-enrichers.js +54 -85
  138. package/dist/indexer/search/search-source.js +1 -4
  139. package/dist/indexer/usage/usage-events.js +2 -7
  140. package/dist/integrations/agent/engine-fallback.js +23 -40
  141. package/dist/integrations/agent/engine-resolution.js +93 -183
  142. package/dist/integrations/agent/execution.js +507 -0
  143. package/dist/integrations/agent/model-map.js +28 -156
  144. package/dist/integrations/agent/request-lowering.js +66 -141
  145. package/dist/integrations/agent/runner-dispatch.js +143 -321
  146. package/dist/integrations/agent/runner.js +54 -14
  147. package/dist/integrations/lockfile.js +53 -101
  148. package/dist/llm/embedders/deterministic.js +2 -3
  149. package/dist/llm/embedders/profile.js +71 -0
  150. package/dist/llm/embedders/remote.js +10 -15
  151. package/dist/llm/graph-extract.js +3 -12
  152. package/dist/llm/index-passes.js +3 -5
  153. package/dist/llm/memory-infer.js +1 -2
  154. package/dist/llm/metadata-enhance.js +1 -2
  155. package/dist/llm/structured-call.js +5 -24
  156. package/dist/output/generic-render.js +23 -11
  157. package/dist/output/html-render.js +13 -10
  158. package/dist/output/render-registry.js +3 -32
  159. package/dist/output/shapes/helpers.js +2 -34
  160. package/dist/output/shapes/passthrough.js +1 -9
  161. package/dist/{indexer/search/ranking-types.js → output/text/bundle-rename.js} +4 -1
  162. package/dist/output/text/command-format.js +60 -23
  163. package/dist/output/text/helpers.js +1 -1
  164. package/dist/output/text/migrate.js +5 -14
  165. package/dist/output/text/proposal-format.js +1 -2
  166. package/dist/output/text/workflow-format.js +0 -32
  167. package/dist/output/text.js +2 -0
  168. package/dist/registry/factory.js +4 -19
  169. package/dist/registry/network.js +66 -220
  170. package/dist/registry/providers/index.js +0 -2
  171. package/dist/registry/providers/skills-sh.js +3 -14
  172. package/dist/registry/providers/static-index.js +24 -26
  173. package/dist/registry/resolve.js +55 -131
  174. package/dist/scripts/akm-migrate-node.js +43937 -93313
  175. package/dist/scripts/akm-migrate.js +43697 -93071
  176. package/dist/setup/registry-stash-loader.js +4 -13
  177. package/dist/setup/semantic-assets.js +3 -44
  178. package/dist/setup/setup.js +1 -1
  179. package/dist/setup/steps/tasks.js +25 -15
  180. package/dist/sources/provider-factory.js +17 -18
  181. package/dist/sources/providers/filesystem.js +2 -3
  182. package/dist/sources/providers/git-install.js +7 -1
  183. package/dist/sources/providers/git-provider.js +0 -3
  184. package/dist/sources/providers/git-stash.js +0 -17
  185. package/dist/sources/providers/npm.js +2 -4
  186. package/dist/sources/providers/provider-utils.js +5 -10
  187. package/dist/sources/providers/website.js +0 -2
  188. package/dist/sources/snapshot-fetchers/website-ingest.js +1 -1
  189. package/dist/sources/website-url.js +2 -2
  190. package/dist/storage/database.js +9 -35
  191. package/dist/storage/repositories/improve-ledger-repository.js +168 -0
  192. package/dist/storage/repositories/index-connection.js +34 -70
  193. package/dist/storage/repositories/index-entries-repository.js +69 -111
  194. package/dist/storage/repositories/index-entry-mapper.js +1 -2
  195. package/dist/storage/repositories/index-entry-schema.js +83 -269
  196. package/dist/storage/repositories/index-fts-repository.js +86 -256
  197. package/dist/storage/repositories/index-llm-cache-repository.js +17 -0
  198. package/dist/storage/repositories/index-meta-repository.js +6 -4
  199. package/dist/storage/repositories/index-schema.js +192 -220
  200. package/dist/storage/repositories/index-utility-repository.js +8 -29
  201. package/dist/storage/repositories/index-vec-repository.js +133 -414
  202. package/dist/storage/repositories/outcome-repository.js +2 -1
  203. package/dist/storage/repositories/proposals-repository.js +35 -0
  204. package/dist/storage/repositories/registry-index-cache-repository.js +100 -0
  205. package/dist/storage/repositories/task-history-repository.js +26 -4
  206. package/dist/storage/repositories/workflow-runs-repository.js +53 -244
  207. package/dist/storage/sqlite-migrations.js +136 -0
  208. package/dist/storage/sqlite-pragmas.js +11 -9
  209. package/dist/storage/sqlite-transaction.js +170 -0
  210. package/dist/storage/state-db-integrity.js +34 -27
  211. package/dist/tasks/activation-config.js +134 -62
  212. package/dist/tasks/backends/cron.js +129 -277
  213. package/dist/tasks/backends/exec-utils.js +2 -5
  214. package/dist/tasks/backends/launchd.js +125 -745
  215. package/dist/tasks/backends/schtasks.js +101 -620
  216. package/dist/tasks/prepare/prepare-support.js +5 -15
  217. package/dist/tasks/prepare/prepare.js +0 -2
  218. package/dist/tasks/resolve-akm-bin.js +20 -79
  219. package/dist/tasks/run/attempt-lifecycle.js +0 -1
  220. package/dist/tasks/scheduler-binding.js +18 -238
  221. package/dist/tasks/scheduler-invocation.js +52 -52
  222. package/dist/tasks/scheduler-lock.js +53 -0
  223. package/dist/tasks/scheduler-sync.js +363 -679
  224. package/dist/tasks/source/parse-task-source.js +160 -10
  225. package/dist/tasks/source/task-source-v3-frozen.js +3 -4
  226. package/dist/tasks/source/task-to-v4.js +2 -2
  227. package/dist/workflows/authoring/authoring.js +3 -12
  228. package/dist/workflows/compile.js +211 -0
  229. package/dist/workflows/concurrency-policy.js +13 -74
  230. package/dist/workflows/exec/child-invocation.js +3 -17
  231. package/dist/workflows/exec/child-workflow.js +32 -141
  232. package/dist/workflows/exec/dispatch-redaction.js +13 -53
  233. package/dist/workflows/exec/environment.js +98 -0
  234. package/dist/workflows/exec/exec-unit.js +33 -140
  235. package/dist/workflows/exec/frozen-judge.js +7 -59
  236. package/dist/workflows/exec/native-executor.js +82 -341
  237. package/dist/workflows/exec/param-secrets.js +29 -47
  238. package/dist/workflows/exec/run-workflow.js +154 -387
  239. package/dist/workflows/exec/scheduler.js +9 -36
  240. package/dist/workflows/exec/step-work.js +127 -430
  241. package/dist/workflows/exec/unit-dispatch.js +11 -63
  242. package/dist/workflows/exec/unit-writer.js +8 -52
  243. package/dist/workflows/exec/worktree.js +39 -273
  244. package/dist/workflows/freeze/child-output-references.js +4 -15
  245. package/dist/workflows/freeze/environment.js +99 -92
  246. package/dist/workflows/freeze/freeze.js +172 -0
  247. package/dist/workflows/freeze/step-values.js +19 -21
  248. package/dist/workflows/freeze/targets/child-workflow.js +23 -92
  249. package/dist/workflows/freeze/targets/command.js +10 -33
  250. package/dist/workflows/freeze/targets/script.js +5 -12
  251. package/dist/workflows/freeze/targets/shell.js +3 -6
  252. package/dist/workflows/freeze/targets/task.js +25 -80
  253. package/dist/workflows/freeze/task-bindings.js +20 -67
  254. package/dist/workflows/{source-ir/github-yaml.js → github-yaml.js} +88 -206
  255. package/dist/workflows/ir/params.js +6 -51
  256. package/dist/workflows/ir/plan-hash.js +2 -34
  257. package/dist/workflows/parser.js +140 -43
  258. package/dist/{commands/improve/consolidate/types.js → workflows/plan.js} +2 -1
  259. package/dist/workflows/renderer.js +36 -69
  260. package/dist/workflows/resource-limits.js +12 -120
  261. package/dist/workflows/runtime/agent-identity.js +8 -40
  262. package/dist/workflows/runtime/run-outputs.js +3 -6
  263. package/dist/workflows/runtime/run-plan.js +316 -0
  264. package/dist/workflows/runtime/runs.js +48 -200
  265. package/dist/workflows/runtime/workflow-asset-loader.js +24 -57
  266. package/dist/workflows/{source-ir/semantics.js → source-semantics.js} +16 -20
  267. package/dist/workflows/validate-summary.js +2 -7
  268. package/docs/integration/bundling-akm.md +49 -42
  269. package/docs/migration/README.md +1 -0
  270. package/docs/migration/release-notes/0.9.17.md +41 -0
  271. package/docs/migration/v0.9.1-to-v0.9.2.md +19 -7
  272. package/docs/reference/cli.md +182 -125
  273. package/docs/reference/configuration.md +49 -56
  274. package/docs/reference/data-and-telemetry.md +19 -20
  275. package/docs/reference/tasks.md +86 -38
  276. package/docs/reference/workflow-schema.md +14 -18
  277. package/docs/reference/workflows.md +6 -9
  278. package/package.json +1 -1
  279. package/schemas/akm-config.json +87 -406
  280. package/dist/commands/health/advisories.js +0 -150
  281. package/dist/commands/health/metrics.js +0 -329
  282. package/dist/commands/health/surfaces.js +0 -102
  283. package/dist/commands/improve/anti-collapse.js +0 -83
  284. package/dist/commands/improve/collapse-detector.js +0 -432
  285. package/dist/commands/improve/consolidate/eligibility.js +0 -48
  286. package/dist/commands/improve/consolidate/merge.js +0 -146
  287. package/dist/commands/improve/distill/promote-memory.js +0 -329
  288. package/dist/commands/improve/distill/quality-gate.js +0 -500
  289. package/dist/commands/improve/memory/memory-contradiction-detect.js +0 -291
  290. package/dist/commands/improve/proposal-envelope.js +0 -31
  291. package/dist/commands/improve/run-context.js +0 -123
  292. package/dist/commands/improve/shared.js +0 -21
  293. package/dist/commands/improve/source-identity.js +0 -28
  294. package/dist/commands/improve/triage.js +0 -96
  295. package/dist/commands/proposal/drain-policies.js +0 -151
  296. package/dist/commands/sources/update-transaction.js +0 -220
  297. package/dist/core/action-contributors.js +0 -28
  298. package/dist/core/config/config-version-shim.js +0 -101
  299. package/dist/core/config/retired-experimental-keys-shim.js +0 -62
  300. package/dist/core/fs-txn.js +0 -405
  301. package/dist/core/lexical-score.js +0 -25
  302. package/dist/core/maintenance-barrier.js +0 -167
  303. package/dist/execution/executable-identity.js +0 -105
  304. package/dist/execution/guarded-source.js +0 -427
  305. package/dist/indexer/graph/graph-boost.js +0 -427
  306. package/dist/indexer/graph/graph-dedup.js +0 -95
  307. package/dist/indexer/search/name-match.js +0 -35
  308. package/dist/indexer/search/ranking-contributors.js +0 -515
  309. package/dist/indexer/walk/project-context.js +0 -192
  310. package/dist/integrations/agent/execution-cascade.js +0 -566
  311. package/dist/integrations/agent/execution-definitions.js +0 -202
  312. package/dist/integrations/agent/execution-lowering.js +0 -841
  313. package/dist/integrations/agent/execution-preparation.js +0 -98
  314. package/dist/integrations/agent/inline-execution.js +0 -74
  315. package/dist/registry/create-provider-registry.js +0 -29
  316. package/dist/registry/pinned-request-helper.js +0 -247
  317. package/dist/registry/pinned-transport.js +0 -717
  318. package/dist/sources/providers/index.js +0 -14
  319. package/dist/storage/engines/sqlite-migrations.js +0 -271
  320. package/dist/storage/repositories/canaries-repository.js +0 -107
  321. package/dist/storage/repositories/embedding-salvage-repository.js +0 -184
  322. package/dist/storage/repositories/registry-cache.js +0 -113
  323. package/dist/tasks/scheduler-sync-preview.js +0 -52
  324. package/dist/workflows/freeze/resolve-steps.js +0 -86
  325. package/dist/workflows/freeze/source-freeze.js +0 -64
  326. package/dist/workflows/ir/compile.js +0 -321
  327. package/dist/workflows/ir/environment-v4.js +0 -330
  328. package/dist/workflows/ir/freeze-v4.js +0 -153
  329. package/dist/workflows/ir/schema-v4.js +0 -745
  330. package/dist/workflows/ir/schema.js +0 -354
  331. package/dist/workflows/program/schema.js +0 -78
  332. package/dist/workflows/runtime/checkin.js +0 -57
  333. package/dist/workflows/runtime/plan-classifier.js +0 -196
  334. package/dist/workflows/runtime/unit-checkin.js +0 -45
  335. package/dist/workflows/runtime/unit-phases.js +0 -20
  336. package/dist/workflows/schema.js +0 -4
  337. package/dist/workflows/source-ir/compile.js +0 -200
  338. package/dist/workflows/source-ir/program.js +0 -50
  339. package/dist/workflows/source-ir/result.js +0 -26
  340. package/dist/workflows/source-ir/schema.js +0 -786
  341. package/dist/workflows/source-ir/triggers.js +0 -79
  342. package/dist/workflows/source-ir/uses.js +0 -40
  343. package/dist/workflows/validator.js +0 -60
@@ -2,37 +2,24 @@
2
2
  // License, v. 2.0. If a copy of the MPL was not distributed with this
3
3
  // file, You can obtain one at https://mozilla.org/MPL/2.0/.
4
4
  /**
5
- * `akm extract` — session-insight extractor.
5
+ * `akm extract` — read native session logs (claude, opencode) through the
6
+ * session-log harnesses, pre-filter the noise, and ask the model for
7
+ * memory/lesson/knowledge candidates the agent did not already save. Each
8
+ * candidate is queued as a proposal (`source: "extract"`), never written.
6
9
  *
7
- * Replaces the akm-plugin session-checkpoint hook with an on-demand extractor
8
- * that reads native session files (claude JSONL, opencode storage tree)
9
- * through the {@link SessionLogHarness} registry, pre-filters noise, and asks
10
- * a bounded in-tree LLM to produce candidate memory/lesson/knowledge proposals
11
- * for content the agent did NOT preserve via inline `akm remember`/`akm feedback`.
12
- *
13
- * Architectural notes:
14
- * - Stateless. All file/LLM access goes through injectable seams so tests
15
- * never touch a real platform.
16
- * - Bounded LLM call routed through `callStructured`. Improve-stage
17
- * enablement comes from the active strategy; explicit `akm extract` always
18
- * runs regardless of that stage toggle.
19
- * - Proposals routed via `createProposal({ source: "extract", ... })` — the
20
- * same review queue as reflect / distill / consolidate. Never direct-write.
21
- * - Per-candidate body assembly merges description (+ when_to_use for lessons)
22
- * into the body's YAML frontmatter so the accept-time
23
- * descriptionQualityValidator passes — same pattern as the
24
- * consolidate-writer fix.
10
+ * A session is skipped with zero LLM calls when its content hash is unchanged
11
+ * since the last extraction, when it is nearly empty, or when the optional
12
+ * heuristic triage scores it below threshold.
25
13
  */
26
14
  import fs from "node:fs";
27
15
  import path from "node:path";
28
16
  import { assembleAsset } from "../../core/asset/asset-serialize.js";
29
- import { timestampForFilename } from "../../core/common.js";
17
+ import { resolveStashDir, timestampForFilename } from "../../core/common.js";
30
18
  import { getImproveProcessConfig, loadConfig } from "../../core/config/config.js";
31
19
  import { ConfigError, UsageError } from "../../core/errors.js";
32
20
  import { appendEvent } from "../../core/events.js";
33
21
  import { createLockPayload, probeLock, reclaimStaleLock, releaseLock, tryAcquireLockSync, } from "../../core/file-lock.js";
34
22
  import { EXTRACT_INFRASTRUCTURE_SKIP_REASONS } from "../../core/improve-types.js";
35
- import { tryAcquireMaintenanceBarrier } from "../../core/maintenance-barrier.js";
36
23
  import { redactErrorBody } from "../../core/redaction.js";
37
24
  import { resolveStashStandards } from "../../core/standards/resolve-stash-standards.js";
38
25
  import { resolveTypeConventions, typeConventionRef } from "../../core/standards/resolve-type-conventions.js";
@@ -42,63 +29,33 @@ import { repairTruncatedDescription } from "../../core/text-truncation.js";
42
29
  import { DURATION_UNITS, parseDuration } from "../../core/time.js";
43
30
  import { warn, warnVerbose } from "../../core/warn.js";
44
31
  import { indexWrittenAssets } from "../../indexer/index-written-assets.js";
45
- import { disposeLoweredExecutionDispatchLease, } from "../../integrations/agent/execution-lowering.js";
32
+ import { assertRunnerCredentials } from "../../integrations/agent/runner-dispatch.js";
46
33
  import { getAvailableHarnesses } from "../../integrations/session-logs/index.js";
47
34
  import { preFilterSession } from "../../integrations/session-logs/pre-filter.js";
48
35
  import { isJsonSchemaKnownUnsupported } from "../../llm/client.js";
49
- import { callStructured, preflightStructuredLlmRunner } from "../../llm/structured-call.js";
50
- import { sha256Hex } from "../../runtime.js";
51
36
  import { getExtractedSessionsMap, getLastExtractRunAt, shouldSkipAlreadyExtractedSession, upsertExtractedSession, } from "../../storage/repositories/extract-sessions-repository.js";
52
37
  import { openSqliteReadSnapshot } from "../../storage/sqlite-read-snapshot.js";
53
- import { isProposalSkipped } from "../proposal/repository.js";
38
+ import { contentHash } from "./content-hash.js";
54
39
  import { resolveImproveLlmExecution } from "./execution.js";
55
40
  import { buildExtractPrompt, EXTRACT_JSON_SCHEMA, parseExtractPayload, } from "./extract-prompt.js";
56
- import { resolveImproveStrategy, resolveProcessEnabled } from "./improve-strategies.js";
57
- import { emitProposal } from "./proposal-envelope.js";
58
- import { createRunContext, resolveRunStashDir } from "./run-context.js";
41
+ import { cloneAndFreeze, resolveImproveStrategy, resolveProcessEnabled } from "./improve-strategies.js";
42
+ import { isLedgerBlocked, ledgerKey, loadLedgerSnapshot } from "./ledger.js";
59
43
  import { buildSessionSummaryPrompt, parseSessionSummary, SESSION_SUMMARY_JSON_SCHEMA, sessionMeetsDurationGate, writeSessionAsset, } from "./session-asset.js";
60
- import { resolveTriageConfig, scoreSessionTriage } from "./triage.js";
61
- /** Default minimum session duration (minutes) for session indexing (#561). */
44
+ import { callStage, mintProposal, noticeSet } from "./stage.js";
45
+ /** Minimum session duration (minutes) for writing a session asset. */
62
46
  const DEFAULT_MIN_SESSION_DURATION_MINUTES = 5;
63
- /**
64
- * Default minimum raw session size (chars) below which the extract LLM call is
65
- * skipped (#595/#596). Deliberately tiny: analysis of 218 candidate-producing
66
- * sessions showed sessions of 22–368 raw chars regularly yield 1–5 candidates,
67
- * so size is not a reliable proxy for value — only truly empty sessions
68
- * (0 chars, journal files) are safe to skip.
69
- */
47
+ /** Raw session size (chars) below which the LLM call is skipped; only truly empty sessions are safe to skip. */
70
48
  const DEFAULT_MIN_CONTENT_CHARS = 10;
71
- /**
72
- * Default cap on NEW sessions the extract pass will LLM-process in a single run
73
- * (`processes.extract.maxSessionsPerRun` overrides; `0` disables). Bounds per-run
74
- * wall time + token spend so a backlog of accumulated sessions can't run a single
75
- * pass past its scheduled-task timeout. Overflow sessions stay unseen and are
76
- * processed by subsequent runs, so coverage is preserved — just spread out.
77
- */
49
+ /** New sessions LLM-processed per run (`processes.extract.maxSessionsPerRun`, 0 disables); the rest wait for later runs. */
78
50
  const DEFAULT_MAX_SESSIONS_PER_RUN = 25;
79
51
  /**
80
- * Floor for the default discovery window (48h). When no explicit `--since` /
81
- * `defaultSince` is configured, discovery looks back to the LAST recorded
82
- * extract run for the harness (so an intermittently-online host that was off for
83
- * days still rediscovers sessions that ended during the gap), but never LESS
84
- * than this — looking back less than the prior window could drop a session that
85
- * a previous run deferred via `maxSessionsPerRun`. Widening is free of redundant
86
- * LLM cost: the content-hash ledger skips unchanged sessions with zero LLM calls.
52
+ * Without an explicit window, discovery looks back to the last extract run for
53
+ * the harness (a host that was off still finds sessions that ended meanwhile),
54
+ * but never less than 48h. The content-hash ledger makes the overlap free.
87
55
  */
88
56
  const DEFAULT_SINCE_FLOOR_MS = 48 * 60 * 60 * 1000;
89
- /**
90
- * Staleness window for the per-session extract lock. A single session's
91
- * processing is bounded by the per-session LLM timeout (default 60s) plus the
92
- * session-summary call, so a lock older than this must belong to a crashed
93
- * holder and is safe to reclaim.
94
- */
57
+ /** A per-session lock older than this belongs to a crashed holder. */
95
58
  const EXTRACT_SESSION_LOCK_STALE_MS = 5 * 60 * 1000;
96
- /**
97
- * Resolve the discovery `sinceMs` cutoff when no explicit `since`/`defaultSince`
98
- * is set: the later of (last recorded extract run for this harness) and
99
- * (now − 48h). See {@link DEFAULT_SINCE_FLOOR_MS}. Best-effort — any state.db
100
- * error falls back to the 48h floor.
101
- */
102
59
  function resolveDefaultSinceMs(harnessName, now, opts) {
103
60
  const floor = now - DEFAULT_SINCE_FLOOR_MS;
104
61
  if (opts.skipTracking)
@@ -122,30 +79,22 @@ function resolveDefaultSinceMs(harnessName, now, opts) {
122
79
  snapshot?.close();
123
80
  }
124
81
  }
125
- /** Filesystem-safe per-session lock path, co-located with the state.db. */
126
- function getExtractSessionLockPath(harness, sessionId, stateDbPath) {
82
+ function extractSessionLockPath(harness, sessionId, stateDbPath) {
127
83
  const safe = `${harness}-${sessionId}`.replace(/[^A-Za-z0-9._-]/g, "_");
128
84
  return path.join(path.dirname(stateDbPath), "extract-locks", `extract-${safe}.lock`);
129
85
  }
130
86
  function extractSessionLockIsUnavailable(harness, sessionId, stateDbPath) {
131
- const lockPath = getExtractSessionLockPath(harness, sessionId, stateDbPath);
132
- const probe = probeLock(lockPath, { staleAfterMs: EXTRACT_SESSION_LOCK_STALE_MS });
87
+ const probe = probeLock(extractSessionLockPath(harness, sessionId, stateDbPath), {
88
+ staleAfterMs: EXTRACT_SESSION_LOCK_STALE_MS,
89
+ });
133
90
  return probe.state === "held" || probe.state === "inaccessible";
134
91
  }
135
92
  /**
136
- * Try to claim the per-session extract lock so a concurrent extract (e.g. a
137
- * session-end hook firing `--session-id` while the hourly improve pass runs
138
- * discovery) cannot double-process the SAME session — duplicate LLM spend and
139
- * near-duplicate proposals. Reclaims a stale lock (dead holder PID or age past
140
- * {@link EXTRACT_SESSION_LOCK_STALE_MS}). Returns false when another LIVE run
141
- * holds it — the caller then skips the session without any LLM call. Best-effort:
142
- * any filesystem error resolves to `true` (proceed) so locking never blocks
143
- * extraction outright.
93
+ * Claim a session so a concurrent extract (a session-end hook racing the
94
+ * hourly improve run) cannot process it twice. A stale lock is reclaimed; a
95
+ * filesystem error proceeds, so locking never blocks extraction outright.
144
96
  */
145
97
  function acquireExtractSessionLock(lockPath) {
146
- const releaseBarrier = tryAcquireMaintenanceBarrier();
147
- if (!releaseBarrier)
148
- return { proceed: false };
149
98
  try {
150
99
  fs.mkdirSync(path.dirname(lockPath), { recursive: true });
151
100
  let ownership = tryAcquireLockSync(lockPath, createLockPayload());
@@ -154,7 +103,6 @@ function acquireExtractSessionLock(lockPath) {
154
103
  const probe = probeLock(lockPath, { staleAfterMs: EXTRACT_SESSION_LOCK_STALE_MS });
155
104
  if (probe.state === "held")
156
105
  return { proceed: false };
157
- // Absent (released between attempt + probe) or successfully reclaimed stale lock → retry once.
158
106
  if (probe.state === "stale" && !reclaimStaleLock(lockPath, probe))
159
107
  return { proceed: false };
160
108
  ownership = tryAcquireLockSync(lockPath, createLockPayload());
@@ -163,38 +111,22 @@ function acquireExtractSessionLock(lockPath) {
163
111
  catch {
164
112
  return { proceed: true };
165
113
  }
166
- finally {
167
- releaseBarrier();
168
- }
169
114
  }
170
- function cloneAndFreeze(value) {
171
- const clone = structuredClone(value);
172
- const freeze = (item) => {
173
- if (typeof item !== "object" || item === null || Object.isFrozen(item))
174
- return;
175
- for (const child of Object.values(item))
176
- freeze(child);
177
- Object.freeze(item);
178
- };
179
- freeze(clone);
180
- return clone;
181
- }
182
- /** Resolve standalone extract selection once before discovery, auto iteration, or watch startup. */
115
+ /** Resolve standalone extract selection once, before discovery, auto iteration or watch startup. */
183
116
  export function resolveStandaloneExtractPlan(config, selection) {
184
117
  if (selection.engine && selection.strategy) {
185
118
  throw new UsageError("--engine and --strategy are mutually exclusive. Pick one.", "INVALID_FLAG_VALUE");
186
119
  }
187
120
  const selected = resolveImproveStrategy(selection.strategy, config);
188
121
  const process = cloneAndFreeze(getImproveProcessConfig("extract", selected.config) ?? {});
189
- const invocation = {
190
- ...(selection.engine ? { engine: selection.engine } : {}),
191
- ...(Object.hasOwn(selection, "timeoutMs") ? { timeoutMs: selection.timeoutMs ?? null } : {}),
192
- };
193
122
  const resolved = resolveImproveLlmExecution({
194
123
  config,
195
124
  profile: selected.config,
196
125
  process,
197
- current: invocation,
126
+ current: {
127
+ ...(selection.engine ? { engine: selection.engine } : {}),
128
+ ...(Object.hasOwn(selection, "timeoutMs") ? { timeoutMs: selection.timeoutMs ?? null } : {}),
129
+ },
198
130
  processName: "extract",
199
131
  });
200
132
  if (!resolved) {
@@ -204,8 +136,7 @@ export function resolveStandaloneExtractPlan(config, selection) {
204
136
  return Object.freeze({
205
137
  strategy: selected.name,
206
138
  engine: runner.engine,
207
- // `akm extract` is an explicit operation. The strategy supplies behavior,
208
- // but its improve-stage enablement gate does not disable this command.
139
+ // An explicit `akm extract` runs regardless of the strategy's improve-stage toggle.
209
140
  enabled: true,
210
141
  process,
211
142
  runner: cloneAndFreeze(runner),
@@ -214,8 +145,15 @@ export function resolveStandaloneExtractPlan(config, selection) {
214
145
  ...(resolved.notices.length > 0 ? { notices: cloneAndFreeze(resolved.notices) } : {}),
215
146
  });
216
147
  }
217
- // ── Helpers ──────────────────────────────────────────────────────────────────
218
- /** An extract envelope for a run that processed no sessions; `engine`/`engineKind` only once a runner is resolved. */
148
+ const NO_PREFILTER = { inputCount: 0, outputCount: 0, truncatedCount: 0 };
149
+ function sessionOutcome(sessionId, harness, fields) {
150
+ return { sessionId, harness, candidateCount: 0, proposalIds: [], preFilter: NO_PREFILTER, warnings: [], ...fields };
151
+ }
152
+ function preFilterStats(filtered) {
153
+ const { inputCount, outputCount, truncatedCount } = filtered.stats;
154
+ return { inputCount, outputCount, truncatedCount };
155
+ }
156
+ /** An extract envelope for a run that processed no sessions. */
219
157
  function emptyExtractResult(args) {
220
158
  return {
221
159
  schemaVersion: 1,
@@ -234,23 +172,13 @@ function emptyExtractResult(args) {
234
172
  };
235
173
  }
236
174
  /**
237
- * Parse a since-string into an absolute ms-epoch cutoff. Accepts:
238
- * - ISO timestamps (parsed via Date.parse)
239
- * - Relative durations: `<n>m`, `<n>h`, `<n>d` (minutes / hours / days)
240
- *
241
- * Throws UsageError on unparseable input so the CLI surfaces a clear error
242
- * rather than silently defaulting.
243
- *
244
- * The recognizer is deliberately CASE-INSENSITIVE and whitespace-tolerant —
245
- * `5M` means 5 MINUTES here, diverging from the core grammar's case-sensitive
246
- * `M`=months (pinned by tests/commands/goldens-duration-flags.test.ts); only
247
- * the unit arithmetic is delegated to the canonical {@link DURATION_UNITS}
248
- * table via {@link parseDuration}.
175
+ * A since-string as an epoch-ms cutoff: an ISO timestamp or `<n>m|h|d`
176
+ * (case-insensitive here — `5M` is five minutes). Default 24h; anything else
177
+ * is a usage error.
249
178
  */
250
179
  export function parseSinceArg(value, now = Date.now()) {
251
- if (!value || value.trim() === "") {
252
- return now - 24 * 60 * 60 * 1000; // default: 24h
253
- }
180
+ if (!value || value.trim() === "")
181
+ return now - 24 * 60 * 60 * 1000;
254
182
  const trimmed = value.trim();
255
183
  const relMatch = trimmed.match(/^(\d+)\s*([mhd])$/i);
256
184
  if (relMatch) {
@@ -264,35 +192,17 @@ export function parseSinceArg(value, now = Date.now()) {
264
192
  throw new UsageError(`--since value "${value}" could not be parsed (expected ISO timestamp or duration like 24h / 7d / 30m)`, "INVALID_FLAG_VALUE");
265
193
  }
266
194
  /**
267
- * Resolve a harness instance for the given type, either from the explicit
268
- * `harnesses` seam or the {@link getAvailableHarnesses} registry. Returns
269
- * `undefined` when no harness matches (the caller surfaces that as a warning).
270
- */
271
- function resolveHarness(type, harnesses) {
272
- const pool = harnesses ?? getAvailableHarnesses();
273
- return pool.find((h) => h.name === type);
274
- }
275
- /**
276
- * Build the ref + content for a candidate. The body must contain a
277
- * frontmatter block carrying `description` (and `when_to_use` for lessons)
278
- * so the accept-time descriptionQualityValidator passes — same pattern as
279
- * the consolidate-writer fix at consolidate.ts.
195
+ * A candidate's ref and content. `description` (and a lesson's `when_to_use`)
196
+ * go into the body's frontmatter so accept-time validation sees them.
280
197
  */
281
198
  function buildCandidateProposal(candidate, sourceRef, sessionAssetRef) {
282
199
  const ref = deriveExtractCandidateRef(candidate, sourceRef);
283
- // Post-generation repair pass (#556): deterministically complete a
284
- // description the LLM sliced mid-sentence before it reaches the
285
- // auto-accept validators. No-op (byte-identical) for valid descriptions.
200
+ // Complete a description the model cut mid-sentence (no-op for valid ones).
286
201
  const description = repairTruncatedDescription(candidate.description, candidate.body);
287
- const fm = {
288
- description,
289
- ...(sessionAssetRef ? { xrefs: [sessionAssetRef] } : {}),
290
- };
291
- if (candidate.type === "lesson" && candidate.when_to_use) {
202
+ const fm = { description, ...(sessionAssetRef ? { xrefs: [sessionAssetRef] } : {}) };
203
+ if (candidate.type === "lesson" && candidate.when_to_use)
292
204
  fm.when_to_use = candidate.when_to_use;
293
- }
294
- const content = assembleAsset(fm, candidate.body);
295
- return { ref, content, description };
205
+ return { ref, content: assembleAsset(fm, candidate.body), description };
296
206
  }
297
207
  function canonicalSegment(value) {
298
208
  return value
@@ -303,13 +213,11 @@ function canonicalSegment(value) {
303
213
  .replace(/^-|-$/g, "");
304
214
  }
305
215
  export function deriveExtractCandidateRef(candidate, sourceRef) {
306
- const candidateParts = candidate.name.split("/").map(canonicalSegment).filter(Boolean);
307
- const leaf = candidateParts.at(-1) ?? "extracted-insight";
216
+ const leaf = candidate.name.split("/").map(canonicalSegment).filter(Boolean).at(-1) ?? "extracted-insight";
308
217
  if (candidate.type === "memory" || candidate.type === "lesson") {
309
218
  const projectName = sourceRef.projectHint?.split(/[\\/]/).filter(Boolean).at(-1);
310
219
  const scope = projectName ? canonicalSegment(projectName) : "";
311
- const subdir = candidate.type === "memory" ? "memories" : "lessons";
312
- return `${subdir}/${scope ? `${scope}/` : ""}${leaf}`;
220
+ return `${candidate.type === "memory" ? "memories" : "lessons"}/${scope ? `${scope}/` : ""}${leaf}`;
313
221
  }
314
222
  return `knowledge/${leaf}`;
315
223
  }
@@ -326,222 +234,120 @@ function resolveExtractStandards(stashDir) {
326
234
  return sections.join("\n\n");
327
235
  }
328
236
  /**
329
- * Canonicalize a session's content into a single deterministic string for
330
- * hashing (#602). Each event is rendered `<role>\n<text>` and events are joined
331
- * with a NUL-delimited separator (`\n\0\n`) so event boundaries cannot be forged
332
- * by text that itself contains newlines.
333
- *
334
- * The input is the RAW `data.events` stream — NOT the pre-filtered / truncated
335
- * set — so the hash is stable across `maxTotalChars` (and any other pre-filter)
336
- * config changes: changing config must NEVER change the hash (idempotency AC).
337
- * `inlineRefs` and ref metadata (title, startedAt/endedAt timestamps) are
338
- * deliberately EXCLUDED so clock/title churn (and an agent adding an inline
339
- * `akm remember` mid-session) does not change the hash.
340
- */
341
- function canonicalizeSessionContent(data) {
342
- return data.events.map((e) => `${e.role ?? "unknown"}\n${e.text}`).join("\n\0\n");
343
- }
344
- /**
345
- * sha256 (hex) of the normalized session content (#602). This is the byte-exact,
346
- * clock-independent skip authority that replaced the old `session_ended_at`
347
- * timestamp comparison. See {@link canonicalizeSessionContent} for exactly what
348
- * is (and is not) hashed.
237
+ * The skip authority for "already extracted": a hash of the raw event stream
238
+ * (`role\ntext` per event, NUL-separated so boundaries cannot be forged). It
239
+ * excludes titles, timestamps and inline refs, and ignores pre-filter config.
349
240
  */
350
241
  export function hashSessionContent(data) {
351
- return sha256Hex(canonicalizeSessionContent(data));
242
+ return contentHash(data.events.map((e) => `${e.role ?? "unknown"}\n${e.text}`).join("\n\0\n"));
352
243
  }
244
+ // ── Session triage (heuristic, zero LLM cost, default off) ───────────────────
245
+ const DEFAULT_TRIAGE_MIN_SCORE = 2;
246
+ const TRIAGE_MARKER_RE = /\b(error|failed|fix(?:ed)?|root cause|turns out|because|decided|instead|gotcha|workaround|regress(?:ed)?|broke|TIL)\b/i;
247
+ const TRIAGE_EDIT_COMMIT_RE = /\b(Edit|Write|MultiEdit|git commit|diff)\b/i;
353
248
  /**
354
- * Process one session through the full pipeline: read → pre-filter → LLM →
355
- * parse → createProposal-per-candidate. Returns the per-session result.
356
- *
357
- * On any non-fatal failure (LLM error, unparseable response, individual
358
- * proposal validation failure) the session result records a warning and
359
- * keeps going — one session's bad luck never aborts a multi-session run.
249
+ * Score a session for extraction worth: learning markers (capped 2), tool
250
+ * density, edits/commits, and the substantive assistant/tool share.
360
251
  */
252
+ function sessionTriagePasses(data, minScore) {
253
+ const events = data.events;
254
+ const count = (predicate) => events.filter(predicate).length;
255
+ const markers = Math.min(count((e) => TRIAGE_MARKER_RE.test(e.text)), 2);
256
+ const toolDensity = Math.min(count((e) => e.role === "tool") * 0.25, 1.5);
257
+ const editCommit = Math.min(count((e) => Boolean(e.filePath) || TRIAGE_EDIT_COMMIT_RE.test(e.text)) * 0.25, 1.5);
258
+ const substantive = count((e) => (e.role === "assistant" || e.role === "tool") && e.text.length >= 40);
259
+ const substantiveRatio = Math.min(events.length > 0 ? substantive / events.length : 0, 1);
260
+ return markers + toolDensity + editCommit + substantiveRatio >= minScore;
261
+ }
361
262
  /**
362
- * The zero-LLM pre-flight gates for one session: read, the #602 content-hash
363
- * already-extracted skip, the #595/#596 minContentChars floor, and the #626
364
- * heuristic triage gate. Returns a terminal skip result, or the read `data` +
365
- * pre-filtered events + content hash to carry into the extraction prompt.
366
- * Extracted verbatim from `processSession` — every skip shape/reason is
367
- * byte-identical.
263
+ * The zero-LLM gates for one session: read, the content-hash skip (only
264
+ * `--force` overrides it, even for `--session-id`), the raw-size floor and the
265
+ * triage score. Returns a skip, or what the prompt needs.
368
266
  */
369
- function runPreLlmSessionGates(args) {
370
- const { harness, sessionRef, prior, force, maxTotalChars, minContentChars, triage } = args;
267
+ function runPreLlmSessionGates(harness, sessionRef, prior, force, gates) {
268
+ const { sessionId } = sessionRef;
371
269
  let data;
372
270
  try {
373
271
  data = harness.readSession(sessionRef);
374
272
  }
375
273
  catch (err) {
376
274
  return {
377
- skip: {
378
- sessionId: sessionRef.sessionId,
379
- harness: harness.name,
380
- candidateCount: 0,
381
- proposalIds: [],
382
- preFilter: { inputCount: 0, outputCount: 0, truncatedCount: 0 },
275
+ skip: sessionOutcome(sessionId, harness.name, {
383
276
  warnings: [`readSession failed: ${err instanceof Error ? err.message : String(err)}`],
384
277
  skipped: true,
385
278
  skipReason: "read_failed",
386
- },
279
+ }),
387
280
  };
388
281
  }
389
- // #602 — content-hash skip. Computed on the RAW event stream immediately after
390
- // a successful read, BEFORE the pre-filter / minContentChars / triage gates, so
391
- // an unchanged session never reaches the LLM. Hash-based ⇒ clock-independent
392
- // (immune to the Jun 11-12 timestamp double-extract/over-throttle bug). The skip
393
- // applies UNIFORMLY — including explicit `--session-id` targeting (so a
394
- // session-end hook firing `extract --session-id <id>` is idempotent). ONLY
395
- // `--force` overrides it to re-extract a previously-extracted session.
396
- const contentHash = hashSessionContent(data);
397
- if (!force && shouldSkipAlreadyExtractedSession(prior, contentHash)) {
398
- return { skip: alreadyExtractedResult(harness.name, sessionRef.sessionId, prior, contentHash) };
399
- }
400
- // #840 — harvest-without-prompting hybrid: the LLM prompt is built only from
401
- // parent-origin events (folding stays as infrastructure for hashing above
402
- // and inline-ref harvesting on `data.inlineRefs`, both of which still see
403
- // the FULL folded stream). Subagent-origin events never reach
404
- // `preFilterSession`, so #839's `dedupeTaskNotifications` naturally becomes
405
- // a no-op on this path — a subagent's own event can no longer be in the
406
- // kept set for a notification to be deduped against, leaving the parent's
407
- // `<task-notification>` (the only surviving trace of that delegated work)
408
- // untouched. See docs/plans/subagent-extraction-design.md §6.
409
- const parentOriginData = {
410
- ...data,
411
- events: data.events.filter((e) => e.filePath === data.ref.filePath),
412
- };
413
- const filtered = preFilterSession(parentOriginData, {
414
- ...(typeof maxTotalChars === "number" ? { maxTotalChars } : {}),
415
- });
416
- // #595/#596 — minContentChars gate: skip the LLM call for sessions whose RAW
417
- // size is below threshold. Measured on the raw event text BEFORE the noise
418
- // pre-filter, NOT on post-filter output — the pre-filter strips boilerplate
419
- // so aggressively that even signal-bearing sessions can have tiny output
420
- // (#596: gating post-filter filtered out 100% of sessions). Note: the 0.8.x
421
- // fix gated on `filtered.stats.inputCount`, which is an EVENT count, not a
422
- // char count — this port measures actual raw chars so the threshold matches
423
- // the config key's documented unit.
424
- // #840 — deliberately measured on the FULL folded `data.events` (parent +
425
- // subagents), not the parent-origin view above: narrowing this to
426
- // parent-origin chars would newly skip delegation-heavy sessions with a
427
- // thin parent transcript before extraction runs at all, even though their
428
- // subagent work is still fully harvested via `data.inlineRefs` above. The
429
- // full-stream measurement is today's unchanged behavior, so the worst case
430
- // this preserves is an LLM call over a small parent-only prompt, not a
431
- // missed extraction.
432
- const rawContentChars = data.events.reduce((sum, event) => sum + event.text.length, 0);
433
- if (minContentChars > 0 && rawContentChars < minContentChars) {
282
+ const hash = hashSessionContent(data);
283
+ if (!force && shouldSkipAlreadyExtractedSession(prior, hash)) {
434
284
  return {
435
- skip: {
436
- sessionId: sessionRef.sessionId,
437
- harness: harness.name,
438
- candidateCount: 0,
439
- proposalIds: [],
440
- preFilter: {
441
- inputCount: filtered.stats.inputCount,
442
- outputCount: filtered.stats.outputCount,
443
- truncatedCount: filtered.stats.truncatedCount,
444
- },
445
- warnings: [],
285
+ skip: sessionOutcome(sessionId, harness.name, {
286
+ warnings: [`already extracted (content unchanged) at ${prior?.processed_at}; pass --force to re-process`],
446
287
  skipped: true,
447
- skipReason: "too_short",
448
- contentHash,
449
- },
288
+ skipReason: "already_extracted",
289
+ contentHash: hash,
290
+ }),
450
291
  };
451
292
  }
452
- // #626 — pre-LLM heuristic triage gate. Runs AFTER minContentChars + the
453
- // already-extracted skip check (both in the caller / above), BEFORE the
454
- // extraction prompt and the session-asset write. When the session scores below
455
- // the configured threshold we triage it out: no chat() call, no session asset,
456
- // no proposals. Pure-heuristic — zero added LLM cost. Default-off → skipped.
457
- if (triage.enabled) {
458
- const t = scoreSessionTriage(data, triage.minScore);
459
- if (!t.pass) {
460
- return {
461
- skip: {
462
- sessionId: sessionRef.sessionId,
463
- harness: harness.name,
464
- candidateCount: 0,
465
- proposalIds: [],
466
- preFilter: {
467
- inputCount: filtered.stats.inputCount,
468
- outputCount: filtered.stats.outputCount,
469
- truncatedCount: filtered.stats.truncatedCount,
470
- },
471
- warnings: [],
472
- skipped: true,
473
- skipReason: "triaged_out",
474
- contentHash,
475
- },
476
- };
477
- }
293
+ // The prompt sees only parent-origin events; subagent work still reaches
294
+ // the hash above and the inline-ref harvest.
295
+ const filtered = preFilterSession({ ...data, events: data.events.filter((e) => e.filePath === data.ref.filePath) }, typeof gates.maxTotalChars === "number" ? { maxTotalChars: gates.maxTotalChars } : {});
296
+ // Measured on the full raw stream: pre-filtered size says little about value.
297
+ const rawChars = data.events.reduce((sum, event) => sum + event.text.length, 0);
298
+ const skipReason = gates.minContentChars > 0 && rawChars < gates.minContentChars
299
+ ? "too_short"
300
+ : gates.triage.enabled && !sessionTriagePasses(data, gates.triage.minScore)
301
+ ? "triaged_out"
302
+ : undefined;
303
+ if (skipReason) {
304
+ return {
305
+ skip: sessionOutcome(sessionId, harness.name, {
306
+ preFilter: preFilterStats(filtered),
307
+ skipped: true,
308
+ skipReason,
309
+ contentHash: hash,
310
+ }),
311
+ };
478
312
  }
479
- return { data, filtered, contentHash };
480
- }
481
- function alreadyExtractedResult(harness, sessionId, prior, contentHash) {
482
- return {
483
- sessionId,
484
- harness,
485
- candidateCount: 0,
486
- proposalIds: [],
487
- preFilter: { inputCount: 0, outputCount: 0, truncatedCount: 0 },
488
- warnings: [`already extracted (content unchanged) at ${prior?.processed_at}; pass --force to re-process`],
489
- skipped: true,
490
- skipReason: "already_extracted",
491
- contentHash,
492
- };
313
+ return { data, filtered, contentHash: hash };
493
314
  }
494
315
  function lockedConcurrentResult(harness, summary) {
495
- return {
496
- sessionId: summary.sessionId,
497
- harness,
498
- candidateCount: 0,
499
- proposalIds: [],
500
- preFilter: { inputCount: 0, outputCount: 0, truncatedCount: 0 },
316
+ return sessionOutcome(summary.sessionId, harness, {
501
317
  warnings: ["concurrent extract holds this session's lock — skipped (handled by the other run)"],
502
318
  skipped: true,
503
319
  skipReason: "locked_concurrent",
504
- };
320
+ });
505
321
  }
322
+ /**
323
+ * Classify candidates read-only, up to `maxSessionsPerRun` model sessions
324
+ * (explicit `--session-id` and `--force` are uncapped); the rest are deferred.
325
+ */
506
326
  function planExtractSessions(args) {
507
- const { candidates, options, harness, seenMap, maxSessionsPerRun, trackingEnabled, dryRun } = args;
327
+ const { candidates, options, harness, maxSessionsPerRun, locking } = args;
328
+ const lockUnavailable = (summary) => locking &&
329
+ extractSessionLockIsUnavailable(harness.name, summary.sessionId, options.stateDbPath ?? getStateDbPath());
508
330
  const plans = [];
509
331
  let modelCount = 0;
510
332
  for (let index = 0; index < candidates.length; index++) {
511
- if (options.signal?.aborted)
512
- return { plans, deferredCandidates: candidates.slice(index) };
513
- if (!options.sessionId && !options.force && maxSessionsPerRun > 0 && modelCount >= maxSessionsPerRun) {
333
+ const capped = !options.sessionId && !options.force && maxSessionsPerRun > 0 && modelCount >= maxSessionsPerRun;
334
+ if (options.signal?.aborted || capped)
514
335
  return { plans, deferredCandidates: candidates.slice(index) };
515
- }
516
336
  const summary = candidates[index];
517
337
  if (!summary)
518
338
  continue;
519
- if (trackingEnabled && !dryRun && !options.stateDb) {
520
- if (extractSessionLockIsUnavailable(harness.name, summary.sessionId, options.stateDbPath ?? getStateDbPath())) {
521
- plans.push({ kind: "skip", summary, result: lockedConcurrentResult(harness.name, summary) });
522
- continue;
523
- }
339
+ if (lockUnavailable(summary)) {
340
+ plans.push({ kind: "skip", summary, result: lockedConcurrentResult(harness.name, summary) });
341
+ continue;
524
342
  }
525
- const gate = runPreLlmSessionGates({
526
- harness,
527
- sessionRef: summary,
528
- prior: seenMap.get(summary.sessionId),
529
- force: options.force === true,
530
- maxTotalChars: args.maxTotalChars,
531
- minContentChars: args.minContentChars,
532
- triage: args.triage,
533
- });
343
+ const prior = args.seenMap.get(summary.sessionId);
344
+ const gate = runPreLlmSessionGates(harness, summary, prior, options.force === true, args.gates);
534
345
  if ("skip" in gate) {
535
346
  plans.push({ kind: "skip", summary, result: gate.skip });
536
347
  continue;
537
348
  }
538
- // Reading and classifying a session can take long enough for a concurrent
539
- // session-end hook to claim its lock. Re-probe the fully classified model
540
- // plan before it consumes a cap slot or forces credential materialization.
541
- if (trackingEnabled &&
542
- !dryRun &&
543
- !options.stateDb &&
544
- extractSessionLockIsUnavailable(harness.name, summary.sessionId, options.stateDbPath ?? getStateDbPath())) {
349
+ // Classifying can take long enough for a session-end hook to claim the lock.
350
+ if (lockUnavailable(summary)) {
545
351
  plans.push({ kind: "skip", summary, result: lockedConcurrentResult(harness.name, summary) });
546
352
  continue;
547
353
  }
@@ -551,42 +357,37 @@ function planExtractSessions(args) {
551
357
  return { plans, deferredCandidates: [] };
552
358
  }
553
359
  const EXTRACT_LLM_UNAVAILABLE = Symbol("extract-llm-unavailable");
554
- async function runSessionExtractionLlmCall(args) {
555
- const { config, llmRunner, lease, chat, prompt, timeoutMs, signal, onNotices } = args;
360
+ /**
361
+ * One session's extraction call. A connection without structured output gets
362
+ * one corrective retry; configuration errors escape before any state is written.
363
+ */
364
+ async function extractFromSession(run, prompt) {
365
+ const { llmRunner } = run;
556
366
  try {
557
367
  const result = await runStructured({
558
368
  dispatch: async (feedback) => {
559
- const content = feedback ? `${prompt}\n\n## Corrective output instruction\n\n${feedback}` : prompt;
560
- const dispatched = await callStructured({
369
+ const outcome = await callStage({
561
370
  feature: "session_extraction",
562
- akmConfig: config,
563
371
  runner: llmRunner,
564
- lease,
565
- messages: [{ role: "user", content }],
372
+ prompt: feedback ? `${prompt}\n\n## Corrective output instruction\n\n${feedback}` : prompt,
373
+ gate: { config: run.config },
566
374
  request: {
567
- timeoutMs,
375
+ timeoutMs: run.timeoutMs,
568
376
  responseSchema: EXTRACT_JSON_SCHEMA,
569
- ...(signal ? { signal } : {}),
570
- ...(chat ? { chat } : {}),
377
+ ...(run.options.signal ? { signal: run.options.signal } : {}),
378
+ ...(run.options.chat ? { chat: run.options.chat } : {}),
571
379
  },
572
- onNotices,
573
- parse: (raw) => ({ kind: "response", raw: raw ?? "" }),
574
- onError: () => ({ kind: "unavailable" }),
575
- fallback: { kind: "unavailable" },
380
+ onNotices: run.notices.add,
576
381
  });
577
- if (dispatched.kind === "unavailable")
382
+ if (!outcome.ok)
578
383
  throw EXTRACT_LLM_UNAVAILABLE;
579
- return dispatched.raw;
384
+ return outcome.raw;
580
385
  },
581
386
  parse: (raw) => {
582
387
  const payload = parseExtractPayload(raw);
583
388
  return payload.parseFailure ? undefined : payload;
584
389
  },
585
390
  validate: (payload) => ({ ok: true, value: payload }),
586
- // One attempt when structured output is expected to work (not explicitly
587
- // disabled, and this connection hasn't already proven otherwise this
588
- // process — see `isJsonSchemaKnownUnsupported`); two when it's known
589
- // unsupported and extraction is relying on looser prompt-contract JSON.
590
391
  maxAttempts: llmRunner.connection.supportsJsonSchema !== false && !isJsonSchemaKnownUnsupported(llmRunner.connection)
591
392
  ? 1
592
393
  : 2,
@@ -594,13 +395,14 @@ async function runSessionExtractionLlmCall(args) {
594
395
  });
595
396
  if (result.ok)
596
397
  return { kind: "success", payload: result.value, attempts: result.attempts };
597
- const payload = parseExtractPayload(result.raw);
598
398
  return {
599
399
  kind: "malformed",
600
400
  raw: result.raw,
601
401
  attempts: result.attempts,
602
- failure: payload.parseFailure ??
603
- { code: "invalid_payload", message: result.errors.join("; ") },
402
+ failure: parseExtractPayload(result.raw).parseFailure ?? {
403
+ code: "invalid_payload",
404
+ message: result.errors.join("; "),
405
+ },
604
406
  };
605
407
  }
606
408
  catch (err) {
@@ -609,333 +411,148 @@ async function runSessionExtractionLlmCall(args) {
609
411
  throw err;
610
412
  }
611
413
  }
612
- function extractNoticeFields(getNotices) {
613
- const notices = getNotices();
614
- return notices.length > 0 ? { notices } : {};
615
- }
616
- function extractPreFilterStats(filtered) {
617
- return {
618
- inputCount: filtered.stats.inputCount,
619
- outputCount: filtered.stats.outputCount,
620
- truncatedCount: filtered.stats.truncatedCount,
621
- };
622
- }
623
- function malformedExtractionResult(args) {
624
- const { extraction, sessionRef, harness, preFilter, contentHash, notices } = args;
625
- const diagnostic = `malformed_model_output: ${extraction.failure.message}; attempts=${extraction.attempts}; responseLength=${extraction.raw.length}; responseSha256=${sha256Hex(extraction.raw)}`;
626
- warnVerbose(`[extract] malformed model output for session ${sessionRef.sessionId}: ${redactErrorBody(extraction.raw)}`);
627
- return {
628
- sessionId: sessionRef.sessionId,
629
- harness,
630
- candidateCount: 0,
631
- proposalIds: [],
632
- preFilter,
633
- warnings: [diagnostic],
634
- skipped: true,
635
- skipReason: "malformed_model_output",
636
- contentHash,
637
- ...notices,
638
- };
639
- }
640
- function unavailableExtractionResult(args) {
641
- return {
642
- sessionId: args.sessionRef.sessionId,
643
- harness: args.harness,
644
- candidateCount: 0,
645
- proposalIds: [],
646
- preFilter: args.preFilter,
647
- warnings: ["session_extraction feature returned empty (disabled / timeout / error)"],
648
- skipped: true,
649
- skipReason: "llm_unavailable",
650
- contentHash: args.contentHash,
651
- ...args.notices,
652
- };
653
- }
654
- // #561 — ADDITIVE session indexing. Generate + write the session asset
655
- // (`sessions/<harness>/<id>.md`). FAIL-OPEN: any failure only returns a
656
- // warning; it NEVER changes the proposal/skip outcome of extract. Returns the
657
- // frontmatter fields to merge into the per-session result for state-db
658
- // correlation. When disabled this makes NO LLM call and writes NOTHING.
659
- async function maybeWriteSessionAsset(runCtx, session) {
660
- const { stashDir, lease, sessionIndexing, dryRun } = runCtx;
661
- const { data } = session.gate;
662
- if (!sessionIndexing.enabled || dryRun)
414
+ /**
415
+ * Write the session's searchable asset (`sessions/<harness>/<id>.md`). Fails
416
+ * open: a failure is only a warning and never changes the extract outcome.
417
+ */
418
+ async function maybeWriteSessionAsset(run, data) {
419
+ const { sessionIndexing } = run;
420
+ if (!sessionIndexing.enabled || run.dryRun)
663
421
  return {};
664
422
  if (!sessionMeetsDurationGate(data, sessionIndexing.minDurationMinutes))
665
423
  return {};
666
424
  try {
667
- const result = await writeSessionAsset(data, stashDir, (summaryData) => sessionIndexing.generate(summaryData, lease));
668
- if (result.written) {
669
- // Write-path indexing (itself fail-open): standalone `akm extract`
670
- // (session-end hook) has no post-loop reindex to pick this file up.
671
- if (result.filePath)
672
- await indexWrittenAssets(stashDir, [result.filePath]);
673
- return {
674
- ...(result.ref ? { sessionAssetRef: result.ref } : {}),
675
- ...(result.logPath ? { sessionLogPath: result.logPath } : {}),
676
- };
677
- }
425
+ const result = await writeSessionAsset(data, run.stashDir, (summaryData) => sessionIndexing.generate(summaryData));
426
+ if (!result.written)
427
+ return {};
428
+ // A standalone extract has no post-loop reindex to pick the file up.
429
+ if (result.filePath)
430
+ await indexWrittenAssets(run.stashDir, [result.filePath]);
431
+ return {
432
+ ...(result.ref ? { sessionAssetRef: result.ref } : {}),
433
+ ...(result.logPath ? { sessionLogPath: result.logPath } : {}),
434
+ };
678
435
  }
679
436
  catch (err) {
680
437
  if (err instanceof ConfigError)
681
438
  throw err;
682
439
  return { warning: `session asset write failed: ${err instanceof Error ? err.message : String(err)}` };
683
440
  }
684
- return {};
685
441
  }
686
- async function processSession(runCtx, session) {
687
- const { harness, stashDir, config, llmRunner, lease, onNotices, getNotices, chat, ctx, eventsCtx, sourceRun, dryRun, timeoutMs, signal, standardsContext, } = runCtx;
688
- const { sessionRef, gate } = session;
689
- const warnings = [];
690
- const { data, filtered, contentHash } = gate;
691
- if (!lease)
692
- throw new TypeError("extract model work requires an operation dispatch lease");
693
- const prompt = buildExtractPrompt({
442
+ async function processSession(run, sessionRef, gate) {
443
+ const { harness, stashDir, sourceRun, dryRun, options } = run;
444
+ const { data, filtered, contentHash: hash } = gate;
445
+ const base = { preFilter: preFilterStats(filtered), contentHash: hash };
446
+ const extraction = await extractFromSession(run, buildExtractPrompt({
694
447
  data,
695
448
  events: filtered.events,
696
449
  inlineRefs: data.inlineRefs,
697
- ...(standardsContext.trim() ? { standardsContext } : {}),
698
- });
699
- const extraction = await runSessionExtractionLlmCall({
700
- config,
701
- llmRunner,
702
- lease,
703
- chat,
704
- prompt,
705
- timeoutMs,
706
- signal,
707
- onNotices,
708
- });
450
+ ...(run.standardsContext.trim() ? { standardsContext: run.standardsContext } : {}),
451
+ }));
709
452
  if (extraction.kind === "unavailable") {
710
- // The seam took the fallback path (disabled / timeout / error). Return skipped.
711
- return unavailableExtractionResult({
712
- sessionRef,
713
- harness: harness.name,
714
- preFilter: extractPreFilterStats(filtered),
715
- contentHash,
716
- notices: extractNoticeFields(getNotices),
453
+ return sessionOutcome(sessionRef.sessionId, harness.name, {
454
+ ...base,
455
+ warnings: ["session_extraction feature returned empty (disabled / timeout / error)"],
456
+ skipped: true,
457
+ skipReason: "llm_unavailable",
458
+ ...run.notices.fields(),
717
459
  });
718
460
  }
719
461
  if (extraction.kind === "malformed") {
720
- return malformedExtractionResult({
721
- extraction,
722
- sessionRef,
723
- harness: harness.name,
724
- preFilter: extractPreFilterStats(filtered),
725
- contentHash,
726
- notices: extractNoticeFields(getNotices),
462
+ warnVerbose(`[extract] malformed model output for session ${sessionRef.sessionId}: ${redactErrorBody(extraction.raw)}`);
463
+ return sessionOutcome(sessionRef.sessionId, harness.name, {
464
+ ...base,
465
+ warnings: [
466
+ `malformed_model_output: ${extraction.failure.message}; attempts=${extraction.attempts}; responseLength=${extraction.raw.length}; responseSha256=${contentHash(extraction.raw)}`,
467
+ ],
468
+ skipped: true,
469
+ skipReason: "malformed_model_output",
470
+ ...run.notices.fields(),
727
471
  });
728
472
  }
729
473
  const { payload } = extraction;
474
+ const warnings = [];
475
+ // Provenance xrefs are added only after the cited session asset exists.
476
+ const { warning, ...sessionAsset } = await maybeWriteSessionAsset(run, data);
477
+ if (warning)
478
+ warnings.push(warning);
730
479
  const proposalIds = [];
731
- // Provenance refs are added only after the cited session asset exists.
732
- const sessionAsset = await maybeWriteSessionAsset(runCtx, session);
733
- if (sessionAsset.warning)
734
- warnings.push(sessionAsset.warning);
735
- if (payload.candidates.length === 0) {
736
- appendEvent({
737
- eventType: "extract_invoked",
738
- ...(sessionAsset.sessionAssetRef ? { ref: sessionAsset.sessionAssetRef } : {}),
739
- metadata: {
740
- outcome: "no_candidates",
741
- sessionId: sessionRef.sessionId,
742
- harness: harness.name,
743
- sourceRun,
744
- rationale: payload.rationale_if_empty,
745
- repairAttempts: extraction.attempts - 1,
746
- preFilterInput: filtered.stats.inputCount,
747
- preFilterOutput: filtered.stats.outputCount,
748
- },
749
- }, eventsCtx);
750
- return {
751
- sessionId: sessionRef.sessionId,
752
- harness: harness.name,
753
- candidateCount: 0,
754
- proposalIds: [],
755
- ...(payload.rationale_if_empty ? { rationaleIfEmpty: payload.rationale_if_empty } : {}),
756
- preFilter: {
757
- inputCount: filtered.stats.inputCount,
758
- outputCount: filtered.stats.outputCount,
759
- truncatedCount: filtered.stats.truncatedCount,
760
- },
761
- warnings,
762
- contentHash,
763
- ...sessionAsset,
764
- ...extractNoticeFields(getNotices),
765
- };
766
- }
767
- // §23.6 fingerprint model-id term: the profile resolved for this session's
768
- // LLM call (best-effort — an unconfigured profile leaves the term empty).
769
- const extractModelId = llmRunner.connection.model;
770
- for (const candidate of payload.candidates) {
771
- const built = buildCandidateProposal(candidate, data.ref, sessionAsset.sessionAssetRef);
772
- if (dryRun) {
773
- proposalIds.push(`dry-run:${built.ref}`);
774
- continue;
775
- }
776
- try {
777
- const { ref, content, description } = built;
778
- const result = emitProposal({ stashDir, proposalsCtx: ctx }, {
779
- ref,
780
- source: "extract",
781
- sourceRun,
782
- // §23.6 fingerprint model-id term (WI-6.4). The LLM already ran for
783
- // this session, so the profile is resolvable; guard anyway.
784
- ...(extractModelId ? { modelId: extractModelId } : {}),
785
- payload: {
786
- content,
787
- frontmatter: {
788
- description,
789
- ...(candidate.when_to_use ? { when_to_use: candidate.when_to_use } : {}),
790
- confidence: candidate.confidence,
791
- ...(sessionAsset.sessionAssetRef ? { xrefs: [sessionAsset.sessionAssetRef] } : {}),
792
- evidence: candidate.evidence,
480
+ if (payload.candidates.length > 0) {
481
+ // A candidate the ledger holds a live window for (pending, or recently rejected) is not queued again.
482
+ const ledger = loadLedgerSnapshot({ proposalsCtx: options.ctx, eventsCtx: options.eventsCtx, ...(dryRun ? { readOnly: true } : {}) }, stashDir, ["extract"]);
483
+ const nowIso = new Date().toISOString();
484
+ for (const candidate of payload.candidates) {
485
+ const built = buildCandidateProposal(candidate, data.ref, sessionAsset.sessionAssetRef);
486
+ const ledgerRow = ledger.get(ledgerKey("extract", built.ref));
487
+ if (isLedgerBlocked(ledgerRow, nowIso)) {
488
+ warnings.push(`candidate ${candidate.type}:${candidate.name} skipped: ${ledgerRow?.outcome} until ${ledgerRow?.nextEligibleAt}`);
489
+ continue;
490
+ }
491
+ if (dryRun) {
492
+ proposalIds.push(`dry-run:${built.ref}`);
493
+ continue;
494
+ }
495
+ try {
496
+ const proposal = mintProposal(stashDir, options.ctx, {
497
+ ref: built.ref,
498
+ source: "extract",
499
+ sourceRun,
500
+ attemptedRefs: [built.ref],
501
+ payload: {
502
+ content: built.content,
503
+ frontmatter: {
504
+ description: built.description,
505
+ ...(candidate.when_to_use ? { when_to_use: candidate.when_to_use } : {}),
506
+ confidence: candidate.confidence,
507
+ ...(sessionAsset.sessionAssetRef ? { xrefs: [sessionAsset.sessionAssetRef] } : {}),
508
+ evidence: candidate.evidence,
509
+ },
793
510
  },
794
- },
795
- });
796
- if (isProposalSkipped(result)) {
797
- warnings.push(`candidate ${candidate.type}:${candidate.name} skipped: ${result.reason}: ${result.message}`);
511
+ });
512
+ proposalIds.push(proposal.id);
798
513
  }
799
- else {
800
- proposalIds.push(result.id);
514
+ catch (err) {
515
+ warnings.push(`candidate ${candidate.type}:${candidate.name} failed: ${err instanceof Error ? err.message : String(err)}`);
801
516
  }
802
517
  }
803
- catch (err) {
804
- warnings.push(`candidate ${candidate.type}:${candidate.name} failed: ${err instanceof Error ? err.message : String(err)}`);
805
- }
806
518
  }
519
+ const empty = payload.candidates.length === 0;
807
520
  appendEvent({
808
521
  eventType: "extract_invoked",
809
522
  ...(sessionAsset.sessionAssetRef ? { ref: sessionAsset.sessionAssetRef } : {}),
810
523
  metadata: {
811
- outcome: "candidates_queued",
524
+ outcome: empty ? "no_candidates" : "candidates_queued",
812
525
  sessionId: sessionRef.sessionId,
813
526
  harness: harness.name,
814
527
  sourceRun,
815
- candidateCount: payload.candidates.length,
816
- proposalCount: proposalIds.length,
528
+ ...(empty
529
+ ? { rationale: payload.rationale_if_empty }
530
+ : { candidateCount: payload.candidates.length, proposalCount: proposalIds.length }),
817
531
  preFilterInput: filtered.stats.inputCount,
818
532
  preFilterOutput: filtered.stats.outputCount,
819
533
  repairAttempts: extraction.attempts - 1,
820
534
  },
821
- }, eventsCtx);
822
- return {
823
- sessionId: sessionRef.sessionId,
824
- harness: harness.name,
535
+ }, options.eventsCtx);
536
+ return sessionOutcome(sessionRef.sessionId, harness.name, {
537
+ ...base,
825
538
  candidateCount: payload.candidates.length,
826
539
  proposalIds,
827
- preFilter: {
828
- inputCount: filtered.stats.inputCount,
829
- outputCount: filtered.stats.outputCount,
830
- truncatedCount: filtered.stats.truncatedCount,
831
- },
540
+ ...(empty && payload.rationale_if_empty ? { rationaleIfEmpty: payload.rationale_if_empty } : {}),
832
541
  warnings,
833
- contentHash,
834
542
  ...sessionAsset,
835
- ...extractNoticeFields(getNotices),
836
- };
837
- }
838
- function recordExtractSessionOutcome(args) {
839
- const { stateDb, trackingEnabled, dryRun, harness, summary, result, sourceRun } = args;
840
- if (!trackingEnabled ||
841
- !stateDb ||
842
- dryRun ||
843
- result.skipReason === "already_extracted" ||
844
- result.skipReason === "locked_concurrent")
845
- return;
846
- try {
847
- const outcome = result.skipped
848
- ? result.skipReason === "read_failed" ||
849
- result.skipReason === "exception" ||
850
- result.skipReason === "malformed_model_output"
851
- ? "failed"
852
- : "skipped"
853
- : result.candidateCount === 0
854
- ? "no_candidates"
855
- : "candidates_queued";
856
- upsertExtractedSession(stateDb, {
857
- harness,
858
- sessionId: summary.sessionId,
859
- processedAt: new Date().toISOString(),
860
- sessionEndedAt: summary.endedAt ?? null,
861
- outcome,
862
- candidateCount: result.candidateCount,
863
- proposalCount: result.proposalIds.length,
864
- rationale: result.rationaleIfEmpty ?? null,
865
- sourceRun,
866
- contentHash: result.skipReason === "llm_unavailable" ||
867
- result.skipReason === "triaged_out" ||
868
- result.skipReason === "malformed_model_output"
869
- ? null
870
- : (result.contentHash ?? null),
871
- metadata: {
872
- preFilterInputCount: result.preFilter.inputCount,
873
- preFilterOutputCount: result.preFilter.outputCount,
874
- preFilterTruncatedCount: result.preFilter.truncatedCount,
875
- engine: result.engine,
876
- ...(result.skipReason ? { skipReason: result.skipReason } : {}),
877
- ...(result.sessionLogPath ? { logPath: result.sessionLogPath } : {}),
878
- ...(result.sessionAssetRef ? { sessionAssetRef: result.sessionAssetRef } : {}),
879
- },
880
- });
881
- }
882
- catch (err) {
883
- warn(`[extract] failed to record session ${summary.sessionId} in state.db: ${err instanceof Error ? err.message : String(err)}`);
884
- }
885
- }
886
- function accountExtractSessionResult(result, triageEnabled, output) {
887
- const stamped = { ...result, engine: output.engine };
888
- output.sessions.push(stamped);
889
- if (triageEnabled) {
890
- const preempted = result.skipReason === "read_failed" ||
891
- result.skipReason === "too_short" ||
892
- result.skipReason === "already_extracted" ||
893
- result.skipReason === "locked_concurrent";
894
- if (!preempted) {
895
- output.triageEvaluated += 1;
896
- if (result.skipReason === "triaged_out")
897
- output.triagedOut += 1;
898
- else
899
- output.triagePassed += 1;
900
- }
901
- }
902
- if (result.skipped)
903
- output.skippedCount += 1;
904
- else
905
- output.processedCount += 1;
906
- output.allProposalIds.push(...result.proposalIds);
907
- return stamped;
543
+ ...run.notices.fields(),
544
+ });
908
545
  }
909
546
  /**
910
- * Iterate the discovered candidate sessions: enforce the per-run cap, take the
911
- * per-session cross-process lock, dispatch to {@link processSession}, aggregate
912
- * the #626 triage counters, and persist each seen-row outcome. Extracted verbatim
913
- * from `akmExtract` — the maxSessionsPerRun break, lock/skip accounting, triage
914
- * aggregation, and seen-row upsert are byte-identical.
547
+ * Work the plans: claim each model session's lock, re-read and re-gate it
548
+ * under the lock (the log may have changed since planning), extract, and
549
+ * record the outcome in the seen-session table. A skipped or locked session
550
+ * refills its model slot from the deferred candidates.
915
551
  */
916
- async function runExtractSessionLoop(args) {
917
- const { plans, deferredCandidates, seenMap, options, harness, stateDb, trackingEnabled, dryRun, stashDir, config, llmRunner, lease, onNotices, getNotices, chat, sourceRun, timeoutMs, triage, sessionIndexing, extractStandardsContext, topLevelWarnings, } = args;
918
- // WI-7.7 §2: run-scoped processSession inputs, resolved once per run.
919
- const sessionRunCtx = {
920
- harness,
921
- stashDir,
922
- config,
923
- llmRunner,
924
- lease,
925
- onNotices,
926
- getNotices,
927
- chat,
928
- ctx: options.ctx,
929
- eventsCtx: options.eventsCtx,
930
- sourceRun,
931
- dryRun,
932
- timeoutMs,
933
- sessionIndexing,
934
- signal: options.signal,
935
- standardsContext: extractStandardsContext,
936
- };
937
- const output = {
938
- engine: llmRunner.engine,
552
+ async function runExtractSessionLoop(run, planned, seenMap, stateDb, tracking, topLevelWarnings) {
553
+ const { options, harness, gates, dryRun } = run;
554
+ const locking = tracking && !dryRun && !options.stateDb;
555
+ const tally = {
939
556
  sessions: [],
940
557
  processedCount: 0,
941
558
  skippedCount: 0,
@@ -945,103 +562,78 @@ async function runExtractSessionLoop(args) {
945
562
  allProposalIds: [],
946
563
  deferred: 0,
947
564
  };
948
- const workPlans = [...plans];
949
- let remainingCandidates = deferredCandidates;
565
+ const account = (result) => {
566
+ const stamped = { ...result, engine: run.llmRunner.engine };
567
+ tally.sessions.push(stamped);
568
+ const preempted = ["read_failed", "too_short", "already_extracted", "locked_concurrent"].includes(result.skipReason ?? "");
569
+ if (gates.triage.enabled && !preempted) {
570
+ tally.triageEvaluated += 1;
571
+ if (result.skipReason === "triaged_out")
572
+ tally.triagedOut += 1;
573
+ else
574
+ tally.triagePassed += 1;
575
+ }
576
+ if (result.skipped)
577
+ tally.skippedCount += 1;
578
+ else
579
+ tally.processedCount += 1;
580
+ tally.allProposalIds.push(...result.proposalIds);
581
+ return stamped;
582
+ };
583
+ const accountAndRecord = (summary, result) => {
584
+ recordSessionOutcome(stateDb, tracking && !dryRun, harness.name, summary, account(result), run.sourceRun);
585
+ };
586
+ const workPlans = [...planned.plans];
587
+ let remaining = planned.deferredCandidates;
950
588
  const refillModelSlot = () => {
951
- if (remainingCandidates.length === 0 || options.signal?.aborted)
589
+ if (remaining.length === 0 || options.signal?.aborted)
952
590
  return;
953
591
  const refill = planExtractSessions({
954
- candidates: remainingCandidates,
592
+ candidates: remaining,
955
593
  options,
956
594
  harness,
957
595
  seenMap,
958
- maxTotalChars: args.maxTotalChars,
959
- minContentChars: args.minContentChars,
596
+ gates,
960
597
  maxSessionsPerRun: 1,
961
- triage,
962
- trackingEnabled,
963
- dryRun,
598
+ locking,
964
599
  });
965
600
  workPlans.push(...refill.plans);
966
- remainingCandidates = refill.deferredCandidates;
601
+ remaining = refill.deferredCandidates;
967
602
  };
968
603
  for (const plan of workPlans) {
969
604
  if (options.signal?.aborted)
970
605
  break;
971
606
  const { summary } = plan;
972
607
  if (plan.kind === "skip") {
973
- const accounted = accountExtractSessionResult(plan.result, triage.enabled, output);
974
- recordExtractSessionOutcome({
975
- stateDb,
976
- trackingEnabled,
977
- dryRun,
978
- harness: harness.name,
979
- summary,
980
- result: accounted,
981
- sourceRun,
982
- });
608
+ accountAndRecord(summary, plan.result);
983
609
  continue;
984
610
  }
985
- let sessionLockOwnership;
986
- if (trackingEnabled && !dryRun && !options.stateDb) {
987
- const sessionLockPath = getExtractSessionLockPath(harness.name, summary.sessionId, options.stateDbPath ?? getStateDbPath());
988
- const sessionLock = acquireExtractSessionLock(sessionLockPath);
989
- if (!sessionLock.proceed) {
990
- accountExtractSessionResult(lockedConcurrentResult(harness.name, summary), triage.enabled, output);
611
+ let lockOwnership;
612
+ if (locking) {
613
+ const lock = acquireExtractSessionLock(extractSessionLockPath(harness.name, summary.sessionId, options.stateDbPath ?? getStateDbPath()));
614
+ if (!lock.proceed) {
615
+ account(lockedConcurrentResult(harness.name, summary));
991
616
  refillModelSlot();
992
617
  continue;
993
618
  }
994
- sessionLockOwnership = sessionLock.ownership;
619
+ lockOwnership = lock.ownership;
995
620
  }
996
621
  try {
997
- // Planning stays read-only so a credential failure creates no state. Once
998
- // this run owns the session lock, read and gate the session again: the log
999
- // may have grown, become too short after replacement, or been completed by
1000
- // another extractor between the planning snapshot and acquisition.
1001
- const currentPrior = stateDb
622
+ const prior = stateDb
1002
623
  ? getExtractedSessionsMap(stateDb, harness.name, [summary.sessionId]).get(summary.sessionId)
1003
624
  : seenMap.get(summary.sessionId);
1004
- const executionGate = runPreLlmSessionGates({
1005
- harness,
1006
- sessionRef: summary,
1007
- prior: currentPrior,
1008
- force: options.force === true,
1009
- maxTotalChars: args.maxTotalChars,
1010
- minContentChars: args.minContentChars,
1011
- triage,
1012
- });
1013
- if ("skip" in executionGate) {
1014
- const accounted = accountExtractSessionResult(executionGate.skip, triage.enabled, output);
1015
- recordExtractSessionOutcome({
1016
- stateDb,
1017
- trackingEnabled,
1018
- dryRun,
1019
- harness: harness.name,
1020
- summary,
1021
- result: accounted,
1022
- sourceRun,
1023
- });
625
+ const gate = runPreLlmSessionGates(harness, summary, prior, options.force === true, gates);
626
+ if ("skip" in gate) {
627
+ accountAndRecord(summary, gate.skip);
1024
628
  refillModelSlot();
1025
629
  continue;
1026
630
  }
1027
- const result = await processSession(sessionRunCtx, {
1028
- sessionRef: summary,
1029
- gate: executionGate,
1030
- });
631
+ const result = await processSession(run, summary, gate);
1031
632
  if (result.skipReason === "malformed_model_output") {
1032
633
  for (const warning of result.warnings)
1033
634
  topLevelWarnings.push(`session ${summary.sessionId}: ${warning}`);
1034
635
  }
1035
- const accounted = accountExtractSessionResult(result, triage.enabled, output);
1036
- recordExtractSessionOutcome({
1037
- stateDb,
1038
- trackingEnabled,
1039
- dryRun,
1040
- harness: harness.name,
1041
- summary,
1042
- result: accounted,
1043
- sourceRun,
1044
- });
636
+ accountAndRecord(summary, result);
1045
637
  }
1046
638
  catch (err) {
1047
639
  if (err instanceof ConfigError)
@@ -1049,171 +641,158 @@ async function runExtractSessionLoop(args) {
1049
641
  const msg = err instanceof Error ? err.message : String(err);
1050
642
  warn(`[extract] session ${summary.sessionId} threw: ${msg}`);
1051
643
  topLevelWarnings.push(`session ${summary.sessionId} threw: ${msg}`);
1052
- accountExtractSessionResult({
1053
- sessionId: summary.sessionId,
1054
- harness: harness.name,
1055
- candidateCount: 0,
1056
- proposalIds: [],
1057
- preFilter: { inputCount: 0, outputCount: 0, truncatedCount: 0 },
644
+ account(sessionOutcome(summary.sessionId, harness.name, {
1058
645
  warnings: [msg],
1059
646
  skipped: true,
1060
647
  skipReason: "exception",
1061
- ...extractNoticeFields(getNotices),
1062
- }, triage.enabled, output);
648
+ ...run.notices.fields(),
649
+ }));
1063
650
  }
1064
651
  finally {
1065
- if (sessionLockOwnership)
1066
- releaseLock(sessionLockOwnership);
652
+ if (lockOwnership)
653
+ releaseLock(lockOwnership);
1067
654
  }
1068
655
  }
1069
- output.deferred = remainingCandidates.length;
1070
- return output;
656
+ tally.deferred = remaining.length;
657
+ return tally;
1071
658
  }
1072
659
  /**
1073
- * Resolve the run-scoped LLM/engine, budget, triage, and session-indexing
1074
- * settings for one extract invocation (throwing when no engine is configured).
1075
- * Extracted verbatim from `akmExtract` — the timeout precedence chain, the
1076
- * session-summary generator seam, and the default resolutions are byte-identical.
660
+ * Persist a session's outcome in the seen-session table. A session skipped as
661
+ * already extracted or locked is not rewritten; one that failed for a
662
+ * transient reason keeps a null hash so it is retried.
1077
663
  */
1078
- function resolveExtractRunConfig(options, config, extractProcess, activeProfile) {
1079
- const executionNotices = new Map();
1080
- const onNotices = (notices) => {
1081
- for (const notice of notices)
1082
- executionNotices.set(JSON.stringify(notice), notice);
1083
- };
1084
- const getNotices = () => Object.freeze([...executionNotices.values()]);
1085
- // Improve supplies its invocation-owned symbolic runner. Standalone extract
1086
- // resolves the selected process engine through the shared execution planner.
664
+ function recordSessionOutcome(stateDb, enabled, harness, summary, result, sourceRun) {
665
+ if (!enabled || !stateDb)
666
+ return;
667
+ if (result.skipReason === "already_extracted" || result.skipReason === "locked_concurrent")
668
+ return;
669
+ const reason = result.skipReason ?? "";
670
+ try {
671
+ upsertExtractedSession(stateDb, {
672
+ harness,
673
+ sessionId: summary.sessionId,
674
+ processedAt: new Date().toISOString(),
675
+ sessionEndedAt: summary.endedAt ?? null,
676
+ outcome: result.skipped
677
+ ? ["read_failed", "exception", "malformed_model_output"].includes(reason)
678
+ ? "failed"
679
+ : "skipped"
680
+ : result.candidateCount === 0
681
+ ? "no_candidates"
682
+ : "candidates_queued",
683
+ candidateCount: result.candidateCount,
684
+ proposalCount: result.proposalIds.length,
685
+ rationale: result.rationaleIfEmpty ?? null,
686
+ sourceRun,
687
+ contentHash: ["llm_unavailable", "triaged_out", "malformed_model_output"].includes(reason)
688
+ ? null
689
+ : (result.contentHash ?? null),
690
+ metadata: {
691
+ preFilterInputCount: result.preFilter.inputCount,
692
+ preFilterOutputCount: result.preFilter.outputCount,
693
+ preFilterTruncatedCount: result.preFilter.truncatedCount,
694
+ engine: result.engine,
695
+ ...(result.skipReason ? { skipReason: result.skipReason } : {}),
696
+ ...(result.sessionLogPath ? { logPath: result.sessionLogPath } : {}),
697
+ ...(result.sessionAssetRef ? { sessionAssetRef: result.sessionAssetRef } : {}),
698
+ },
699
+ });
700
+ }
701
+ catch (err) {
702
+ warn(`[extract] failed to record session ${summary.sessionId} in state.db: ${err instanceof Error ? err.message : String(err)}`);
703
+ }
704
+ }
705
+ /** The run's runner, budgets, gates and session-indexing settings (throws without an engine). */
706
+ function resolveExtractRun(options, config, process, activeProfile) {
707
+ const notices = noticeSet();
1087
708
  let llmRunner;
1088
709
  if (options.resolvedPlan) {
1089
710
  llmRunner = options.resolvedPlan.runner;
1090
- onNotices(options.resolvedPlan.notices ?? []);
711
+ notices.add(options.resolvedPlan.notices ?? []);
1091
712
  }
1092
713
  else if (options.llmRunner) {
1093
714
  llmRunner = options.llmRunner;
1094
715
  }
1095
716
  else {
1096
- const resolved = resolveImproveLlmExecution({
1097
- config,
1098
- profile: activeProfile,
1099
- process: extractProcess,
1100
- processName: "extract",
1101
- });
717
+ const resolved = resolveImproveLlmExecution({ config, profile: activeProfile, process, processName: "extract" });
1102
718
  llmRunner = resolved?.runner;
1103
719
  if (resolved)
1104
- onNotices(resolved.notices);
720
+ notices.add(resolved.notices);
1105
721
  }
1106
722
  if (!llmRunner) {
1107
723
  throw new ConfigError("No LLM engine configured for extract. Set defaults.llmEngine or improve.strategies.<name>.processes.extract.engine.", "LLM_NOT_CONFIGURED");
1108
724
  }
725
+ const runner = llmRunner;
1109
726
  const timeoutMs = options.resolvedPlan
1110
727
  ? options.resolvedPlan.timeoutMs
1111
728
  : Object.hasOwn(options, "timeoutMs")
1112
729
  ? (options.timeoutMs ?? null)
1113
- : Object.hasOwn(llmRunner, "timeoutMs")
1114
- ? (llmRunner.timeoutMs ?? null)
730
+ : Object.hasOwn(runner, "timeoutMs")
731
+ ? (runner.timeoutMs ?? null)
1115
732
  : 600_000;
1116
- // Pre-filter budget — process config can raise it for large-context models.
1117
- const maxTotalChars = typeof extractProcess?.maxTotalChars === "number" ? extractProcess.maxTotalChars : undefined;
1118
- // #595/#596 — minimum raw session size; sessions below it skip the LLM call
1119
- // entirely. Set `processes.extract.minContentChars: 0` to disable the gate.
1120
- const minContentChars = typeof extractProcess?.minContentChars === "number" ? extractProcess.minContentChars : DEFAULT_MIN_CONTENT_CHARS;
1121
- // Cap on NEW sessions LLM-processed per run; 0 disables. Absent = default.
1122
- // Bounds per-run wall time / LLM cost so a backlog can't push a run past its
1123
- // task timeout — the overflow stays unseen and is picked up by later runs.
1124
- const maxSessionsPerRun = options.since
1125
- ? 0
1126
- : typeof extractProcess?.maxSessionsPerRun === "number"
1127
- ? extractProcess.maxSessionsPerRun
1128
- : DEFAULT_MAX_SESSIONS_PER_RUN;
1129
- // Default discovery window — process config can override the built-in 24h.
1130
- const effectiveSince = options.since ?? extractProcess?.defaultSince;
1131
- // #626 — resolve the triage gate config once per run. Default-off → the
1132
- // per-session path never calls the scorer and emits no telemetry.
1133
- const triage = resolveTriageConfig(extractProcess);
1134
- // #561 — resolve session-indexing config. Default ON: we only reach this code
1135
- // when `session_extraction` is enabled AND an LLM is configured (both checked
1136
- // above), so defaulting on costs nothing offline (the summary call fails open)
1137
- // while making sessions searchable in the common LLM-configured case. Set
1138
- // `processes.extract.indexSessions: false` for byte-identical legacy behaviour.
1139
- const sessionIndexingEnabled = extractProcess?.indexSessions ?? true;
1140
- const minSessionDuration = typeof extractProcess?.minSessionDuration === "number"
1141
- ? extractProcess.minSessionDuration
1142
- : DEFAULT_MIN_SESSION_DURATION_MINUTES;
1143
- // Production summary generator: a bounded in-tree LLM call wrapped in the
1144
- // same fail-open `callStructured` seam as the rest of extract. Returns
1145
- // `undefined` on disablement / timeout / error so no asset is written.
1146
- // Tests inject a fake.
1147
- const defaultSessionSummaryGenerator = async (data, lease) => {
1148
- let raw = "";
1149
- await callStructured({
733
+ const triage = process?.triage;
734
+ // The default summary generator fails open: no summary, no session asset.
735
+ const generate = async (data) => {
736
+ const outcome = await callStage({
1150
737
  feature: "session_extraction",
1151
- akmConfig: config,
1152
- runner: llmRunner,
1153
- ...(lease ? { lease } : {}),
1154
- messages: [{ role: "user", content: buildSessionSummaryPrompt(data) }],
738
+ runner,
739
+ prompt: buildSessionSummaryPrompt(data),
740
+ gate: { config },
1155
741
  request: {
1156
742
  timeoutMs,
1157
743
  responseSchema: SESSION_SUMMARY_JSON_SCHEMA,
1158
744
  ...(options.signal ? { signal: options.signal } : {}),
1159
745
  ...(options.chat ? { chat: options.chat } : {}),
1160
746
  },
1161
- onNotices,
1162
- parse: (r) => {
1163
- raw = r ?? "";
1164
- return raw;
1165
- },
1166
- onError: () => "",
1167
- fallback: "",
747
+ onNotices: notices.add,
1168
748
  });
1169
- return parseSessionSummary(raw);
1170
- };
1171
- const sessionIndexing = {
1172
- enabled: sessionIndexingEnabled,
1173
- minDurationMinutes: minSessionDuration,
1174
- generate: options.generateSessionSummary ?? defaultSessionSummaryGenerator,
749
+ return parseSessionSummary(outcome.ok ? outcome.raw : "");
1175
750
  };
1176
751
  return {
752
+ llmRunner: runner,
753
+ notices,
1177
754
  timeoutMs,
1178
- llmRunner,
1179
- onNotices,
1180
- getNotices,
1181
- maxTotalChars,
1182
- minContentChars,
1183
- maxSessionsPerRun,
1184
- effectiveSince,
1185
- triage,
1186
- sessionIndexing,
755
+ gates: {
756
+ maxTotalChars: typeof process?.maxTotalChars === "number" ? process.maxTotalChars : undefined,
757
+ minContentChars: typeof process?.minContentChars === "number" ? process.minContentChars : DEFAULT_MIN_CONTENT_CHARS,
758
+ triage: {
759
+ enabled: triage?.enabled === true,
760
+ minScore: typeof triage?.minScore === "number" ? triage.minScore : DEFAULT_TRIAGE_MIN_SCORE,
761
+ },
762
+ },
763
+ maxSessionsPerRun: options.since
764
+ ? 0
765
+ : typeof process?.maxSessionsPerRun === "number"
766
+ ? process.maxSessionsPerRun
767
+ : DEFAULT_MAX_SESSIONS_PER_RUN,
768
+ effectiveSince: options.since ?? process?.defaultSince,
769
+ sessionIndexing: {
770
+ enabled: process?.indexSessions ?? true,
771
+ minDurationMinutes: typeof process?.minSessionDuration === "number"
772
+ ? process.minSessionDuration
773
+ : DEFAULT_MIN_SESSION_DURATION_MINUTES,
774
+ generate: options.generateSessionSummary ?? generate,
775
+ },
1187
776
  };
1188
777
  }
1189
- /**
1190
- * Resolve the session set to process: the single `--session-id` target (or a
1191
- * not-found envelope) or the discovery-window listing. Extracted verbatim from
1192
- * `akmExtract`; the 48h default-since floor and location filter are unchanged.
1193
- */
778
+ /** The sessions to process: the `--session-id` target (or a not-found envelope) or the discovery window. */
1194
779
  function discoverExtractCandidates(options, harness, effectiveSince, startMs, dryRun, llmRunner) {
780
+ const location = options.location ? { location: options.location } : {};
1195
781
  if (options.sessionId) {
1196
- const all = harness.listSessions({
1197
- ...(options.location ? { location: options.location } : {}),
1198
- });
1199
- const target = all.find((s) => s.sessionId === options.sessionId);
1200
- if (!target) {
1201
- return {
1202
- notFound: emptyExtractResult({
1203
- ok: false,
1204
- dryRun,
1205
- type: options.type,
1206
- warning: `session ${options.sessionId} not found for harness ${options.type}`,
1207
- startMs,
1208
- llmRunner,
1209
- }),
1210
- };
1211
- }
1212
- return { candidates: [target] };
782
+ const target = harness.listSessions(location).find((s) => s.sessionId === options.sessionId);
783
+ if (target)
784
+ return { candidates: [target] };
785
+ return {
786
+ notFound: emptyExtractResult({
787
+ ok: false,
788
+ dryRun,
789
+ type: options.type,
790
+ warning: `session ${options.sessionId} not found for harness ${options.type}`,
791
+ startMs,
792
+ llmRunner,
793
+ }),
794
+ };
1213
795
  }
1214
- // No explicit `--since`/`defaultSince` → default to "since the last run"
1215
- // (floored at 48h) so an intermittently-online host doesn't lose sessions
1216
- // that ended while it was off. See {@link resolveDefaultSinceMs}.
1217
796
  const sinceMs = effectiveSince
1218
797
  ? parseSinceArg(effectiveSince)
1219
798
  : resolveDefaultSinceMs(harness.name, startMs, {
@@ -1221,38 +800,10 @@ function discoverExtractCandidates(options, harness, effectiveSince, startMs, dr
1221
800
  ...(options.stateDbPath ? { stateDbPath: options.stateDbPath } : {}),
1222
801
  ...(options.skipTracking ? { skipTracking: options.skipTracking } : {}),
1223
802
  });
1224
- return {
1225
- candidates: harness.listSessions({
1226
- sinceMs,
1227
- ...(options.location ? { location: options.location } : {}),
1228
- }),
1229
- };
803
+ return { candidates: harness.listSessions({ sinceMs, ...location }) };
1230
804
  }
1231
- // ── Public entrypoint ────────────────────────────────────────────────────────
1232
- /**
1233
- * WI-9.10: build one `akm extract` run's {@link RunContext} from values
1234
- * `akmExtract` has already resolved by the time it calls this (config,
1235
- * stashDir, dryRun, sourceRun, and `resolveExtractRunConfig`'s symbolic runner)
1236
- * — no second config load, credential materialization, or new db handle.
1237
- */
1238
- function buildExtractRunContext(args) {
1239
- const { options, config, stashDir, dryRun, sourceRun, llmRunner } = args;
1240
- return createRunContext({
1241
- stashDir,
1242
- config,
1243
- eventsCtx: options.eventsCtx ?? {},
1244
- // Not yet wired into any proposal call site this stage (mirrors
1245
- // buildImproveRunContext's proposalsCtx comment in improve.ts).
1246
- proposalsCtx: options.ctx ?? {},
1247
- getLlmRunner: () => llmRunner,
1248
- sourceRun,
1249
- dryRun,
1250
- signal: options.signal,
1251
- });
1252
- }
1253
- function loadExtractSeenMapReadOnly(args) {
1254
- const { options, harness, candidates, trackingEnabled, warnings } = args;
1255
- if (!trackingEnabled || candidates.length === 0)
805
+ function loadSeenMapReadOnly(options, harness, candidates, warnings) {
806
+ if (options.skipTracking === true || candidates.length === 0)
1256
807
  return new Map();
1257
808
  let snapshot;
1258
809
  try {
@@ -1273,84 +824,22 @@ function loadExtractSeenMapReadOnly(args) {
1273
824
  snapshot?.close();
1274
825
  }
1275
826
  }
1276
- function openExtractLiveStateDb(args) {
1277
- const { options, trackingEnabled, hasModelWork, dryRun, warnings } = args;
1278
- if (!trackingEnabled)
1279
- return undefined;
1280
- if (options.stateDb)
1281
- return options.stateDb;
1282
- if (!hasModelWork || dryRun)
1283
- return undefined;
1284
- try {
1285
- return openStateDatabase(options.stateDbPath);
1286
- }
1287
- catch (err) {
1288
- const msg = err instanceof Error ? err.message : String(err);
1289
- warn(`[extract] state.db unavailable, processing without skip-tracking: ${msg}`);
1290
- warnings.push(`state.db unavailable: ${msg}`);
1291
- return undefined;
1292
- }
1293
- }
1294
- function emitExtractTriageEvent(args) {
1295
- const { modelPlanCount, triageEnabled, result, sourceRun, eventsCtx } = args;
1296
- if (modelPlanCount === 0 || !triageEnabled || result.triageEvaluated === 0)
1297
- return;
1298
- appendEvent({
1299
- eventType: "extract_triaged",
1300
- metadata: {
1301
- evaluated: result.triageEvaluated,
1302
- passed: result.triagePassed,
1303
- triagedOut: result.triagedOut,
1304
- sourceRun,
1305
- },
1306
- }, eventsCtx);
1307
- }
1308
- /**
1309
- * Count every session's `skipReason` (#912) and push one warning line per
1310
- * infrastructure reason in {@link EXTRACT_INFRASTRUCTURE_SKIP_REASONS}.
1311
- * `undefined` when nothing was skipped, so the envelope carries no key.
1312
- */
1313
- function buildExtractSkipAggregate(sessions, engine, warnings) {
1314
- const counts = {};
1315
- for (const session of sessions) {
1316
- if (!session.skipReason)
1317
- continue;
1318
- counts[session.skipReason] = (counts[session.skipReason] ?? 0) + 1;
1319
- }
1320
- if (Object.keys(counts).length === 0)
1321
- return undefined;
1322
- const total = sessions.length;
1323
- for (const reason of EXTRACT_INFRASTRUCTURE_SKIP_REASONS) {
1324
- const n = counts[reason];
1325
- if (n)
1326
- warnings.push(`${n} of ${total} sessions skipped: ${reason} (engine "${engine}")`);
1327
- }
1328
- return counts;
1329
- }
1330
827
  export async function akmExtract(options) {
1331
828
  const startMs = Date.now();
1332
829
  if (!options.type || options.type.trim() === "") {
1333
830
  throw new UsageError("--type is required. Pass a harness name (e.g. --type claude).", "MISSING_REQUIRED_ARGUMENT");
1334
831
  }
1335
832
  const config = options.config ?? loadConfig();
1336
- const stashDir = resolveRunStashDir(options.stashDir);
833
+ const stashDir = options.stashDir ?? resolveStashDir();
1337
834
  const dryRun = options.dryRun ?? false;
1338
835
  const sourceRun = options.sourceRun ?? `extract-${timestampForFilename()}`;
1339
- // Read process behavior from the frozen standalone plan or the active improve
1340
- // strategy. This prevents config changes during watch mode from changing later
1341
- // triggers and prevents one improve strategy from overriding another.
836
+ // Behavior comes from the frozen standalone plan or the active improve strategy.
1342
837
  const activeProfile = options.improveProfile ?? (options.resolvedPlan ? undefined : resolveImproveStrategy(undefined, config).config);
1343
- const extractProcess = options.resolvedPlan?.process ?? getImproveProcessConfig("extract", activeProfile);
1344
- // The `extract.enabled` process toggle gates extract as a STAGE of `akm improve`
1345
- // (the activeProfile path) — consistent with #593/#594 where the active profile,
1346
- // not `default`, is the source of truth. An EXPLICIT `akm extract` invocation
1347
- // (no activeProfile) is a direct user/cron action and always runs; gating it on
1348
- // the default improve profile's stage toggle was a footgun — dropping extract
1349
- // from the daily improve profile would silently disable the standalone command.
1350
- const extractEnabled = options.resolvedPlan?.enabled ??
838
+ const process = options.resolvedPlan?.process ?? getImproveProcessConfig("extract", activeProfile);
839
+ // The extract toggle gates extract as an improve STAGE; an explicit `akm extract` always runs.
840
+ const enabled = options.resolvedPlan?.enabled ??
1351
841
  (options.improveProfile ? resolveProcessEnabled("extract", options.improveProfile) : true);
1352
- // Feature-gate early so we get a clean "skipped because disabled" envelope.
1353
- if (!extractEnabled) {
842
+ if (!enabled) {
1354
843
  return emptyExtractResult({
1355
844
  ok: true,
1356
845
  dryRun,
@@ -1359,101 +848,72 @@ export async function akmExtract(options) {
1359
848
  startMs,
1360
849
  });
1361
850
  }
1362
- const { timeoutMs, llmRunner, onNotices, getNotices, maxTotalChars, minContentChars, maxSessionsPerRun, effectiveSince, triage, sessionIndexing, } = resolveExtractRunConfig(options, config, extractProcess, activeProfile);
1363
- // WI-9.10: construct this run's RunContext (extracted to
1364
- // buildExtractRunContext to keep akmExtract under the fn-size bar — R31).
1365
- const ctx = buildExtractRunContext({ options, config, stashDir, dryRun, sourceRun, llmRunner });
1366
- const harness = resolveHarness(options.type, options.harnesses);
1367
- if (!harness) {
851
+ const resolved = resolveExtractRun(options, config, process, activeProfile);
852
+ const { llmRunner, notices } = resolved;
853
+ const harness = (options.harnesses ?? getAvailableHarnesses()).find((h) => h.name === options.type);
854
+ const unavailable = !harness
855
+ ? `no available harness matches type "${options.type}" (check that the platform is installed)`
856
+ : !harness.isAvailable()
857
+ ? `harness ${options.type} is registered but reports not-available (no session data on this machine)`
858
+ : undefined;
859
+ if (!harness || unavailable) {
1368
860
  return emptyExtractResult({
1369
861
  ok: false,
1370
862
  dryRun,
1371
863
  type: options.type,
1372
- warning: `no available harness matches type "${options.type}" (check that the platform is installed)`,
864
+ warning: unavailable ?? "",
1373
865
  startMs,
1374
866
  llmRunner,
1375
867
  });
1376
868
  }
1377
- if (!harness.isAvailable()) {
1378
- return emptyExtractResult({
1379
- ok: false,
1380
- dryRun,
1381
- type: options.type,
1382
- warning: `harness ${options.type} is registered but reports not-available (no session data on this machine)`,
1383
- startMs,
1384
- llmRunner,
1385
- });
1386
- }
1387
- // Decide which sessions to process: explicit sessionId OR discovery via since.
1388
- const discovery = discoverExtractCandidates(options, harness, effectiveSince, startMs, dryRun, llmRunner);
869
+ const discovery = discoverExtractCandidates(options, harness, resolved.effectiveSince, startMs, dryRun, llmRunner);
1389
870
  if ("notFound" in discovery)
1390
871
  return discovery.notFound;
1391
- const candidates = discovery.candidates;
1392
872
  const topLevelWarnings = [];
1393
- const trackingEnabled = options.skipTracking !== true;
1394
- const seenMap = loadExtractSeenMapReadOnly({
1395
- options,
1396
- harness: harness.name,
1397
- candidates,
1398
- trackingEnabled,
1399
- warnings: topLevelWarnings,
1400
- });
873
+ const tracking = options.skipTracking !== true;
874
+ const seenMap = loadSeenMapReadOnly(options, harness.name, discovery.candidates, topLevelWarnings);
1401
875
  const planned = planExtractSessions({
1402
- candidates,
876
+ candidates: discovery.candidates,
1403
877
  options,
1404
878
  harness,
1405
879
  seenMap,
1406
- maxTotalChars,
1407
- minContentChars,
1408
- maxSessionsPerRun,
1409
- triage,
1410
- trackingEnabled,
1411
- dryRun,
880
+ gates: resolved.gates,
881
+ maxSessionsPerRun: resolved.maxSessionsPerRun,
882
+ locking: tracking && !dryRun && !options.stateDb,
1412
883
  });
1413
884
  const modelPlanCount = planned.plans.filter((plan) => plan.kind === "model").length;
1414
- // Eligible dry-runs still dispatch to produce their candidate preview. Only
1415
- // deterministic no-work plans are credential-free. Materialize once after
1416
- // every read-only gate and before opening live state or acquiring a lock.
1417
- const dispatchLease = modelPlanCount > 0 ? await preflightStructuredLlmRunner(llmRunner) : undefined;
885
+ // Credentials are materialized once, after every read-only gate and before live state or a lock.
886
+ if (modelPlanCount > 0)
887
+ assertRunnerCredentials(llmRunner);
1418
888
  let stateDb;
1419
- let loopResult;
889
+ let tally;
1420
890
  try {
1421
- stateDb = openExtractLiveStateDb({
1422
- options,
1423
- trackingEnabled,
1424
- hasModelWork: modelPlanCount > 0,
1425
- dryRun,
1426
- warnings: topLevelWarnings,
1427
- });
1428
- // Stash authoring standards (convention/meta fact bodies) for non-wiki
1429
- // extract output. Resolved ONCE per run and threaded into each session's
1430
- // prompt so facts are not re-read per session.
1431
- const extractStandardsContext = modelPlanCount > 0 ? resolveExtractStandards(stashDir) : "";
1432
- loopResult = await runExtractSessionLoop({
1433
- plans: planned.plans,
1434
- deferredCandidates: planned.deferredCandidates,
1435
- seenMap,
891
+ if (tracking) {
892
+ if (options.stateDb) {
893
+ stateDb = options.stateDb;
894
+ }
895
+ else if (modelPlanCount > 0 && !dryRun) {
896
+ try {
897
+ stateDb = openStateDatabase(options.stateDbPath);
898
+ }
899
+ catch (err) {
900
+ const msg = err instanceof Error ? err.message : String(err);
901
+ warn(`[extract] state.db unavailable, processing without skip-tracking: ${msg}`);
902
+ topLevelWarnings.push(`state.db unavailable: ${msg}`);
903
+ }
904
+ }
905
+ }
906
+ const run = {
907
+ ...resolved,
1436
908
  options,
1437
909
  harness,
1438
- stateDb,
1439
- trackingEnabled,
1440
- dryRun,
1441
910
  stashDir,
1442
911
  config,
1443
- llmRunner,
1444
- lease: dispatchLease,
1445
- onNotices,
1446
- getNotices,
1447
- chat: options.chat,
1448
912
  sourceRun,
1449
- timeoutMs,
1450
- maxTotalChars,
1451
- minContentChars,
1452
- triage,
1453
- sessionIndexing,
1454
- extractStandardsContext,
1455
- topLevelWarnings,
1456
- });
913
+ dryRun,
914
+ standardsContext: modelPlanCount > 0 ? resolveExtractStandards(stashDir) : "",
915
+ };
916
+ tally = await runExtractSessionLoop(run, planned, seenMap, stateDb, tracking, topLevelWarnings);
1457
917
  }
1458
918
  finally {
1459
919
  if (stateDb && !options.stateDb) {
@@ -1461,70 +921,62 @@ export async function akmExtract(options) {
1461
921
  stateDb.close();
1462
922
  }
1463
923
  catch {
1464
- // best-effort close
924
+ // best-effort
1465
925
  }
1466
926
  }
1467
- if (dispatchLease)
1468
- disposeLoweredExecutionDispatchLease(dispatchLease);
1469
927
  }
1470
- const { sessions, processedCount, skippedCount, allProposalIds } = loopResult;
1471
- if (loopResult.deferred > 0) {
1472
- topLevelWarnings.push(`Reached maxSessionsPerRun=${maxSessionsPerRun}; ${loopResult.deferred} session(s) deferred to a later run.`);
928
+ if (tally.deferred > 0) {
929
+ topLevelWarnings.push(`Reached maxSessionsPerRun=${resolved.maxSessionsPerRun}; ${tally.deferred} session(s) deferred to a later run.`);
930
+ }
931
+ // Every skip reason is counted; infrastructure failures also get a warning line.
932
+ const counts = {};
933
+ for (const session of tally.sessions) {
934
+ if (session.skipReason)
935
+ counts[session.skipReason] = (counts[session.skipReason] ?? 0) + 1;
936
+ }
937
+ for (const reason of EXTRACT_INFRASTRUCTURE_SKIP_REASONS) {
938
+ const n = counts[reason];
939
+ if (n)
940
+ topLevelWarnings.push(`${n} of ${tally.sessions.length} sessions skipped: ${reason} (engine "${llmRunner.engine}")`);
941
+ }
942
+ if (modelPlanCount > 0 && resolved.gates.triage.enabled && tally.triageEvaluated > 0) {
943
+ appendEvent({
944
+ eventType: "extract_triaged",
945
+ metadata: {
946
+ evaluated: tally.triageEvaluated,
947
+ passed: tally.triagePassed,
948
+ triagedOut: tally.triagedOut,
949
+ sourceRun,
950
+ },
951
+ }, options.eventsCtx);
1473
952
  }
1474
- const skipReasons = buildExtractSkipAggregate(sessions, llmRunner.engine, topLevelWarnings);
1475
- emitExtractTriageEvent({
1476
- modelPlanCount,
1477
- triageEnabled: triage.enabled,
1478
- result: loopResult,
1479
- sourceRun,
1480
- eventsCtx: options.eventsCtx,
1481
- });
1482
953
  return {
1483
954
  schemaVersion: 1,
1484
955
  ok: true,
1485
956
  shape: "extract-result",
1486
- // Sourced from ctx (identical value to the local `dryRun` — see the
1487
- // RunContext construction above) so the constructed RunContext has a
1488
- // genuine downstream reference in this verb, which currently has no
1489
- // content-read site to route through ctx.readAsset (see the WI-9.10c
1490
- // report).
1491
- dryRun: ctx.dryRun,
957
+ dryRun,
1492
958
  type: options.type,
1493
- sessionsProcessed: processedCount,
1494
- sessionsSkipped: skippedCount,
1495
- candidatesCreated: allProposalIds.length,
1496
- proposals: allProposalIds,
1497
- sessions,
959
+ sessionsProcessed: tally.processedCount,
960
+ sessionsSkipped: tally.skippedCount,
961
+ candidatesCreated: tally.allProposalIds.length,
962
+ proposals: tally.allProposalIds,
963
+ sessions: tally.sessions,
1498
964
  warnings: topLevelWarnings,
1499
965
  durationMs: Date.now() - startMs,
1500
- ...(getNotices().length > 0 ? { notices: getNotices() } : {}),
1501
- ...(skipReasons ? { skipReasons } : {}),
966
+ ...notices.fields(),
967
+ ...(Object.keys(counts).length > 0 ? { skipReasons: counts } : {}),
1502
968
  engine: llmRunner.engine,
1503
969
  engineKind: llmRunner.kind,
1504
970
  };
1505
971
  }
1506
972
  /**
1507
- * Count NEW (unseen, in-window) extract candidate sessions across all available
1508
- * harnesses WITHOUT making any LLM calls. Mirrors the discovery + seen-filter
1509
- * logic in {@link akmExtract} so the `#554 minNewSessions` gate in `improve`
1510
- * can decide whether the extract pass is worth running before any work begins.
1511
- *
1512
- * #602 — this gate is intentionally CHEAP: it does NOT read session bodies, so
1513
- * it cannot compute the content hash that {@link shouldSkipAlreadyExtractedSession}
1514
- * now uses. It therefore uses a CONSERVATIVE row-presence approximation: a
1515
- * session counts as "new" when there is NO prior row OR the prior row's
1516
- * `content_hash` is null (never-seen or backfill-eligible). A prior row WITH a
1517
- * non-null content_hash counts as NOT new — it MIGHT have changed, but the
1518
- * precise per-session hash check happens downstream in processSession, so an
1519
- * over-/under-count here only affects whether the pass RUNS, never whether a
1520
- * changed session is actually re-processed.
973
+ * Count new in-window sessions across the available harnesses, with no LLM
974
+ * call, for improve's `minNewSessions` gate. It does not read session bodies,
975
+ * so a session counts as new when it has no seen row or a row without a
976
+ * content hash; the exact hash check happens at extraction.
1521
977
  */
1522
978
  export function countNewExtractCandidates(_config, options = {}) {
1523
- const extractProcess = getImproveProcessConfig("extract", options.improveProfile);
1524
- const effectiveSince = options.since ?? extractProcess?.defaultSince;
1525
- // Mirror akmExtract: when no explicit window is set, default per-harness to
1526
- // "since the last run" (floored at 48h) instead of a fixed 24h. Keeps this
1527
- // gate's discovery window identical to what akmExtract will actually scan.
979
+ const effectiveSince = options.since ?? getImproveProcessConfig("extract", options.improveProfile)?.defaultSince;
1528
980
  const explicitSinceMs = effectiveSince ? parseSinceArg(effectiveSince) : undefined;
1529
981
  const harnesses = (options.harnesses ?? getAvailableHarnesses()).filter((h) => h.isAvailable());
1530
982
  let stateDb = options.stateDb;
@@ -1538,20 +990,15 @@ export function countNewExtractCandidates(_config, options = {}) {
1538
990
  ...(options.stateDbPath ? { stateDbPath: options.stateDbPath } : {}),
1539
991
  ...(options.readOnly && !options.stateDb ? { skipTracking: true } : {}),
1540
992
  });
1541
- const candidates = harness.listSessions({
1542
- sinceMs,
1543
- ...(options.readOnly ? { isolatedSnapshot: true } : {}),
1544
- });
993
+ const candidates = harness.listSessions({ sinceMs, ...(options.readOnly ? { isolatedSnapshot: true } : {}) });
1545
994
  if (candidates.length === 0)
1546
995
  continue;
1547
- // A dry planner with no pre-existing state database has no seen-session
1548
- // ledger by definition. Count the discovered sessions directly instead
1549
- // of creating state.db merely to prove that it is empty.
996
+ // A dry planner without state.db has no seen-session ledger by definition.
1550
997
  if (options.readOnly && !stateDb) {
1551
998
  total += candidates.length;
1552
999
  continue;
1553
1000
  }
1554
- let seenMap = new Map();
1001
+ let seenMap;
1555
1002
  try {
1556
1003
  if (!stateDb) {
1557
1004
  stateDb = openStateDatabase(options.stateDbPath);
@@ -1560,9 +1007,7 @@ export function countNewExtractCandidates(_config, options = {}) {
1560
1007
  seenMap = getExtractedSessionsMap(stateDb, harness.name, candidates.map((c) => c.sessionId));
1561
1008
  }
1562
1009
  catch (err) {
1563
- // state.db unavailable — treat every in-window session as a new
1564
- // candidate (fail-open: never let a transient sqlite error wrongly
1565
- // trip the gate and skip a pass that should have run).
1010
+ // Fail open: a transient sqlite error must not skip a pass that should run.
1566
1011
  const msg = err instanceof Error ? err.message : String(err);
1567
1012
  warn(`[extract] state.db unavailable while counting candidates, treating all as new: ${msg}`);
1568
1013
  total += candidates.length;
@@ -1570,12 +1015,8 @@ export function countNewExtractCandidates(_config, options = {}) {
1570
1015
  }
1571
1016
  for (const summary of candidates) {
1572
1017
  const prior = seenMap.get(summary.sessionId);
1573
- // #602 row-presence approximation (see fn doc): a prior row WITH a
1574
- // non-null content_hash is treated as not-new here; everything else
1575
- // (never-seen, or null-hash backfill-eligible) counts as new.
1576
- if (prior && prior.content_hash != null)
1577
- continue;
1578
- total += 1;
1018
+ if (!(prior && prior.content_hash != null))
1019
+ total += 1;
1579
1020
  }
1580
1021
  }
1581
1022
  }
@@ -1585,7 +1026,7 @@ export function countNewExtractCandidates(_config, options = {}) {
1585
1026
  stateDb.close();
1586
1027
  }
1587
1028
  catch {
1588
- // best-effort close
1029
+ // best-effort
1589
1030
  }
1590
1031
  }
1591
1032
  }