akm-cli 0.9.16 → 0.9.17-alpha.10

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (403) hide show
  1. package/CHANGELOG.md +2101 -0
  2. package/STABILITY.md +11 -10
  3. package/dist/akm +124 -193
  4. package/dist/akm-migrate +38 -19
  5. package/dist/assets/hints/cli-hints-full.md +6 -7
  6. package/dist/assets/improve-strategies/catchup.json +0 -3
  7. package/dist/assets/improve-strategies/consolidate.json +0 -1
  8. package/dist/assets/improve-strategies/default.json +1 -2
  9. package/dist/assets/improve-strategies/proactive-maintenance.json +1 -2
  10. package/dist/assets/improve-strategies/quick.json +1 -2
  11. package/dist/assets/improve-strategies/reflect-distill.json +1 -2
  12. package/dist/assets/improve-strategies/thorough.json +0 -3
  13. package/dist/assets/prompts/consolidate-pair.md +20 -0
  14. package/dist/assets/prompts/consolidate-system.md +4 -11
  15. package/dist/assets/prompts/retrieval-relevance-judge.md +6 -0
  16. package/dist/assets/stash-skeleton/facts/conventions/backlinks.md +20 -20
  17. package/dist/assets/stash-skeleton/facts/conventions/domains.md +2 -2
  18. package/dist/assets/templates/html/health.html +3 -5
  19. package/dist/cli/retired-commands.js +1 -1
  20. package/dist/cli/shared.js +6 -2
  21. package/dist/cli/unknown-flags.js +24 -1
  22. package/dist/cli.js +68 -10
  23. package/dist/commands/agent/agent-dispatch.js +1 -1
  24. package/dist/commands/command/command-execution.js +24 -62
  25. package/dist/commands/feedback-cli.js +0 -1
  26. package/dist/commands/health/accept-rate.js +6 -0
  27. package/dist/commands/health/archive-usage.js +92 -0
  28. package/dist/commands/health/checks.js +83 -74
  29. package/dist/commands/health/config-skew.js +38 -0
  30. package/dist/commands/health/data-dir-usage.js +25 -13
  31. package/dist/commands/health/egress.js +54 -0
  32. package/dist/commands/health/html-report.js +1 -42
  33. package/dist/commands/health/improve-metrics.js +136 -591
  34. package/dist/commands/health/md-report.js +1 -6
  35. package/dist/commands/health/plugin-staleness.js +53 -3
  36. package/dist/commands/health/renderers.js +12 -4
  37. package/dist/commands/health/report-view-model.js +14 -120
  38. package/dist/commands/health/types-improve.js +4 -19
  39. package/dist/commands/health/windows.js +64 -74
  40. package/dist/commands/health.js +145 -143
  41. package/dist/commands/improve/consolidate/chunking.js +26 -117
  42. package/dist/commands/improve/consolidate/continuity-check.js +137 -0
  43. package/dist/commands/improve/consolidate/pair-pass.js +791 -0
  44. package/dist/commands/improve/consolidate/sanitize.js +54 -149
  45. package/dist/commands/improve/consolidate.js +589 -1127
  46. package/dist/commands/improve/content-hash.js +16 -24
  47. package/dist/commands/improve/distill/content-repair.js +18 -100
  48. package/dist/commands/improve/distill-guards.js +20 -81
  49. package/dist/commands/improve/distill-promotion-policy.js +23 -243
  50. package/dist/commands/improve/distill.js +608 -1041
  51. package/dist/commands/improve/eligibility.js +126 -390
  52. package/dist/commands/improve/execution.js +8 -10
  53. package/dist/commands/improve/extract-prompt.js +1 -2
  54. package/dist/commands/improve/extract.js +487 -1046
  55. package/dist/commands/improve/feedback-valence.js +0 -25
  56. package/dist/commands/improve/improve-cli.js +75 -169
  57. package/dist/commands/improve/improve-result-file.js +10 -66
  58. package/dist/commands/improve/improve-strategies.js +52 -4
  59. package/dist/commands/improve/improve-usage-report.js +18 -64
  60. package/dist/commands/improve/improve.js +480 -1074
  61. package/dist/commands/improve/ledger.js +119 -0
  62. package/dist/commands/improve/locks.js +2 -8
  63. package/dist/commands/improve/loop-stages.js +415 -1073
  64. package/dist/commands/improve/memory/derived-ref.js +12 -77
  65. package/dist/commands/improve/memory/memory-belief.js +16 -118
  66. package/dist/commands/improve/memory/memory-improve.js +266 -14
  67. package/dist/commands/improve/outcome-loop.js +28 -156
  68. package/dist/commands/improve/planner.js +5 -15
  69. package/dist/commands/improve/preparation.js +779 -2319
  70. package/dist/commands/improve/proactive-maintenance.js +34 -101
  71. package/dist/commands/improve/reflect-noise.js +104 -280
  72. package/dist/commands/improve/reflect.js +642 -1353
  73. package/dist/commands/improve/retrieval-gate.js +127 -0
  74. package/dist/commands/improve/retrieval-scope.js +92 -0
  75. package/dist/commands/improve/salience.js +41 -240
  76. package/dist/commands/improve/session-asset.js +19 -100
  77. package/dist/commands/improve/stage.js +322 -0
  78. package/dist/commands/lint/base-linter.js +37 -15
  79. package/dist/commands/proposal/drain.js +261 -578
  80. package/dist/commands/proposal/proposal-cli.js +19 -20
  81. package/dist/commands/proposal/proposal-types.js +31 -24
  82. package/dist/commands/proposal/proposal.js +38 -8
  83. package/dist/commands/proposal/propose.js +134 -160
  84. package/dist/commands/proposal/repository.js +1097 -1394
  85. package/dist/commands/proposal/validators/proposal-quality-validators.js +71 -174
  86. package/dist/commands/proposal/validators/proposal-validators.js +1 -1
  87. package/dist/commands/proposal/validators/proposals.js +22 -89
  88. package/dist/commands/read/curate.js +105 -462
  89. package/dist/commands/read/knowledge.js +3 -2
  90. package/dist/commands/read/search-cli.js +16 -33
  91. package/dist/commands/read/search.js +17 -23
  92. package/dist/commands/read/show.js +57 -108
  93. package/dist/commands/sources/bundle-cli.js +25 -2
  94. package/dist/commands/sources/bundle-config-ops.js +4 -0
  95. package/dist/commands/sources/dangerous-env-audit.js +1 -2
  96. package/dist/commands/sources/info.js +127 -29
  97. package/dist/commands/sources/installed-stashes.js +197 -746
  98. package/dist/commands/sources/schema-repair.js +98 -129
  99. package/dist/commands/sources/source-add.js +62 -12
  100. package/dist/commands/sources/source-manage.js +9 -2
  101. package/dist/commands/sources/stash-cli.js +24 -4
  102. package/dist/commands/tasks/explain.js +10 -13
  103. package/dist/commands/tasks/tasks-cli.js +12 -13
  104. package/dist/commands/tasks/tasks.js +350 -936
  105. package/dist/commands/tasks/validate.js +26 -24
  106. package/dist/commands/workflow/plan.js +22 -29
  107. package/dist/commands/workflow-cli.js +4 -4
  108. package/dist/core/adapter/adapters/akm-adapter.js +2 -1
  109. package/dist/core/adapter/adapters/akm-lint.js +2 -3
  110. package/dist/core/adapter/adapters/akm-metadata.js +42 -12
  111. package/dist/core/adapter/adapters/akm-task-adapter.js +29 -8
  112. package/dist/core/adapter/adapters/akm-workflow-adapter.js +1 -1
  113. package/dist/core/adapter/execution-source.js +17 -29
  114. package/dist/core/asset/asset-placement.js +4 -13
  115. package/dist/core/asset/frontmatter.js +106 -1
  116. package/dist/core/asset/resolve-ref.js +1 -1
  117. package/dist/core/bundle-id.js +42 -5
  118. package/dist/core/bundle-rename.js +285 -0
  119. package/dist/core/config/config-io.js +1 -2
  120. package/dist/core/config/config-schema.js +9 -34
  121. package/dist/core/config/config-walker.js +1 -1
  122. package/dist/core/config/config.js +184 -111
  123. package/dist/core/config/engine-semantics.js +0 -2
  124. package/dist/core/config/legacy-source-shape-shim.js +38 -9
  125. package/dist/core/config/schema/embedding.js +20 -5
  126. package/dist/core/config/schema/engines.js +5 -0
  127. package/dist/core/config/schema/execution.js +1 -1
  128. package/dist/core/config/schema/experimental.js +1 -1
  129. package/dist/core/config/schema/improve-processes.js +54 -125
  130. package/dist/core/config/schema/improve.js +4 -42
  131. package/dist/core/config/schema/index-config.js +9 -48
  132. package/dist/core/config/schema/scheduler.js +12 -12
  133. package/dist/core/config/schema/search.js +6 -22
  134. package/dist/core/env-secret-ref.js +0 -1
  135. package/dist/core/errors.js +8 -9
  136. package/dist/core/file-change.js +13 -5
  137. package/dist/core/file-lock.js +76 -173
  138. package/dist/core/improve-result.js +35 -7
  139. package/dist/core/improve-types.js +0 -1
  140. package/dist/core/logs-db.js +2 -2
  141. package/dist/core/loopback.js +7 -12
  142. package/dist/core/non-task-input.js +20 -0
  143. package/dist/core/parse.js +13 -16
  144. package/dist/core/paths.js +0 -24
  145. package/dist/core/redaction.js +109 -2
  146. package/dist/core/run-lock.js +2 -5
  147. package/dist/core/spawn-env.js +1 -1
  148. package/dist/core/state/migrations.js +123 -61
  149. package/dist/core/state-db-scope.js +2 -4
  150. package/dist/core/state-db.js +126 -692
  151. package/dist/core/time.js +0 -20
  152. package/dist/core/type-presentation.js +1 -9
  153. package/dist/core/write-source.js +294 -1005
  154. package/dist/execution/input-contract.js +1 -1
  155. package/dist/execution/resolved-request.js +135 -689
  156. package/dist/execution/source.js +63 -257
  157. package/dist/execution/target-ref.js +1 -1
  158. package/dist/indexer/bundle-identity-guard.js +2 -2
  159. package/dist/indexer/db/llm-cache.js +2 -2
  160. package/dist/indexer/ensure-index.js +77 -73
  161. package/dist/indexer/index-rebuild-lock.js +3 -11
  162. package/dist/indexer/index-writer-lock.js +8 -17
  163. package/dist/indexer/index-written-assets.js +141 -154
  164. package/dist/indexer/indexer.js +400 -1124
  165. package/dist/indexer/links/declared-links.js +90 -0
  166. package/dist/indexer/materialize-embeddings.js +60 -397
  167. package/dist/indexer/passes/memory-inference.js +96 -90
  168. package/dist/indexer/passes/metadata.js +132 -219
  169. package/dist/indexer/read-preflight.js +0 -7
  170. package/dist/indexer/scan/doc-to-entry.js +2 -3
  171. package/dist/indexer/scan/drain-dir.js +1 -1
  172. package/dist/indexer/search/db-search.js +190 -590
  173. package/dist/indexer/search/fts-query.js +30 -41
  174. package/dist/indexer/search/ranking.js +28 -154
  175. package/dist/indexer/search/search-attribution.js +12 -32
  176. package/dist/indexer/search/search-fields.js +11 -15
  177. package/dist/indexer/search/search-hit-enrichers.js +54 -85
  178. package/dist/indexer/search/search-source.js +1 -4
  179. package/dist/indexer/usage/usage-events.js +36 -7
  180. package/dist/indexer/walk/walker.js +3 -4
  181. package/dist/integrations/agent/engine-fallback.js +23 -40
  182. package/dist/integrations/agent/engine-resolution.js +93 -183
  183. package/dist/integrations/agent/execution.js +507 -0
  184. package/dist/integrations/agent/model-map.js +28 -156
  185. package/dist/integrations/agent/request-lowering.js +66 -141
  186. package/dist/integrations/agent/runner-dispatch.js +143 -321
  187. package/dist/integrations/agent/runner.js +54 -14
  188. package/dist/integrations/lockfile.js +53 -101
  189. package/dist/llm/client.js +18 -6
  190. package/dist/llm/embedders/deterministic.js +2 -3
  191. package/dist/llm/embedders/profile.js +71 -0
  192. package/dist/llm/embedders/remote.js +11 -17
  193. package/dist/llm/feature-gate.js +0 -8
  194. package/dist/llm/index-passes.js +3 -5
  195. package/dist/llm/memory-infer.js +1 -2
  196. package/dist/llm/structured-call.js +5 -24
  197. package/dist/output/generic-render.js +23 -11
  198. package/dist/output/html-render.js +13 -10
  199. package/dist/output/render-registry.js +3 -32
  200. package/dist/output/shapes/helpers.js +25 -38
  201. package/dist/output/shapes/passthrough.js +1 -9
  202. package/dist/{indexer/graph/graph-types.js → output/text/bundle-rename.js} +4 -1
  203. package/dist/output/text/command-format.js +69 -31
  204. package/dist/output/text/helpers.js +1 -1
  205. package/dist/output/text/migrate.js +5 -14
  206. package/dist/output/text/proposal-format.js +48 -3
  207. package/dist/output/text/show-format.js +13 -17
  208. package/dist/output/text/workflow-format.js +0 -32
  209. package/dist/output/text.js +2 -0
  210. package/dist/registry/factory.js +4 -19
  211. package/dist/registry/network.js +66 -220
  212. package/dist/registry/providers/index.js +0 -2
  213. package/dist/registry/providers/skills-sh.js +3 -14
  214. package/dist/registry/providers/static-index.js +24 -26
  215. package/dist/registry/resolve.js +55 -131
  216. package/dist/scripts/akm-migrate-node.js +42948 -92369
  217. package/dist/scripts/akm-migrate.js +42935 -92354
  218. package/dist/setup/registry-stash-loader.js +4 -13
  219. package/dist/setup/semantic-assets.js +3 -44
  220. package/dist/setup/setup.js +1 -1
  221. package/dist/setup/steps/connection.js +5 -6
  222. package/dist/setup/steps/platforms.js +2 -2
  223. package/dist/setup/steps/tasks.js +25 -15
  224. package/dist/sources/provider-factory.js +17 -18
  225. package/dist/sources/providers/filesystem.js +2 -3
  226. package/dist/sources/providers/git-install.js +7 -1
  227. package/dist/sources/providers/git-provider.js +0 -3
  228. package/dist/sources/providers/git-stash.js +83 -21
  229. package/dist/sources/providers/npm.js +2 -4
  230. package/dist/sources/providers/provider-utils.js +5 -10
  231. package/dist/sources/providers/website.js +0 -2
  232. package/dist/sources/snapshot-fetchers/website-ingest.js +1 -1
  233. package/dist/sources/website-url.js +2 -2
  234. package/dist/storage/database.js +9 -35
  235. package/dist/storage/repositories/improve-ledger-repository.js +209 -0
  236. package/dist/storage/repositories/index-connection.js +39 -72
  237. package/dist/storage/repositories/index-entries-repository.js +131 -129
  238. package/dist/storage/repositories/index-entry-mapper.js +1 -2
  239. package/dist/storage/repositories/index-entry-schema.js +101 -268
  240. package/dist/storage/repositories/index-fts-repository.js +86 -256
  241. package/dist/storage/repositories/index-links-repository.js +143 -0
  242. package/dist/storage/repositories/index-llm-cache-repository.js +7 -9
  243. package/dist/storage/repositories/index-meta-repository.js +6 -4
  244. package/dist/storage/repositories/index-schema.js +257 -325
  245. package/dist/storage/repositories/index-utility-repository.js +8 -29
  246. package/dist/storage/repositories/index-vec-repository.js +133 -414
  247. package/dist/storage/repositories/outcome-repository.js +2 -1
  248. package/dist/storage/repositories/proposals-repository.js +104 -1
  249. package/dist/storage/repositories/registry-index-cache-repository.js +100 -0
  250. package/dist/storage/repositories/salience-repository.js +1 -19
  251. package/dist/storage/repositories/task-history-repository.js +26 -4
  252. package/dist/storage/repositories/workflow-runs-repository.js +53 -244
  253. package/dist/storage/sqlite-migrations.js +136 -0
  254. package/dist/storage/sqlite-pragmas.js +11 -9
  255. package/dist/storage/sqlite-transaction.js +170 -0
  256. package/dist/storage/state-db-integrity.js +130 -0
  257. package/dist/tasks/activation-config.js +134 -62
  258. package/dist/tasks/backends/cron.js +191 -302
  259. package/dist/tasks/backends/exec-utils.js +2 -5
  260. package/dist/tasks/backends/launchd.js +141 -748
  261. package/dist/tasks/backends/schtasks.js +119 -623
  262. package/dist/tasks/prepare/prepare-support.js +5 -15
  263. package/dist/tasks/prepare/prepare.js +0 -2
  264. package/dist/tasks/resolve-akm-bin.js +20 -79
  265. package/dist/tasks/run/attempt-lifecycle.js +0 -1
  266. package/dist/tasks/run/load-task.js +1 -1
  267. package/dist/tasks/scheduler-binding.js +20 -238
  268. package/dist/tasks/scheduler-invocation.js +136 -244
  269. package/dist/tasks/scheduler-lock.js +53 -0
  270. package/dist/tasks/scheduler-sync.js +368 -679
  271. package/dist/tasks/source/parse-task-source.js +55 -9
  272. package/dist/tasks/source/task-source-v3-frozen.js +3 -4
  273. package/dist/tasks/source/task-to-v4.js +464 -88
  274. package/dist/workflows/authoring/authoring.js +3 -12
  275. package/dist/workflows/compile.js +211 -0
  276. package/dist/workflows/concurrency-policy.js +13 -74
  277. package/dist/workflows/exec/child-invocation.js +3 -17
  278. package/dist/workflows/exec/child-workflow.js +32 -141
  279. package/dist/workflows/exec/dispatch-redaction.js +13 -53
  280. package/dist/workflows/exec/environment.js +98 -0
  281. package/dist/workflows/exec/exec-unit.js +33 -140
  282. package/dist/workflows/exec/frozen-judge.js +7 -59
  283. package/dist/workflows/exec/native-executor.js +82 -341
  284. package/dist/workflows/exec/param-secrets.js +29 -47
  285. package/dist/workflows/exec/run-workflow.js +154 -387
  286. package/dist/workflows/exec/scheduler.js +9 -36
  287. package/dist/workflows/exec/step-work.js +127 -430
  288. package/dist/workflows/exec/unit-dispatch.js +11 -63
  289. package/dist/workflows/exec/unit-writer.js +8 -52
  290. package/dist/workflows/exec/worktree.js +39 -273
  291. package/dist/workflows/freeze/child-output-references.js +4 -15
  292. package/dist/workflows/freeze/environment.js +99 -92
  293. package/dist/workflows/freeze/freeze.js +172 -0
  294. package/dist/workflows/freeze/step-values.js +19 -21
  295. package/dist/workflows/freeze/targets/child-workflow.js +23 -92
  296. package/dist/workflows/freeze/targets/command.js +10 -33
  297. package/dist/workflows/freeze/targets/script.js +5 -12
  298. package/dist/workflows/freeze/targets/shell.js +3 -6
  299. package/dist/workflows/freeze/targets/task.js +25 -80
  300. package/dist/workflows/freeze/task-bindings.js +20 -67
  301. package/dist/workflows/{source-ir/github-yaml.js → github-yaml.js} +88 -206
  302. package/dist/workflows/ir/params.js +6 -51
  303. package/dist/workflows/ir/plan-hash.js +2 -34
  304. package/dist/workflows/parser.js +140 -43
  305. package/dist/{commands/improve/consolidate/types.js → workflows/plan.js} +2 -1
  306. package/dist/workflows/renderer.js +36 -69
  307. package/dist/workflows/resource-limits.js +12 -120
  308. package/dist/workflows/runtime/agent-identity.js +8 -40
  309. package/dist/workflows/runtime/run-outputs.js +3 -6
  310. package/dist/workflows/runtime/run-plan.js +316 -0
  311. package/dist/workflows/runtime/runs.js +48 -200
  312. package/dist/workflows/runtime/workflow-asset-loader.js +24 -57
  313. package/dist/workflows/{source-ir/semantics.js → source-semantics.js} +16 -20
  314. package/dist/workflows/validate-summary.js +2 -7
  315. package/docs/integration/bundling-akm.md +49 -42
  316. package/docs/migration/README.md +1 -0
  317. package/docs/migration/release-notes/0.9.17.md +43 -0
  318. package/docs/migration/v0.9.1-to-v0.9.2.md +23 -7
  319. package/docs/reference/cli.md +232 -135
  320. package/docs/reference/configuration.md +71 -57
  321. package/docs/reference/data-and-telemetry.md +20 -21
  322. package/docs/reference/tasks.md +105 -39
  323. package/docs/reference/workflow-schema.md +14 -18
  324. package/docs/reference/workflows.md +6 -9
  325. package/package.json +1 -1
  326. package/schemas/akm-config.json +115 -738
  327. package/schemas/akm-workflow.json +1 -0
  328. package/dist/assets/improve-strategies/graph-refresh.json +0 -15
  329. package/dist/assets/prompts/contradiction-judge.md +0 -33
  330. package/dist/assets/prompts/graph-extract-system.md +0 -1
  331. package/dist/assets/prompts/graph-extract-user-prompt.md +0 -35
  332. package/dist/assets/prompts/metadata-enhance-system.md +0 -1
  333. package/dist/assets/tasks/improve/akm-graph-refresh-weekly.yml +0 -4
  334. package/dist/commands/health/advisories.js +0 -150
  335. package/dist/commands/health/metrics.js +0 -329
  336. package/dist/commands/health/surfaces.js +0 -102
  337. package/dist/commands/improve/anti-collapse.js +0 -83
  338. package/dist/commands/improve/collapse-detector.js +0 -432
  339. package/dist/commands/improve/consolidate/eligibility.js +0 -48
  340. package/dist/commands/improve/consolidate/merge.js +0 -149
  341. package/dist/commands/improve/distill/promote-memory.js +0 -291
  342. package/dist/commands/improve/distill/quality-gate.js +0 -337
  343. package/dist/commands/improve/eval-cases.js +0 -52
  344. package/dist/commands/improve/memory/memory-contradiction-detect.js +0 -291
  345. package/dist/commands/improve/proposal-envelope.js +0 -31
  346. package/dist/commands/improve/run-context.js +0 -123
  347. package/dist/commands/improve/shared.js +0 -31
  348. package/dist/commands/improve/source-identity.js +0 -28
  349. package/dist/commands/improve/triage.js +0 -96
  350. package/dist/commands/proposal/drain-policies.js +0 -151
  351. package/dist/commands/sources/update-transaction.js +0 -220
  352. package/dist/core/action-contributors.js +0 -28
  353. package/dist/core/config/config-version-shim.js +0 -101
  354. package/dist/core/fs-txn.js +0 -405
  355. package/dist/core/lexical-score.js +0 -25
  356. package/dist/core/maintenance-barrier.js +0 -167
  357. package/dist/execution/executable-identity.js +0 -105
  358. package/dist/execution/guarded-source.js +0 -427
  359. package/dist/indexer/db/graph-db.js +0 -444
  360. package/dist/indexer/graph/graph-boost.js +0 -427
  361. package/dist/indexer/graph/graph-dedup.js +0 -95
  362. package/dist/indexer/graph/graph-extraction.js +0 -1108
  363. package/dist/indexer/search/name-match.js +0 -35
  364. package/dist/indexer/search/ranking-contributors.js +0 -515
  365. package/dist/indexer/search/ranking-types.js +0 -4
  366. package/dist/indexer/walk/project-context.js +0 -192
  367. package/dist/integrations/agent/execution-cascade.js +0 -566
  368. package/dist/integrations/agent/execution-definitions.js +0 -202
  369. package/dist/integrations/agent/execution-lowering.js +0 -841
  370. package/dist/integrations/agent/execution-preparation.js +0 -98
  371. package/dist/integrations/agent/inline-execution.js +0 -74
  372. package/dist/llm/graph-extract.js +0 -728
  373. package/dist/llm/metadata-enhance.js +0 -96
  374. package/dist/registry/create-provider-registry.js +0 -29
  375. package/dist/registry/pinned-request-helper.js +0 -247
  376. package/dist/registry/pinned-transport.js +0 -717
  377. package/dist/sources/providers/index.js +0 -14
  378. package/dist/storage/engines/sqlite-migrations.js +0 -271
  379. package/dist/storage/repositories/canaries-repository.js +0 -107
  380. package/dist/storage/repositories/embedding-salvage-repository.js +0 -184
  381. package/dist/storage/repositories/registry-cache.js +0 -113
  382. package/dist/tasks/scheduler-sync-preview.js +0 -52
  383. package/dist/tasks/source/task-to-v3.js +0 -507
  384. package/dist/workflows/freeze/resolve-steps.js +0 -86
  385. package/dist/workflows/freeze/source-freeze.js +0 -64
  386. package/dist/workflows/ir/compile.js +0 -321
  387. package/dist/workflows/ir/environment-v4.js +0 -330
  388. package/dist/workflows/ir/freeze-v4.js +0 -153
  389. package/dist/workflows/ir/schema-v4.js +0 -745
  390. package/dist/workflows/ir/schema.js +0 -354
  391. package/dist/workflows/program/schema.js +0 -77
  392. package/dist/workflows/runtime/checkin.js +0 -57
  393. package/dist/workflows/runtime/plan-classifier.js +0 -196
  394. package/dist/workflows/runtime/unit-checkin.js +0 -45
  395. package/dist/workflows/runtime/unit-phases.js +0 -20
  396. package/dist/workflows/schema.js +0 -4
  397. package/dist/workflows/source-ir/compile.js +0 -200
  398. package/dist/workflows/source-ir/program.js +0 -50
  399. package/dist/workflows/source-ir/result.js +0 -26
  400. package/dist/workflows/source-ir/schema.js +0 -786
  401. package/dist/workflows/source-ir/triggers.js +0 -79
  402. package/dist/workflows/source-ir/uses.js +0 -40
  403. package/dist/workflows/validator.js +0 -60
@@ -2,24 +2,13 @@
2
2
  // License, v. 2.0. If a copy of the MPL was not distributed with this
3
3
  // file, You can obtain one at https://mozilla.org/MPL/2.0/.
4
4
  /**
5
- * `akm reflect [ref]` — proposal-producing agent command (#226).
5
+ * `akm reflect [ref]` — ask an engine for a revised asset and queue it as a
6
+ * proposal (`source: "reflect"`). Reflect never writes an asset: the proposal
7
+ * queue is the only path, `akm proposal accept` the bridge.
6
8
  *
7
- * Pipeline:
8
- *
9
- * 1. Emit `reflect_invoked` event at command entry (always, even on failure).
10
- * 2. If `ref` is provided, look the asset up via the FTS index and read its
11
- * content. Pull recent feedback (`feedback` events for that ref) and
12
- * lesson-lint findings to surface as schema hints.
13
- * 3. Build the prompt via {@link buildReflectPrompt}.
14
- * 4. Prepare, authorize, lower, and dispatch the frozen engine selection.
15
- * 5. Parse the agent's stdout into a {@link AgentProposalPayload}.
16
- * 6. Insert into the proposal queue via {@link createProposal} with
17
- * `source: "reflect"`.
18
- *
19
- * Failures are surfaced as structured envelopes carrying an
20
- * {@link AgentFailureReason} discriminant. Reflect NEVER calls
21
- * `writeAssetToSource` directly — the proposal queue is the only path to
22
- * a committed asset, and the `accept` flow is the bridge.
9
+ * Every invocation closes with one `reflect_completed` event; `reflect_invoked`
10
+ * is emitted once the dispatch has validated its credentials (deterministic
11
+ * pre-dispatch refusals still emit both).
23
12
  */
24
13
  import fs from "node:fs";
25
14
  import os from "node:os";
@@ -28,6 +17,7 @@ import { assembleAssetFromString, serializeFrontmatter } from "../../core/asset/
28
17
  import { parseFrontmatter } from "../../core/asset/frontmatter.js";
29
18
  import { conceptIdFromTypeName, parseRefInput } from "../../core/asset/resolve-ref.js";
30
19
  import { DESCRIPTION_MAX_CHARS, requiresDescription } from "../../core/authoring-rules.js";
20
+ import { resolveStashDir } from "../../core/common.js";
31
21
  import { loadConfig } from "../../core/config/config.js";
32
22
  import { generatedContentRejection, stripReflectPromptScaffolding } from "../../core/content-safety.js";
33
23
  import { ConfigError, UsageError } from "../../core/errors.js";
@@ -40,77 +30,46 @@ import { warn, warnOnce } from "../../core/warn.js";
40
30
  import { lookup } from "../../indexer/indexer.js";
41
31
  import { DEFAULT_LLM_TIMEOUT_MS } from "../../integrations/agent/config.js";
42
32
  import { fallbackAnnouncement, NO_ENGINE_MESSAGE_SUFFIX, NO_ENGINE_REMEDY, withEngineFallback, } from "../../integrations/agent/engine-fallback.js";
43
- import { acquireLoweredExecutionDispatchLease, dispatchLoweredExecutionRequest, disposeLoweredExecutionDispatchLease, lowerResolvedExecutionRequest, lowerResolvedExecutionRequestWithRunner, } from "../../integrations/agent/execution-lowering.js";
44
- import { prepareInlineExecution, prepareInlineExecutionWithRunner } from "../../integrations/agent/inline-execution.js";
33
+ import { buildExecution, resolveExecution } from "../../integrations/agent/execution.js";
45
34
  import { buildReflectOutputRepairPrompt, buildReflectPrompt, extractDraftConfidence, parseAgentProposalPayload, REFLECT_CONTENT_CAP, REFLECT_TRUNCATION_MARKER, } from "../../integrations/agent/prompts.js";
46
35
  import { runnerIsLlm, runnerSupportsFileWrite } from "../../integrations/agent/runner.js";
47
- import { collectDispatchSensitiveValues } from "../../integrations/agent/runner-dispatch.js";
36
+ import { assertRunnerCredentials, collectDispatchSensitiveValues, runExecution, } from "../../integrations/agent/runner-dispatch.js";
48
37
  import { isJsonSchemaKnownUnsupported, LlmCallError } from "../../llm/client.js";
49
- import { callStructured } from "../../llm/structured-call.js";
50
38
  import { baseFailureFields, enoentHintMessage, isEnoentFailure } from "../agent/agent-support.js";
51
- import { isProposalSkipped, listProposalsReadOnly, proposalContent, recordGateDecision, } from "../proposal/repository.js";
52
39
  import { checkReflectSize, isValidDescription } from "../proposal/validators/proposal-quality-validators.js";
53
40
  import { CHARS_PER_TOKEN, DEFAULT_CONTEXT_LENGTH_TOKENS } from "./consolidate/chunking.js";
54
41
  import { deriveLessonRef } from "./distill.js";
55
- import { runReflectQualityJudge } from "./distill/quality-gate.js";
56
42
  import { findAssetFilePath } from "./eligibility.js";
57
43
  import { resolveImproveLlmExecution } from "./execution.js";
58
- import { emitProposal } from "./proposal-envelope.js";
59
- import { classifyReflectChange } from "./reflect-noise.js";
60
- import { createRunContext, resolveRunStashDir } from "./run-context.js";
61
- import { MAX_REJECTED_PROPOSALS } from "./shared.js";
62
- import { durableImproveRef, improveStateReadRefs } from "./source-identity.js";
63
- function collectLoweringNotices(target, notices) {
64
- for (const notice of notices)
65
- target.set(JSON.stringify(notice), notice);
66
- }
67
- function reflectNoticeFields(notices) {
68
- return notices.size > 0 ? { notices: Object.freeze([...notices.values()]) } : {};
69
- }
44
+ import { recordLedgerAttempt } from "./ledger.js";
45
+ import { classifyReflectChange, splitFrontmatter } from "./reflect-noise.js";
46
+ import { loadRetrievalQueries, runRetrievalRegressionGate } from "./retrieval-gate.js";
47
+ import { callStage, mintProposal, noticeSet, rejectedProposalContext, runReflectQualityJudge, } from "./stage.js";
70
48
  const MAX_FEEDBACK_LINES = 10;
71
49
  const MAX_GLOBAL_FEEDBACK_LINES = 20;
72
- /**
73
- * Pull recent `feedback` events from events.jsonl. When `ref` is present we
74
- * scope to that asset; otherwise we surface the most recent feedback across
75
- * all assets so `akm reflect` can operate in a general "review recent
76
- * signals" mode. Best-effort — a missing or empty events stream returns `[]`.
77
- */
78
50
  function readOnlyEventsContext(ctx) {
79
51
  return ctx?.db ? ctx : { ...(ctx ?? {}), readOnly: true };
80
52
  }
53
+ /** Recent `feedback` lines for `ref` (or across all assets without one). Best-effort. */
81
54
  function readRecentFeedback(ref, eventsCtx) {
82
55
  try {
83
56
  const events = readEvents({ type: "feedback", ...(ref ? { ref } : {}) }, readOnlyEventsContext(eventsCtx)).events;
84
- const lines = [];
85
- const limit = ref ? MAX_FEEDBACK_LINES : MAX_GLOBAL_FEEDBACK_LINES;
86
- for (const event of events.slice(-limit)) {
87
- const md = (event.metadata ?? {});
57
+ return events.slice(-(ref ? MAX_FEEDBACK_LINES : MAX_GLOBAL_FEEDBACK_LINES)).map((event) => {
58
+ const md = event.metadata ?? {};
88
59
  const signal = typeof md.signal === "string" ? md.signal : "?";
89
60
  const note = typeof md.reason === "string" ? md.reason : typeof md.note === "string" ? md.note : "";
90
61
  const details = note ? `[${signal}] ${note}` : `[${signal}]`;
91
- lines.push(!ref && event.ref ? `${event.ref} ${details}` : details);
92
- }
93
- return lines;
62
+ return !ref && event.ref ? `${event.ref} ${details}` : details;
63
+ });
94
64
  }
95
65
  catch {
96
66
  return [];
97
67
  }
98
68
  }
99
69
  /**
100
- * Asset types that reflect is allowed to operate on.
101
- *
102
- * Reflect's canonical output shape is `frontmatter + markdown body`. Running it
103
- * against types whose on-disk form is NOT markdown (executable scripts, env files
104
- * env files, YAML tasks) blindly prepends `---\n…\n---\n` to the asset and
105
- * breaks the runtime contract — for example a `.ts` script with a YAML preamble
106
- * is a TypeScript syntax error.
107
- *
108
- * Whitelisting (rather than blacklisting) keeps the door closed by default as
109
- * new asset types are registered. To allow a custom registered type, extend
110
- * this set explicitly.
111
- *
112
- * Observed regression: proposal `8737ab63` (May 2026) prepended frontmatter to
113
- * a `.ts` script file via reflect. This whitelist prevents that.
70
+ * Types reflect may rewrite: its output is frontmatter + markdown, which would
71
+ * break a script or env file. Another type is allowed only when its current
72
+ * content already has that shape; secrets are never read.
114
73
  */
115
74
  export const REFLECT_ALLOWED_TYPES = new Set([
116
75
  "knowledge",
@@ -122,129 +81,68 @@ export const REFLECT_ALLOWED_TYPES = new Set([
122
81
  "workflow",
123
82
  ]);
124
83
  const REFLECT_REFUSED_TYPES = new Set(["secret"]);
125
- function isReflectableSourceShape(content) {
126
- return parseFrontmatter(content).frontmatter !== null;
127
- }
128
- /**
129
- * Identity / structural frontmatter fields the LLM is NEVER allowed to change.
130
- *
131
- * Renaming `name` on a skill silently breaks ref resolution because the ref is
132
- * derived from the on-disk path. Similar reasoning for `ref`, `id`, `slug`,
133
- * and `type`. The post-processor below restores any of these fields if the
134
- * LLM tried to rewrite them.
135
- *
136
- * Observed regression: proposal `26941510` (May 2026) renamed
137
- * `skills/openpalm-stack-diagnostics`'s `name` field to `"diagnostic-checklist"`.
138
- */
84
+ /** Identity fields the model may never change (a renamed `name` breaks ref resolution). */
139
85
  const PROTECTED_FRONTMATTER_FIELDS = new Set(["name", "ref", "id", "slug", "type"]);
140
86
  /**
141
- * Read the last 1–3 archived rejected proposals for a given ref from the
142
- * proposal store. Returns `[]` when the proposals store is absent (not yet
143
- * created) or the ref is undefined — `listProposalsReadOnly` already handles
144
- * that case; a genuine read failure propagates instead of being swallowed,
145
- * since silently dropping this Reflexion-style context risks re-proposing
146
- * content that was already rejected (arXiv:2303.11366).
147
- */
148
- function readRejectedProposals(stash, ref, proposalsCtx) {
149
- if (!ref)
150
- return [];
151
- return listProposalsReadOnly(stash, { ref, status: "rejected", includeArchive: true }, proposalsCtx)
152
- .sort((a, b) => new Date(b.updatedAt ?? 0).getTime() - new Date(a.updatedAt ?? 0).getTime())
153
- .slice(0, MAX_REJECTED_PROPOSALS)
154
- .map((p) => ({
155
- ref: p.ref,
156
- reason: p.review?.reason ?? "no reason given",
157
- contentPreview: proposalContent(p).slice(0, 500),
158
- }));
159
- }
160
- /**
161
- * Synthesize a tmp draft-file path for the agent/sdk file-write contract.
162
- *
163
- * Mirrors the draft-path synthesis in `src/commands/proposal/propose.ts` —
164
- * when the runner is agent-CLI or the OpenCode SDK, we instruct the agent to
165
- * write the proposal body directly to this file instead of inlining it in
166
- * JSON on stdout. This bypasses two
167
- * known failure modes for long assets: (a) ARG_MAX truncation on prompt
168
- * round-trips through fenced JSON, and (b) embedded-JSON parser brittleness
169
- * on multi-KB bodies (e.g. the `knowledge/systems/KOKORO_USAGE_GUIDE` 8.4KB
170
- * payload that produced 4/5 `parse_error` in May 2026 reflect validation).
171
- *
172
- * The path lives under {@link os.tmpdir} and embeds the (sanitized) ref +
173
- * timestamp + random suffix so concurrent reflect calls cannot collide.
174
- *
175
- * The LLM HTTP runner cannot use this path because chat-completion transport
176
- * has no filesystem access.
87
+ * A fresh tmp path per iteration for the agent/SDK file-write contract (long
88
+ * bodies are written to a file instead of fenced JSON on stdout). The direct
89
+ * LLM runner has no filesystem and never gets one.
177
90
  */
178
91
  function synthesizeReflectDraftPath(ref) {
179
92
  const safeRef = (ref ?? "no-ref").replace(/[^a-z0-9_-]/gi, "_");
180
93
  const rand = Math.random().toString(36).slice(2, 8);
181
94
  return path.join(os.tmpdir(), `akm-reflect-${safeRef}-${Date.now()}-${rand}.md`);
182
95
  }
183
- /**
184
- * Heuristic check that the agent honoured the file-write contract.
185
- * The contract instructs the agent to emit a single `DRAFT_WRITTEN` line on
186
- * stdout when it has finished writing the draft file. Some agents print
187
- * additional log lines; we match anywhere in the captured stdout.
188
- */
189
- function stdoutSignalsDraftWritten(stdout) {
190
- if (!stdout)
191
- return false;
192
- return /\bDRAFT_WRITTEN\b/.test(stdout);
193
- }
194
- /**
195
- * Build schema/lint hints for the prompt. For lesson refs, run the lesson
196
- * lint over the current content and surface any findings — they are a
197
- * concrete starting point for the agent's revision.
198
- */
96
+ /** Lesson lint findings for the prompt: a concrete starting point for the revision. */
199
97
  function buildSchemaHints(type, content) {
200
- if (!content)
201
- return [];
202
- if (type !== "lesson")
98
+ if (!content || type !== "lesson")
203
99
  return [];
204
- const report = lintLessonContent(content, "reflect");
205
- return report.findings.map((f) => `[${f.kind}] ${f.message}`);
100
+ return lintLessonContent(content, "reflect").findings.map((f) => `[${f.kind}] ${f.message}`);
206
101
  }
207
- function hasRelatedSkillSource(content, skillRef) {
208
- const parsed = parseFrontmatter(content);
209
- const sources = parsed.data.sources;
210
- return Array.isArray(sources) && sources.some((source) => typeof source === "string" && source.trim() === skillRef);
211
- }
212
- async function readRelatedLessons(ctx, stash, ref, parsedRef, itemRef) {
102
+ /**
103
+ * Lessons related to a skill: its derived lesson, lessons distilled from it,
104
+ * and lessons citing it in `sources`. Without independent feedback on the skill,
105
+ * lessons reflect itself produced are dropped so its own output is not fed
106
+ * back as evidence.
107
+ */
108
+ async function readRelatedLessons(stash, ref, parsedRef, itemRef, eventsCtx) {
213
109
  if (parsedRef.type !== "skill")
214
110
  return [];
111
+ const cache = new Map();
112
+ const read = (filePath) => {
113
+ const key = path.resolve(filePath);
114
+ const cached = cache.get(key) ?? fs.readFileSync(filePath, "utf8");
115
+ cache.set(key, cached);
116
+ return cached;
117
+ };
215
118
  const related = new Map();
216
119
  const derivedLessonRef = deriveLessonRef(ref);
217
120
  const candidateRefs = new Set([derivedLessonRef]);
218
121
  const derivedLessonPath = path.join(stash, "lessons", `${parseRefInput(derivedLessonRef).name}.md`);
219
122
  if (fs.existsSync(derivedLessonPath)) {
220
- // WI-9.10: genuine content read — routed through the per-invocation asset
221
- // memo (D6). No write to this same path happens later in this invocation,
222
- // so memoizing is safe (see run-context.ts's D6 seam docblock).
223
- related.set(derivedLessonRef, { ref: derivedLessonRef, content: ctx.readAsset(derivedLessonPath) });
123
+ related.set(derivedLessonRef, { ref: derivedLessonRef, content: read(derivedLessonPath) });
224
124
  }
225
125
  try {
226
- // Match events using the candidate's single durable state key.
227
- const distillInvokedKeys = new Set(improveStateReadRefs(ref, itemRef));
228
- const feedbackEvents = readEvents({ type: "distill_invoked" }, readOnlyEventsContext(ctx.eventsCtx)).events.filter((event) => event.ref !== undefined && distillInvokedKeys.has(event.ref));
229
- for (const event of feedbackEvents) {
126
+ const keys = new Set([itemRef ?? ref]);
127
+ for (const event of readEvents({ type: "distill_invoked" }, readOnlyEventsContext(eventsCtx)).events) {
128
+ if (event.ref === undefined || !keys.has(event.ref))
129
+ continue;
230
130
  const proposalRef = typeof event.metadata?.proposalRef === "string" ? event.metadata.proposalRef : undefined;
231
131
  if (proposalRef && lenientRefType(proposalRef) === "lesson")
232
132
  candidateRefs.add(proposalRef);
233
133
  }
234
134
  }
235
135
  catch {
236
- // Best effort only.
136
+ // best-effort
237
137
  }
238
138
  for (const candidateRef of candidateRefs) {
239
139
  try {
240
- const filePath = await findAssetFilePath(durableImproveRef(candidateRef), stash);
241
- if (!filePath || !fs.existsSync(filePath))
242
- continue;
243
- const content = ctx.readAsset(filePath);
244
- related.set(candidateRef, { ref: candidateRef, content });
140
+ const filePath = await findAssetFilePath(candidateRef, stash);
141
+ if (filePath && fs.existsSync(filePath))
142
+ related.set(candidateRef, { ref: candidateRef, content: read(filePath) });
245
143
  }
246
144
  catch {
247
- // Index miss is non-fatal.
145
+ // An index miss is not fatal.
248
146
  }
249
147
  }
250
148
  try {
@@ -253,83 +151,40 @@ async function readRelatedLessons(ctx, stash, ref, parsedRef, itemRef) {
253
151
  for (const fileName of fs.readdirSync(lessonsDir)) {
254
152
  if (!fileName.endsWith(".md"))
255
153
  continue;
256
- const content = ctx.readAsset(path.join(lessonsDir, fileName));
257
- if (!hasRelatedSkillSource(content, ref))
154
+ const content = read(path.join(lessonsDir, fileName));
155
+ const sources = parseFrontmatter(content).data.sources;
156
+ if (!Array.isArray(sources) || !sources.some((s) => typeof s === "string" && s.trim() === ref))
258
157
  continue;
259
- const lessonName = fileName.slice(0, -3);
260
- const lessonRef = conceptIdFromTypeName("lesson", lessonName);
261
- if (!related.has(lessonRef)) {
158
+ const lessonRef = conceptIdFromTypeName("lesson", fileName.slice(0, -3));
159
+ if (!related.has(lessonRef))
262
160
  related.set(lessonRef, { ref: lessonRef, content });
263
- }
264
161
  }
265
162
  }
266
163
  }
267
164
  catch {
268
- // Best effort only.
165
+ // best-effort
269
166
  }
270
- // R-4 / #373: Filter out lessons with `derived_from_reflect: true` unless
271
- // independent feedback exists for the skill. This prevents the echo-chamber
272
- // risk where reflect-output lessons feed back into the next reflect pass as
273
- // "independent" evidence, amplifying their own prior outputs over time.
274
- //
275
- // ExpeL arXiv:2308.10144: rules need differential evidence from independent
276
- // sources (success vs failure traces). A lesson that only ever appeared from
277
- // reflect-internal signals has no such differential signal.
278
- //
279
- // "Independent feedback" = any usage_events "feedback" events for the skill
280
- // ref itself, indicating a human or external system rated the skill.
281
- let hasIndependentFeedback = false;
167
+ let hasIndependentFeedback = true;
282
168
  try {
283
- const feedbackEventsForSkill = readEvents({ type: "feedback", ref }, readOnlyEventsContext(ctx.eventsCtx)).events;
284
- hasIndependentFeedback = feedbackEventsForSkill.length > 0;
169
+ hasIndependentFeedback = readEvents({ type: "feedback", ref }, readOnlyEventsContext(eventsCtx)).events.length > 0;
285
170
  }
286
171
  catch {
287
- // Best effort — if we can't check, allow all lessons through.
288
- hasIndependentFeedback = true;
172
+ // Unknown: keep every lesson.
289
173
  }
290
174
  if (!hasIndependentFeedback) {
291
- // No independent feedback: exclude all reflect-derived lessons to prevent
292
- // echo-chamber amplification.
293
- for (const [lessonRef, lesson] of related.entries()) {
175
+ for (const [lessonRef, lesson] of related) {
294
176
  try {
295
- const lessonFm = parseFrontmatter(lesson.content);
296
- if (lessonFm.data.derived_from_reflect === true) {
177
+ if (parseFrontmatter(lesson.content).data.derived_from_reflect === true)
297
178
  related.delete(lessonRef);
298
- }
299
179
  }
300
180
  catch {
301
- // If we can't parse the frontmatter, keep the lesson (safe default).
181
+ // Unparseable frontmatter: keep it.
302
182
  }
303
183
  }
304
184
  }
305
185
  return [...related.values()];
306
186
  }
307
- /**
308
- * Returns true only when `stdout` is a recognised AKM proposal-skip signal.
309
- *
310
- * Accepted forms are structured JSON: `{ skipped: true }` or
311
- * `{ reason: "<known-skip-reason>" }`.
312
- */
313
- function isStructuredCooldownSignal(stdout) {
314
- try {
315
- const parsed = JSON.parse(stdout.trim());
316
- if (parsed?.skipped === true)
317
- return true;
318
- if (typeof parsed?.reason === "string" && ["fingerprint_match", "rejection_backoff"].includes(parsed.reason))
319
- return true;
320
- }
321
- catch {
322
- // Non-JSON stdout is never a structured cooldown signal.
323
- }
324
- return false;
325
- }
326
- /**
327
- * Best-effort asset type for a maybe-ref string, in the 0.9.0 `[bundle//]conceptId`
328
- * grammar (`""` when it does not parse). Replaces the pre-0.9.0 `ref.split(":")[0]`
329
- * type-extraction, which yielded the whole conceptId (`lessons/my-lesson`) instead
330
- * of the type once refs stopped carrying a `type:` prefix (ref-grammar decision
331
- * D-R3). Lenient by design — the callers degrade gracefully on an empty type.
332
- */
187
+ /** The asset type of a maybe-ref, or `""` when it does not parse. */
333
188
  function lenientRefType(ref) {
334
189
  if (!ref)
335
190
  return "";
@@ -341,81 +196,43 @@ function lenientRefType(ref) {
341
196
  }
342
197
  }
343
198
  /**
344
- * Split a markdown blob into `[frontmatterText, bodyText]`.
345
- *
346
- * Returns `[null, raw]` when the blob does not start with a frontmatter block.
347
- */
348
- function splitFrontmatter(raw) {
349
- const m = raw.match(/^---\r?\n([\s\S]*?)\r?\n---\r?\n?([\s\S]*)$/);
350
- if (!m)
351
- return { fmText: null, body: raw };
352
- return { fmText: m[1], body: m[2] };
353
- }
354
- /**
355
- * Strip an LLM-appended duplicate frontmatter block from a body string.
356
- *
357
- * When the LLM echoes the original source file verbatim after its rewrite,
358
- * the resulting body contains a second `---...---` YAML block. We detect it
359
- * by requiring BOTH a balanced fence (opening + closing `---`) AND YAML-like
360
- * `key: value` content inside, so legitimate Markdown thematic breaks and
361
- * code-fence examples are never truncated.
199
+ * Cut a duplicate frontmatter block the model appended after its rewrite.
200
+ * Requires a balanced fence AND `key:` lines so thematic breaks survive.
362
201
  */
363
202
  function stripAppendedFrontmatter(body) {
364
- const fencePattern = /\n---\r?\n([\s\S]*?)\n---\r?\n/;
365
- const match = body.match(fencePattern);
366
- if (!match)
367
- return body;
368
- // Only strip when the captured block looks like YAML frontmatter.
369
- if (!/^\w[\w-]*:/m.test(match[1]))
203
+ const match = body.match(/\n---\r?\n([\s\S]*?)\n---\r?\n/);
204
+ if (!match || !/^\w[\w-]*:/m.test(match[1]))
370
205
  return body;
371
206
  return body.slice(0, body.indexOf(match[0])).replace(/\s+$/, "");
372
207
  }
373
208
  /**
374
- * #636 — deterministically derive a valid `description` from an asset's existing
375
- * metadata when one is missing. Sources, in priority order: the `title:`
376
- * frontmatter field, the first `# Heading` in the (proposed or source) body, and
377
- * the first sentence of the opening body paragraph. The candidate is normalized
378
- * (whitespace collapsed, trailing punctuation/markdown stripped, clamped to the
379
- * description max) and only returned if it PASSES `isValidDescription` — so this
380
- * never produces a heading-fragment, truncated, or otherwise gate-failing value.
381
- * Returns `undefined` when nothing usable can be derived (caller leaves the
382
- * proposal as-is rather than fabricating prose).
383
- *
384
- * This is intentionally deterministic and lives in the reflect proposal-build
385
- * path — it does NOT touch the validators or the promote-time repair.
209
+ * A description derived from existing metadata (title, first heading, first
210
+ * prose sentence) that passes `isValidDescription`, or `undefined`. Never
211
+ * free-form invention.
386
212
  */
387
213
  function deriveDescriptionFromAsset(title, proposedBody, sourceBody, targetRef) {
388
- // Each candidate is tagged with its kind. A title or `# Heading` is a bare
389
- // fragment ("Paged.js — Named Page") that reads poorly as a description even
390
- // when it is long enough to pass the length gate, so for those we prefer the
391
- // padded sentence form. A prose sentence is already a sentence, so it is used
392
- // as-is (padding it would double-wrap an already-complete sentence).
393
214
  const candidates = [];
394
- // 1. title: frontmatter
395
215
  if (typeof title === "string" && title.trim())
396
216
  candidates.push({ text: title.trim(), kind: "fragment" });
397
- // 2. first `# Heading` (proposed body first, then source body)
398
217
  for (const body of [proposedBody, sourceBody]) {
399
- const headingMatch = body.match(/^#{1,6}\s+(.+?)\s*$/m);
400
- if (headingMatch?.[1])
401
- candidates.push({ text: headingMatch[1].trim(), kind: "fragment" });
218
+ const heading = body.match(/^#{1,6}\s+(.+?)\s*$/m)?.[1];
219
+ if (heading)
220
+ candidates.push({ text: heading.trim(), kind: "fragment" });
402
221
  }
403
- // 3. first sentence of the opening prose paragraph (skip headings, fences,
404
- // list markers, blockquotes — those are not prose).
405
222
  for (const body of [proposedBody, sourceBody]) {
406
- const firstSentence = firstProseSentence(body);
407
- if (firstSentence)
408
- candidates.push({ text: firstSentence, kind: "prose" });
223
+ const sentence = firstProseSentence(body);
224
+ if (sentence)
225
+ candidates.push({ text: sentence, kind: "prose" });
409
226
  }
410
227
  for (const { text, kind } of candidates) {
411
- const normalized = normalizeDescriptionCandidate(text);
228
+ const normalized = text
229
+ .replace(/`/g, "")
230
+ .replace(/^[#>*\-\s]+/, "")
231
+ .replace(/\s+/g, " ")
232
+ .trim();
412
233
  if (!normalized)
413
234
  continue;
414
- // For a title/heading fragment, try the padded sentence form FIRST so the
415
- // result reads as a sentence rather than a bare fragment — a short but valid
416
- // title like "Paged.js — Named Page" (21 chars) would otherwise be returned
417
- // verbatim. Fall back to the bare form only if the padded form fails the
418
- // gate. A prose candidate is already a sentence, so it is used as-is.
235
+ // A bare title/heading reads poorly as a description: prefer the sentence form.
419
236
  const variants = kind === "fragment" ? [`Reference notes on ${normalized}.`, normalized] : [normalized];
420
237
  for (const v of variants) {
421
238
  const clamped = v.length > DESCRIPTION_MAX_CHARS ? v.slice(0, DESCRIPTION_MAX_CHARS).trimEnd() : v;
@@ -425,51 +242,21 @@ function deriveDescriptionFromAsset(title, proposedBody, sourceBody, targetRef)
425
242
  }
426
243
  return undefined;
427
244
  }
428
- /** Extract the first prose sentence from a markdown body, or `""` if none. */
429
245
  function firstProseSentence(body) {
430
246
  for (const rawLine of body.split(/\r?\n/)) {
431
247
  const line = rawLine.trim();
432
- if (!line)
433
- continue;
434
- if (/^(#{1,6}\s|```|~~~|[-*+]\s|\d+\.\s|>|\||<!--)/.test(line))
248
+ if (!line || /^(#{1,6}\s|```|~~~|[-*+]\s|\d+\.\s|>|\||<!--)/.test(line))
435
249
  continue;
436
- const sentenceMatch = line.match(/^(.+?[.!?])(\s|$)/);
437
- return (sentenceMatch?.[1] ?? line).trim();
250
+ return (line.match(/^(.+?[.!?])(\s|$)/)?.[1] ?? line).trim();
438
251
  }
439
252
  return "";
440
253
  }
441
- /** Normalize a description candidate: strip markdown markers, collapse space. */
442
- function normalizeDescriptionCandidate(raw) {
443
- return raw
444
- .replace(/`/g, "")
445
- .replace(/^[#>*\-\s]+/, "")
446
- .replace(/\s+/g, " ")
447
- .trim();
448
- }
449
254
  /**
450
- * Reflect post-processor — enforces the safety rails described at the top of
451
- * this file:
452
- *
453
- * 1. Restore the source frontmatter so reflect never strips load-bearing
454
- * `description`, `when_to_use`, `tags`, etc. The LLM is only allowed to
455
- * change the markdown body. Frontmatter fields proposed by the LLM are
456
- * treated as a *merge on top* of the source — concrete field renames /
457
- * identity changes (`name`, `ref`, `id`, `slug`, `type`) are reverted.
458
- * 2. Reject responses that shrink or expand the body past the configured
459
- * ratio thresholds, when the source body is large enough to be reliable.
460
- * 3. Drop any leading `---` frontmatter block the LLM produced inside the
461
- * body — the prompt asks it to emit body only, and a stray YAML preamble
462
- * on top of an executable-typed asset is dangerous.
463
- *
464
- * Caller branches:
465
- * - On `reject`: surface as a failure with the reported reason.
466
- * - Otherwise: substitute `content` (and optional `frontmatter`) into the
467
- * proposal payload.
468
- *
469
- * Source-less / new-asset case (`sourceContent === undefined`): we still strip
470
- * the LLM's frontmatter block from `content` and re-emit a clean block built
471
- * from `payload.frontmatter` so identity fields can be enforced. Size guard
472
- * is skipped because there is no source to compare against.
255
+ * Reflect's content rails: the source frontmatter is restored and the model's
256
+ * frontmatter merged on top except identity fields; a stray or appended
257
+ * frontmatter block and echoed run-only guidance are stripped; a missing
258
+ * required description is derived deterministically; a body outside the size
259
+ * ratios or echoing the truncation notice is flagged for review.
473
260
  */
474
261
  export function sanitizeReflectPayload(payload, sourceContent, targetRef) {
475
262
  const warnings = [];
@@ -478,13 +265,9 @@ export function sanitizeReflectPayload(payload, sourceContent, targetRef) {
478
265
  : { fmText: null, body: "" };
479
266
  const sourceFm = sourceFmText !== null ? parseFrontmatter(sourceContent ?? "").data : {};
480
267
  const { fmText: llmFmText, body: rawLlmBody } = splitFrontmatter(payload.content);
481
- if (llmFmText !== null) {
482
- warnings.push("LLM emitted frontmatter in content; stripped and merged through identity guard.");
483
- }
484
- // Parse the LLM-emitted frontmatter (if any) so we can merge its non-identity
485
- // keys into the source frontmatter.
486
268
  let llmFm = {};
487
269
  if (llmFmText !== null) {
270
+ warnings.push("LLM emitted frontmatter in content; stripped and merged through identity guard.");
488
271
  try {
489
272
  llmFm = parseFrontmatter(payload.content).data;
490
273
  }
@@ -492,107 +275,60 @@ export function sanitizeReflectPayload(payload, sourceContent, targetRef) {
492
275
  llmFm = {};
493
276
  }
494
277
  }
495
- // Also accept the explicit `frontmatter` field on the payload.
496
- if (payload.frontmatter && typeof payload.frontmatter === "object") {
278
+ if (payload.frontmatter && typeof payload.frontmatter === "object")
497
279
  llmFm = { ...llmFm, ...payload.frontmatter };
498
- }
499
- // Strip protected identity fields from any LLM-supplied frontmatter — they
500
- // must come from the source asset, never from the LLM.
501
280
  for (const field of PROTECTED_FRONTMATTER_FIELDS) {
502
281
  if (field in llmFm && llmFm[field] !== sourceFm[field]) {
503
282
  warnings.push(`LLM attempted to change protected frontmatter field "${field}"; restored from source.`);
504
283
  delete llmFm[field];
505
284
  }
506
285
  }
507
- // Build the effective frontmatter: source overlaid with sanitized LLM fields.
508
- // Source fields always win on identity keys.
509
286
  const mergedFm = { ...sourceFm, ...llmFm };
510
- for (const field of PROTECTED_FRONTMATTER_FIELDS) {
511
- if (field in sourceFm) {
287
+ for (const field of PROTECTED_FRONTMATTER_FIELDS)
288
+ if (field in sourceFm)
512
289
  mergedFm[field] = sourceFm[field];
513
- }
514
- }
515
- const withoutAppendedFrontmatter = stripAppendedFrontmatter(rawLlmBody.replace(/^\s+/, ""));
516
- const promptScaffolding = stripReflectPromptScaffolding(withoutAppendedFrontmatter);
517
- const cleanedBody = promptScaffolding.content;
518
- if (promptScaffolding.stripped) {
290
+ const scaffolding = stripReflectPromptScaffolding(stripAppendedFrontmatter(rawLlmBody.replace(/^\s+/, "")));
291
+ const cleanedBody = scaffolding.content;
292
+ if (scaffolding.stripped) {
519
293
  warnings.push('Removed echoed run-only "Avoid These Patterns" guidance from the proposed asset body (#963).');
520
294
  }
521
- // #636 — deterministic description fallback (reflect-side belt-and-suspenders).
522
- // If the type requires a `description` and the merged frontmatter is still
523
- // MISSING one (source had none AND the model didn't author one), derive a
524
- // description DETERMINISTICALLY from the existing `title:` frontmatter or the
525
- // first `# Heading` / opening body sentence — never free-form invention. This
526
- // runs in the reflect proposal-build path, BEFORE the proposal is created, so
527
- // the validator/promote path is left untouched (no gate fabricates content).
528
- //
529
- // Scope is the issue's target: a source asset that ALREADY carries frontmatter
530
- // (e.g. scraped docs: `source`/`title`/`scraped`) but has a MISSING/empty
531
- // `description`. We deliberately do NOT fire when:
532
- // - the source has no frontmatter block at all (injecting one would be a
533
- // structural change and would defeat the #580 no-op/cosmetic noise gate
534
- // for a pure body echo), or
535
- // - a present-but-otherwise-invalid description exists (too short, a heading
536
- // fragment) — overwriting authored content is out of scope; the prompt
537
- // instruction handles improving it instead.
295
+ // Only a source that already has frontmatter but no description gets one:
296
+ // injecting a whole block, or overwriting an authored one, is out of scope.
538
297
  const refType = lenientRefType(targetRef);
539
- const mergedDesc = mergedFm.description;
540
- const descIsMissing = typeof mergedDesc !== "string" || mergedDesc.trim().length === 0;
298
+ const desc = mergedFm.description;
541
299
  const sourceHadFrontmatter = sourceFmText !== null && Object.keys(sourceFm).length > 0;
542
- if (refType && requiresDescription(refType) && descIsMissing && sourceHadFrontmatter) {
300
+ if (refType &&
301
+ requiresDescription(refType) &&
302
+ (typeof desc !== "string" || desc.trim().length === 0) &&
303
+ sourceHadFrontmatter) {
543
304
  const derived = deriveDescriptionFromAsset(mergedFm.title, cleanedBody, sourceBody, targetRef);
544
305
  if (derived) {
545
306
  mergedFm.description = derived;
546
307
  warnings.push("Synthesized a deterministic `description` from title/heading (#636) — source and proposal lacked one.");
547
308
  }
548
309
  }
549
- // Size guard — only when source body is meaningfully large. The pure
550
- // predicate lives in `core/proposal-quality-validators` so the same check
551
- // also runs inside `runProposalValidators` on `proposal accept`.
552
- const sizeOutcome = checkReflectSize(sourceBody, cleanedBody);
310
+ const size = checkReflectSize(sourceBody, cleanedBody);
553
311
  let sizeGuardRatio;
554
- if (!sizeOutcome.ok) {
555
- const pct = (sizeOutcome.ratio * 100).toFixed(0);
556
- const limit = sizeOutcome.code === "EXCESSIVE_SHRINKAGE" ? "minimum 50%" : "maximum 250%";
557
- const cause = sizeOutcome.code === "EXCESSIVE_SHRINKAGE"
558
- ? "Concrete content was likely deleted."
559
- : "Speculative material was likely added.";
560
- warnings.push(`${sizeOutcome.code} — proposed body is ${pct}% of source (${limit}) for ref ${targetRef}. ${cause} Flagged for review.`);
561
- sizeGuardRatio = { code: sizeOutcome.code, ratio: sizeOutcome.ratio };
312
+ if (!size.ok) {
313
+ const shrink = size.code === "EXCESSIVE_SHRINKAGE";
314
+ warnings.push(`${size.code} — proposed body is ${(size.ratio * 100).toFixed(0)}% of source (${shrink ? "minimum 50%" : "maximum 250%"}) for ref ${targetRef}. ${shrink ? "Concrete content was likely deleted." : "Speculative material was likely added."} Flagged for review.`);
315
+ sizeGuardRatio = { code: size.code, ratio: size.ratio };
562
316
  }
563
- // Truncation-marker leak (#952) — a model that saw a capped/truncated
564
- // asset sometimes echoes the "[truncated ...]" notice verbatim into its
565
- // rewrite instead of proposing real content for the missing tail. The
566
- // body-length ratio check above does not reliably catch this (a leaked
567
- // marker can still fall inside the 50%-250% band). Flag and defer to
568
- // human review — same "degrade with a warning" rung as the size guard,
569
- // not a new hard reject.
570
317
  const truncationMarkerLeaked = cleanedBody.includes(REFLECT_TRUNCATION_MARKER);
571
318
  if (truncationMarkerLeaked) {
572
319
  warnings.push(`Proposed body for ref ${targetRef} contains the truncation-notice text the model was shown for a capped source asset ("${REFLECT_TRUNCATION_MARKER}"). The model likely echoed the notice instead of writing real content. Flagged for review.`);
573
320
  }
574
- // Reassemble final content: merged frontmatter + cleaned body.
575
- // When there is no frontmatter at all (no source fm and no LLM fm), emit body
576
- // only so we don't add a stray `---` to e.g. a script asset that bypassed the
577
- // type guard via a custom registration.
321
+ // No frontmatter at all stays body-only, never gaining a stray `---`.
578
322
  const hasFrontmatter = Object.keys(mergedFm).length > 0;
579
- const reassembled = hasFrontmatter
580
- ? assembleAssetFromString(serializeFrontmatter(mergedFm), cleanedBody)
581
- : cleanedBody;
582
323
  return {
583
- content: reassembled,
324
+ content: hasFrontmatter ? assembleAssetFromString(serializeFrontmatter(mergedFm), cleanedBody) : cleanedBody,
584
325
  ...(hasFrontmatter ? { frontmatter: mergedFm } : {}),
585
326
  warnings,
586
327
  ...(sizeGuardRatio ? { sizeGuardRatio } : {}),
587
328
  ...(truncationMarkerLeaked ? { truncationMarkerLeaked } : {}),
588
329
  };
589
330
  }
590
- /**
591
- * JSON Schema for structured reflect output. Passed to `chatCompletion` when
592
- * {@link wantsJsonSchemaOutput} selects `outputMode: "json_schema"`, so the
593
- * model returns a strict JSON object containing only the target-scoped
594
- * fields AKM cannot derive.
595
- */
331
+ // ── Direct-LLM output contract ───────────────────────────────────────────────
596
332
  const REFLECT_FRONTMATTER_PATCH_JSON_SCHEMA = {
597
333
  type: "object",
598
334
  required: ["description", "when_to_use"],
@@ -602,23 +338,19 @@ const REFLECT_FRONTMATTER_PATCH_JSON_SCHEMA = {
602
338
  when_to_use: { type: ["string", "null"] },
603
339
  },
604
340
  };
341
+ const REFLECT_CONFIDENCE_SCHEMA = {
342
+ type: "number",
343
+ minimum: 0,
344
+ maximum: 1,
345
+ description: "Self-reported quality confidence in [0, 1]. Persisted on the proposal for reviewers and the triage judge to read during adjudication.",
346
+ };
605
347
  export const REFLECT_JSON_SCHEMA = {
606
348
  type: "object",
607
349
  required: ["content", "confidence", "frontmatterPatch"],
608
350
  additionalProperties: false,
609
351
  properties: {
610
352
  content: { type: "string", description: "Complete improved markdown body without YAML frontmatter." },
611
- // Phase 6A (Advantage D6a): self-reported confidence in [0, 1]. When the
612
- // LLM is well-calibrated, scores at or above the configured threshold
613
- // (default 0.8) drive auto-accept in `akm improve`. Out-of-range or
614
- // non-finite values are rejected by direct-output extraction. Agent and SDK
615
- // confidence remains optional on their separate existing contracts.
616
- confidence: {
617
- type: "number",
618
- minimum: 0,
619
- maximum: 1,
620
- description: "Self-reported quality confidence in [0, 1]. Persisted on the proposal for reviewers and the triage judge to read during adjudication.",
621
- },
353
+ confidence: REFLECT_CONFIDENCE_SCHEMA,
622
354
  frontmatterPatch: REFLECT_FRONTMATTER_PATCH_JSON_SCHEMA,
623
355
  },
624
356
  };
@@ -629,44 +361,32 @@ const REFLECT_UNSCOPED_JSON_SCHEMA = {
629
361
  properties: {
630
362
  ref: { type: "string", description: "Selected asset ref as a subdir-qualified conceptId." },
631
363
  content: { type: "string", description: "Complete improved markdown body without YAML frontmatter." },
632
- confidence: {
633
- type: "number",
634
- minimum: 0,
635
- maximum: 1,
636
- description: "Self-reported quality confidence in [0, 1].",
637
- },
364
+ confidence: { type: "number", minimum: 0, maximum: 1, description: "Self-reported quality confidence in [0, 1]." },
638
365
  frontmatterPatch: REFLECT_FRONTMATTER_PATCH_JSON_SCHEMA,
639
366
  },
640
367
  };
641
368
  /**
642
- * Whether to frame the reflect prompt for structured JSON output on this
643
- * connection. Optimistic by default — `chatCompletion` attempts
644
- * `response_format: json_schema` fresh on every call and falls back once on
645
- * a 4xx, so there is no persisted verdict to consult here. `false` only when
646
- * a human/workflow explicitly disabled it, or a real call already proved
647
- * this connection rejects it earlier in the same process.
369
+ * Frame for JSON Schema unless the connection disabled it or already proved
370
+ * this process that it rejects it (the transport retries plain text on a 4xx).
648
371
  */
649
372
  function wantsJsonSchemaOutput(connection) {
650
373
  return connection.supportsJsonSchema !== false && !isJsonSchemaKnownUnsupported(connection);
651
374
  }
652
- /** Critique prompt injected between prior draft and refinement request (Self-Refine loop). */
375
+ /** Injected between the prior draft and the refinement request (self-refine). */
653
376
  const REFLECT_CRITIQUE_PROMPT = "Your previous proposal is shown above. Review it critically and provide an improved version that is more specific, actionable, and avoids any issues with the previous attempt. Return only the improved response using the output contract from the original prompt.";
377
+ function parsedRecord(result) {
378
+ return result.parsed && typeof result.parsed === "object" && !Array.isArray(result.parsed)
379
+ ? result.parsed
380
+ : undefined;
381
+ }
654
382
  function reflectLlmTelemetry(result) {
655
- if (!result.parsed || typeof result.parsed !== "object" || Array.isArray(result.parsed))
656
- return undefined;
657
- const parsed = result.parsed;
658
- if (parsed.outputMode !== "json_schema" && parsed.outputMode !== "framed_markdown")
383
+ const parsed = parsedRecord(result);
384
+ if (!parsed || (parsed.outputMode !== "json_schema" && parsed.outputMode !== "framed_markdown"))
659
385
  return undefined;
660
386
  if (typeof parsed.repairAttempts !== "number")
661
387
  return undefined;
662
388
  return { outputMode: parsed.outputMode, repairAttempts: parsed.repairAttempts };
663
389
  }
664
- function reflectLlmPriorDraft(result) {
665
- if (!result.parsed || typeof result.parsed !== "object" || Array.isArray(result.parsed))
666
- return undefined;
667
- const priorDraft = result.parsed.priorDraft;
668
- return typeof priorDraft === "string" ? priorDraft : undefined;
669
- }
670
390
  function parseReflectConfidence(value) {
671
391
  if (typeof value !== "number" || !Number.isFinite(value) || value < 0 || value > 1) {
672
392
  throw new Error('direct reflect response missing required number field "confidence" in [0, 1]');
@@ -734,88 +454,57 @@ function parseFramedReflectOutput(raw, targetRef) {
734
454
  throw new Error("direct reflect response missing terminal AKM_REFLECT_CONTENT_END marker");
735
455
  }
736
456
  const headerLines = normalized.slice(0, beginIndex).trim().split("\n").filter(Boolean);
737
- const confidenceLine = headerLines.find((line) => line.startsWith("AKM_REFLECT_CONFIDENCE:"));
738
- const refLine = headerLines.find((line) => line.startsWith("AKM_REFLECT_REF:"));
739
- const patchLine = headerLines.find((line) => line.startsWith("AKM_REFLECT_FRONTMATTER_PATCH:"));
740
- const expectedHeaderLines = targetRef ? 2 : 3;
457
+ const header = (prefix) => headerLines.find((line) => line.startsWith(prefix));
458
+ const confidenceLine = header("AKM_REFLECT_CONFIDENCE:");
459
+ const refLine = header("AKM_REFLECT_REF:");
460
+ const patchLine = header("AKM_REFLECT_FRONTMATTER_PATCH:");
741
461
  const invalidRefLine = targetRef ? refLine !== undefined : refLine === undefined;
742
- if (headerLines.length !== expectedHeaderLines || !confidenceLine || !patchLine || invalidRefLine) {
462
+ if (headerLines.length !== (targetRef ? 2 : 3) || !confidenceLine || !patchLine || invalidRefLine) {
743
463
  throw new Error("direct reflect response contained invalid frame metadata");
744
464
  }
745
465
  const confidenceText = confidenceLine.slice("AKM_REFLECT_CONFIDENCE:".length).trim();
746
466
  if (!/^(?:0(?:\.\d+)?|1(?:\.0+)?)$/.test(confidenceText)) {
747
467
  throw new Error("direct reflect frame confidence must be a decimal number in [0, 1]");
748
468
  }
749
- const confidence = parseReflectConfidence(Number(confidenceText));
750
469
  const ref = targetRef ?? refLine?.slice("AKM_REFLECT_REF:".length).trim() ?? "";
751
470
  if (!ref)
752
471
  throw new Error("direct reflect response contained an empty AKM_REFLECT_REF value");
753
472
  const content = normalized.slice(contentStart, endIndex);
754
473
  if (!content.trim())
755
474
  throw new Error("direct reflect response contained empty framed content");
756
- const patchText = patchLine.slice("AKM_REFLECT_FRONTMATTER_PATCH:".length).trim();
757
475
  let parsedPatch;
758
476
  try {
759
- parsedPatch = JSON.parse(patchText);
477
+ parsedPatch = JSON.parse(patchLine.slice("AKM_REFLECT_FRONTMATTER_PATCH:".length).trim());
760
478
  }
761
479
  catch {
762
480
  throw new Error("direct reflect response contained invalid frontmatter patch JSON");
763
481
  }
764
482
  const frontmatter = parseReflectFrontmatterPatch(parsedPatch);
483
+ const confidence = parseReflectConfidence(Number(confidenceText));
765
484
  return { ref, content, confidence, ...(frontmatter ? { frontmatter } : {}) };
766
485
  }
767
- function parseDirectReflectOutput(raw, mode, targetRef) {
768
- return mode === "json_schema" ? parseSchemaReflectOutput(raw, targetRef) : parseFramedReflectOutput(raw, targetRef);
769
- }
770
486
  /**
771
- * Run a single reflect iteration directly via the LLM API (v2 config path).
772
- *
773
- * Returns an {@link AgentRunResult}-shaped object so it can slot into the same
774
- * dispatch loop as agent-based runners. Production calls extract the selected
775
- * direct-LLM contract and normalize it to proposal JSON in `stdout`. Errors
776
- * are captured into the result rather than thrown.
487
+ * One reflect iteration through the direct LLM runner, as an agent-shaped
488
+ * result (errors captured, never thrown except configuration). An unparseable
489
+ * response gets one repair turn within the original deadline.
777
490
  */
778
491
  export async function runReflectViaLlm(opts) {
779
492
  const start = Date.now();
780
493
  let repairAttempts = 0;
781
- const _connection = opts.runner.connection;
782
- const messages = [{ role: "user", content: opts.prompt ?? "" }];
783
494
  const configuredTimeout = Object.hasOwn(opts, "timeoutMs")
784
495
  ? (opts.timeoutMs ?? null)
785
496
  : Object.hasOwn(opts.runner, "timeoutMs")
786
497
  ? (opts.runner.timeoutMs ?? null)
787
498
  : DEFAULT_LLM_TIMEOUT_MS;
788
499
  const deadline = typeof configuredTimeout === "number" ? start + configuredTimeout : undefined;
500
+ const messages = [{ role: "user", content: opts.prompt ?? "" }];
789
501
  if (opts.priorDraft !== undefined && opts.iteration > 0) {
790
- messages.push({ role: "assistant", content: opts.priorDraft });
791
- messages.push({ role: "user", content: REFLECT_CRITIQUE_PROMPT });
502
+ messages.push({ role: "assistant", content: opts.priorDraft }, { role: "user", content: REFLECT_CRITIQUE_PROMPT });
792
503
  }
793
- const call = async (callMessages, repairTimeoutMs) => callStructured({
794
- feature: "reflect_proposal",
795
- runner: opts.runner,
796
- ...(opts.lease ? { lease: opts.lease } : {}),
797
- messages: callMessages,
798
- request: {
799
- ...(repairTimeoutMs !== undefined
800
- ? { timeoutMs: repairTimeoutMs }
801
- : Object.hasOwn(opts, "timeoutMs")
802
- ? { timeoutMs: opts.timeoutMs }
803
- : {}),
804
- ...(opts.signal ? { signal: opts.signal } : {}),
805
- ...(opts.responseSchema !== undefined ? { responseSchema: opts.responseSchema } : {}),
806
- ...(opts.maxTokens !== undefined ? { maxTokens: opts.maxTokens } : {}),
807
- // Reflect requires a machine-readable payload. Visible chain-of-thought
808
- // can consume the output cap before the model reaches the envelope.
809
- enableThinking: false,
810
- ...(opts.chat ? { chat: opts.chat } : {}),
811
- },
812
- ...(opts.onNotices ? { onNotices: opts.onNotices } : {}),
813
- parse: (raw) => raw ?? "",
814
- // Unreachable on the ungated path (errors propagate to the catch below).
815
- onError: () => "",
816
- fallback: "",
817
- });
818
- const failure = (err, reason, repairAttempts, stdout = "", exitCode = 1) => {
504
+ const parse = (raw) => opts.outputMode === "json_schema"
505
+ ? parseSchemaReflectOutput(raw, opts.targetRef)
506
+ : parseFramedReflectOutput(raw, opts.targetRef);
507
+ const failure = (err, reason, stdout = "", exitCode = 1) => {
819
508
  const msg = err instanceof Error ? err.message : String(err);
820
509
  return {
821
510
  ok: false,
@@ -828,6 +517,34 @@ export async function runReflectViaLlm(opts) {
828
517
  parsed: { outputMode: opts.outputMode, repairAttempts },
829
518
  };
830
519
  };
520
+ const call = async (callMessages, repairTimeoutMs) => {
521
+ const outcome = await callStage({
522
+ feature: "reflect_proposal",
523
+ runner: opts.runner,
524
+ prompt: callMessages.at(-1)?.content ?? "",
525
+ history: callMessages.slice(0, -1),
526
+ request: {
527
+ ...(repairTimeoutMs !== undefined
528
+ ? { timeoutMs: repairTimeoutMs }
529
+ : Object.hasOwn(opts, "timeoutMs")
530
+ ? { timeoutMs: opts.timeoutMs }
531
+ : {}),
532
+ ...(opts.signal ? { signal: opts.signal } : {}),
533
+ ...(opts.responseSchema !== undefined ? { responseSchema: opts.responseSchema } : {}),
534
+ ...(opts.maxTokens !== undefined ? { maxTokens: opts.maxTokens } : {}),
535
+ // Visible chain-of-thought can exhaust the output before the envelope.
536
+ enableThinking: false,
537
+ ...(opts.chat ? { chat: opts.chat } : {}),
538
+ },
539
+ ...(opts.onNotices ? { onNotices: opts.onNotices } : {}),
540
+ });
541
+ if (!outcome.ok) {
542
+ throw outcome.reason === "timeout"
543
+ ? new LlmCallError(outcome.error ?? "timeout", "timeout")
544
+ : new Error(outcome.error ?? "LLM call failed");
545
+ }
546
+ return outcome.raw;
547
+ };
831
548
  try {
832
549
  if (opts.signal?.aborted)
833
550
  throw new Error("Reflect request aborted");
@@ -835,33 +552,25 @@ export async function runReflectViaLlm(opts) {
835
552
  let payload;
836
553
  let acceptedOutput = stdout;
837
554
  try {
838
- payload = parseDirectReflectOutput(stdout, opts.outputMode, opts.targetRef);
555
+ payload = parse(stdout);
839
556
  }
840
557
  catch (err) {
841
558
  if (opts.allowRepair === false)
842
- return failure(err, "parse_error", 0, stdout, 0);
559
+ return failure(err, "parse_error", stdout, 0);
843
560
  if (opts.signal?.aborted)
844
- return failure(new Error("Reflect request aborted"), "aborted", 0, stdout);
561
+ return failure(new Error("Reflect request aborted"), "aborted", stdout);
845
562
  const remaining = deadline === undefined ? undefined : deadline - Date.now();
846
563
  if (remaining !== undefined && remaining <= 0) {
847
- return failure(new LlmCallError("Reflect request timed out before output repair", "timeout"), "timeout", 0, stdout);
564
+ return failure(new LlmCallError("Reflect request timed out before output repair", "timeout"), "timeout", stdout);
848
565
  }
849
566
  repairAttempts = 1;
850
- const repairMessages = [
851
- ...messages,
852
- { role: "assistant", content: stdout },
853
- {
854
- role: "user",
855
- content: buildReflectOutputRepairPrompt(opts.outputMode, opts.targetRef !== undefined),
856
- },
857
- ];
858
- const repaired = await call(repairMessages, remaining);
859
- acceptedOutput = repaired;
567
+ const repairPrompt = buildReflectOutputRepairPrompt(opts.outputMode, opts.targetRef !== undefined);
568
+ acceptedOutput = await call([...messages, { role: "assistant", content: stdout }, { role: "user", content: repairPrompt }], remaining);
860
569
  try {
861
- payload = parseDirectReflectOutput(repaired, opts.outputMode, opts.targetRef);
570
+ payload = parse(acceptedOutput);
862
571
  }
863
- catch (err) {
864
- return failure(err, "parse_error", repairAttempts, repaired, 0);
572
+ catch (repairErr) {
573
+ return failure(repairErr, "parse_error", acceptedOutput, 0);
865
574
  }
866
575
  }
867
576
  return {
@@ -881,397 +590,114 @@ export async function runReflectViaLlm(opts) {
881
590
  : err instanceof LlmCallError && err.code === "timeout"
882
591
  ? "timeout"
883
592
  : "non_zero_exit";
884
- return failure(err, reason, repairAttempts);
593
+ return failure(err, reason);
885
594
  }
886
595
  }
887
- function failureEnvelope(result, ref, engine, fallbackReason = "non_zero_exit") {
888
- return {
889
- ...baseFailureFields(result, fallbackReason),
890
- schemaVersion: 2,
891
- ...(ref ? { ref } : {}),
892
- ...(engine ? { engine } : {}),
893
- };
894
- }
895
- /**
896
- * Reflect content-preservation + proposal creation: restore/reset protected
897
- * frontmatter and reject unsafe body-size ratios (sanitizeReflectPayload), the
898
- * #580 noise gate, the optional quality judge, then create the proposal (with
899
- * the R-4/#373 lesson provenance stamp) and emit `reflect_completed`. Extracted
900
- * verbatim from `akmReflect`; every reject/skip envelope and event is
901
- * byte-identical.
902
- */
903
- async function finalizeReflectProposal(args) {
904
- const { assetContent, result, options, engineName, config, qualityGateEnabled, qualityGateSkippedNoJudge, qualityJudgeRunner, qualityJudgeLease, feedback, stash, emitReflectFailed, onNotices, } = args;
905
- let payload = args.payload;
906
- const outputTelemetry = reflectLlmTelemetry(result);
907
- // 7. Reflect content-preservation rails:
908
- // - Restore source frontmatter so reflect can never strip indexable
909
- // fields (`description`, `when_to_use`, `tags`, ...).
910
- // - Reset protected identity fields (`name`, `ref`, `id`, `slug`,
911
- // `type`) the LLM tried to change.
912
- // - Reject proposals that shrink/expand the body past safe ratios.
913
- //
914
- // See REFLECT_ALLOWED_TYPES / sanitizeReflectPayload for the underlying
915
- // hypotheses + observed regressions (`8737ab63`, `26941510`, and the
916
- // catastrophic-shrinkage cases from the May 2026 review).
917
- const sanitizeOutcome = sanitizeReflectPayload({ content: payload.content, ...(payload.frontmatter ? { frontmatter: payload.frontmatter } : {}) }, assetContent, payload.ref);
918
- if (sanitizeOutcome.reject) {
596
+ /** The lazy `reflect_invoked` + failure-side `reflect_completed` emitters. */
597
+ function reflectEmitters(options) {
598
+ let invoked = false;
599
+ const emitInvoked = () => {
600
+ if (invoked)
601
+ return;
919
602
  appendEvent({
920
- eventType: "reflect_completed",
921
- ref: payload.ref,
603
+ eventType: "reflect_invoked",
604
+ ...(options.ref ? { ref: options.itemRef ?? options.ref } : {}),
922
605
  metadata: {
923
- source: "reflect",
924
- sanitized: true,
925
- rejected: true,
926
- rejectReason: sanitizeOutcome.reject.error,
927
- ...(sanitizeOutcome.warnings.length > 0 ? { sanitizerWarnings: sanitizeOutcome.warnings } : {}),
928
- ...(outputTelemetry ?? {}),
606
+ ...(options.task ? { task: options.task } : {}),
607
+ ...(options.engine ? { engine: options.engine } : {}),
608
+ ...(options.eligibilitySource ? { eligibilitySource: options.eligibilitySource } : {}),
929
609
  },
930
610
  }, options.eventsCtx);
931
- return {
932
- schemaVersion: 2,
933
- ok: false,
934
- reason: sanitizeOutcome.reject.reason,
935
- error: sanitizeOutcome.reject.error,
936
- ...(options.ref ? { ref: options.ref } : {}),
937
- engine: engineName,
938
- exitCode: result.exitCode,
939
- };
940
- }
941
- payload = {
942
- ...payload,
943
- content: sanitizeOutcome.content,
944
- ...(sanitizeOutcome.frontmatter ? { frontmatter: sanitizeOutcome.frontmatter } : {}),
611
+ invoked = true;
945
612
  };
946
- // 7c. Noise gate (#580): never queue a proposal whose sanitized content is
947
- // identical to the current asset (empty diff) or differs only cosmetically
948
- // (whitespace reflow, code-fence language hints, YAML scalar re-folding).
949
- // Pure deterministic text comparison — see `reflect-noise.ts`. Skipped when
950
- // there is no source asset (new-asset proposals have nothing to diff against).
951
- if (assetContent !== undefined) {
952
- const changeKind = classifyReflectChange(assetContent, payload.content);
953
- // 'low-value' is config-gated (#639). DEFAULT OFF — absent = byte-identical
954
- // pre-#639 behaviour (low-value treated the same as substantive). Resolved
955
- // by the caller from the active improve strategy's
956
- // `processes.reflect.lowValueFilter.enabled` and passed via options, so the
957
- // running strategy decides.
958
- const lowValueFilterEnabled = options.lowValueFilter === true;
959
- const isDeferred = changeKind === "noop" || changeKind === "cosmetic" || (changeKind === "low-value" && lowValueFilterEnabled);
960
- if (isDeferred) {
961
- const subreason = changeKind === "noop"
962
- ? "reflect_skipped_noop"
963
- : changeKind === "low-value"
964
- ? "reflect_skipped_low_value"
965
- : "reflect_skipped_cosmetic";
966
- emitReflectFailed("no_change", subreason, options.ref, { changeKind, ...(outputTelemetry ?? {}) });
967
- return {
968
- schemaVersion: 2,
969
- ok: false,
970
- reason: "no_change",
971
- error: changeKind === "noop"
972
- ? `Reflect skipped: proposed content for ${payload.ref} is identical to the current asset (empty diff); no proposal created.`
973
- : changeKind === "low-value"
974
- ? `Reflect skipped: proposed content for ${payload.ref} is a low-value prose micro-rewrite (few changed tokens, no structural changes); no proposal created.`
975
- : `Reflect skipped: proposed content for ${payload.ref} is a cosmetic-only reformat of the current asset (whitespace/fence/YAML-folding changes); no proposal created.`,
976
- ...(options.ref ? { ref: options.ref } : {}),
977
- engine: engineName,
978
- exitCode: result.exitCode,
979
- };
980
- }
981
- }
982
- // 7c. Judge the exact sanitized content that can be persisted. Fail closed
983
- // on cancellation, transport failure, malformed output, or an invalid score.
984
- // Skipped when the size guard or the truncation-marker leak already fired —
985
- // that content is deferred to human review regardless of what the judge says.
986
- if (qualityGateEnabled && !sanitizeOutcome.sizeGuardRatio && !sanitizeOutcome.truncationMarkerLeaked) {
987
- const judgeResult = await runReflectQualityJudge(config, payload.content, assetContent ?? "", feedback, options.chat, {
988
- runnerSelectionFrozen: true,
989
- ...(qualityJudgeRunner ? { llmRunner: qualityJudgeRunner } : {}),
990
- ...(qualityJudgeLease ? { lease: qualityJudgeLease } : {}),
991
- ...(Object.hasOwn(options, "timeoutMs") ? { timeoutMs: options.timeoutMs } : {}),
992
- ...(options.signal ? { signal: options.signal } : {}),
993
- onNotices,
994
- });
995
- if (!judgeResult.pass) {
996
- appendEvent({
997
- eventType: "reflect_completed",
998
- ref: payload.ref,
999
- metadata: {
1000
- source: "reflect",
1001
- qualityRejected: true,
1002
- qualityScore: judgeResult.score,
1003
- qualityReason: judgeResult.reason,
1004
- ...(outputTelemetry ?? {}),
1005
- },
1006
- }, options.eventsCtx);
1007
- return {
1008
- schemaVersion: 2,
1009
- ok: false,
1010
- reason: "parse_error",
1011
- error: `Reflect proposal quality gate rejected: score=${judgeResult.score}, reason="${judgeResult.reason}"`,
1012
- ...(options.ref ? { ref: options.ref } : {}),
1013
- engine: engineName,
1014
- exitCode: result.exitCode,
1015
- };
1016
- }
1017
- }
1018
- return createReflectProposal({
1019
- payload,
1020
- options,
1021
- stash,
1022
- engineName,
1023
- durationMs: result.durationMs,
1024
- emitReflectFailed,
1025
- outputTelemetry,
1026
- qualityGateSkippedNoJudge,
1027
- sizeGuardRatio: sanitizeOutcome.sizeGuardRatio,
1028
- truncationMarkerLeaked: sanitizeOutcome.truncationMarkerLeaked,
1029
- });
613
+ const emitFailed = (reason, subreason, ref, extra) => {
614
+ emitInvoked();
615
+ appendEvent({
616
+ eventType: "reflect_completed",
617
+ ...(ref ? { ref } : {}),
618
+ metadata: { source: "reflect", ok: false, reason, subreason, ...(extra ?? {}) },
619
+ }, options.eventsCtx);
620
+ };
621
+ return { emitInvoked, emitFailed };
1030
622
  }
1031
- /**
1032
- * Create the reflect proposal from sanitized+judged payload: stamp the R-4/#373
1033
- * lesson provenance marker, call `createProposal`, and emit the terminal
1034
- * `reflect_completed` (or a cooldown skip envelope). Extracted verbatim from
1035
- * `akmReflect`'s finalize tail.
1036
- */
1037
- function createReflectProposal(args) {
1038
- const { payload, options, stash, engineName, durationMs, emitReflectFailed, outputTelemetry, qualityGateSkippedNoJudge, sizeGuardRatio, truncationMarkerLeaked, } = args;
1039
- // 8. Create the proposal. The proposal queue is the ONLY thing reflect
1040
- // writes — promotion to a real asset is gated by `akm proposal accept`.
1041
- //
1042
- // R-4 / #373: Stamp `derived_from_reflect: true` in the frontmatter of any
1043
- // lesson proposal generated by reflect. This provenance marker lets
1044
- // `readRelatedLessons` exclude echo-chamber lessons (lessons that originate
1045
- // from prior reflect runs on the same skill) unless independent feedback
1046
- // evidence exists. ExpeL arXiv:2308.10144 — reject rules without success/
1047
- // failure differential from independent evidence.
1048
- const isLessonProposal = (() => {
1049
- try {
1050
- return parseRefInput(payload.ref).type === "lesson";
1051
- }
1052
- catch {
1053
- return false;
1054
- }
1055
- })();
1056
- const basePayloadFrontmatter = payload.frontmatter ?? {};
1057
- const payloadFrontmatterWithProvenance = isLessonProposal
1058
- ? { ...basePayloadFrontmatter, derived_from_reflect: true }
1059
- : basePayloadFrontmatter;
1060
- const createInput = {
1061
- ref: payload.ref,
1062
- ...(options.target ? { target: options.target } : {}),
1063
- source: "reflect",
1064
- sourceRun: `reflect-${Date.now()}`,
1065
- payload: {
1066
- content: payload.content,
1067
- ...(Object.keys(payloadFrontmatterWithProvenance).length > 0
1068
- ? { frontmatter: payloadFrontmatterWithProvenance }
1069
- : {}),
1070
- },
1071
- // Phase 6A: forward LLM-reported confidence into the proposal record.
1072
- // `parseAgentProposalPayload` already clamps to [0, 1] and drops non-
1073
- // finite values; `createProposal` runs its own sanitizer as a safety net.
1074
- ...(typeof payload.confidence === "number" ? { confidence: payload.confidence } : {}),
1075
- // Attribution tagging: persist the eligibility lane on the proposal so it
1076
- // survives to accept/reject/revert time even across runs. See EligibilitySource.
1077
- ...(options.eligibilitySource ? { eligibilitySource: options.eligibilitySource } : {}),
1078
- // §23.6 fingerprint model-id term (WI-6.4): the engine that generated
1079
- // this draft (reflect resolves engines, not bare model ids).
1080
- modelId: engineName,
623
+ /** A post-dispatch failure envelope (with the run's notices). */
624
+ function reflectFailure(run, result, reason, error, withOutput) {
625
+ return {
626
+ schemaVersion: 2,
627
+ ok: false,
628
+ reason,
629
+ error,
630
+ ...(run.options.ref ? { ref: run.options.ref } : {}),
631
+ engine: run.engineName,
632
+ exitCode: result.exitCode,
633
+ ...(withOutput ? { stdout: result.stdout, ...(result.stderr ? { stderr: result.stderr } : {}) } : {}),
634
+ ...run.notices.fields(),
1081
635
  };
1082
- const proposalResult = emitProposal({ stashDir: stash, proposalsCtx: options.ctx }, createInput);
1083
- if (isProposalSkipped(proposalResult)) {
1084
- // Dedup/cooldown guard fired — surface as a "cooldown" reason (not "parse_error")
1085
- // so the improve orchestrator can distinguish legitimate skips from real failures
1086
- // and exclude them from recentErrors/avoidPatterns injection.
1087
- emitReflectFailed("cooldown", "proposal_skipped", options.ref, {
1088
- proposalSkipReason: proposalResult.reason,
1089
- ...(outputTelemetry ?? {}),
1090
- });
1091
- return {
636
+ }
637
+ function exitCodeMeta(result) {
638
+ return result.exitCode !== null ? { exitCode: result.exitCode } : {};
639
+ }
640
+ function unsupportedTypeFailure(ref, type, detail, emitFailed) {
641
+ emitFailed("unsupported_type", "unsupported_type", ref, { type });
642
+ return {
643
+ failure: {
1092
644
  schemaVersion: 2,
1093
645
  ok: false,
1094
- reason: "cooldown",
1095
- error: `Proposal skipped (${proposalResult.reason}): ${proposalResult.message}`,
1096
- ...(options.ref ? { ref: options.ref } : {}),
1097
- engine: engineName,
646
+ reason: "unsupported_type",
647
+ error: `Reflect refused: asset type "${type}" is not supported by reflect (${detail}). Use \`akm proposal new\` or edit the file directly.`,
648
+ ref,
1098
649
  exitCode: null,
1099
- };
1100
- }
1101
- let proposal = proposalResult;
1102
- const reviewReasons = [];
1103
- if (qualityGateSkippedNoJudge)
1104
- reviewReasons.push("no-judge-configured");
1105
- if (sizeGuardRatio)
1106
- reviewReasons.push("reflect-size-ratio");
1107
- if (truncationMarkerLeaked)
1108
- reviewReasons.push("reflect-truncation-leak");
1109
- if (reviewReasons.length > 0) {
1110
- proposal =
1111
- recordGateDecision(stash, proposal.id, {
1112
- outcome: "deferred",
1113
- reason: reviewReasons.join("+"),
1114
- gate: "reflect",
1115
- ...(sizeGuardRatio ? { measured: Math.round(sizeGuardRatio.ratio * 100) } : {}),
1116
- }, options.ctx) ?? proposal;
1117
- }
1118
- appendEvent({
1119
- eventType: "reflect_completed",
1120
- ref: proposal.ref,
1121
- metadata: {
1122
- proposalId: proposal.id,
1123
- source: "reflect",
1124
- engine: engineName,
1125
- ...(qualityGateSkippedNoJudge ? { qualityGateSkippedNoJudge: true } : {}),
1126
- ...(sizeGuardRatio ? { sizeGuardRatio: sizeGuardRatio.code, sizeGuardRatioValue: sizeGuardRatio.ratio } : {}),
1127
- ...(truncationMarkerLeaked ? { truncationMarkerLeaked: true } : {}),
1128
- ...(outputTelemetry ?? {}),
1129
650
  },
1130
- }, options.eventsCtx);
1131
- return {
1132
- schemaVersion: 2,
1133
- ok: true,
1134
- proposal,
1135
- ref: proposal.ref,
1136
- engine: engineName,
1137
- durationMs,
1138
651
  };
1139
652
  }
1140
- /**
1141
- * Resolve the agent's proposal payload from a successful run: the file-write
1142
- * contract path (read `lastDraftPath`, extract self-rated confidence) or the
1143
- * JSON-stdout path used by direct LLM runners. Returns the payload or a terminal
1144
- * failure envelope.
1145
- */
1146
- function resolveReflectPayload(args) {
1147
- const { result, lastDraftPath, sensitiveValues, options, engineName, emitReflectFailed } = args;
1148
- // 6. Resolve the proposal content.
1149
- //
1150
- // Path A (file-write contract — preferred for agent/sdk runners on long
1151
- // assets): the agent wrote the body to `lastDraftPath` and printed
1152
- // `DRAFT_WRITTEN` on stdout. Load the body from disk and synthesize a
1153
- // payload. The `EXCESSIVE_EXPANSION`/schema-shape gates downstream still
1154
- // apply — they validate content, not transport.
1155
- //
1156
- // Path B (JSON stdout): the direct LLM runner cannot honour file-write.
1157
- const draftFileExists = lastDraftPath !== undefined && fs.existsSync(lastDraftPath) && fs.statSync(lastDraftPath).size > 0;
1158
- const draftSignaled = stdoutSignalsDraftWritten(result.stdout);
1159
- if (draftSignaled && lastDraftPath && !draftFileExists) {
1160
- // Agent claimed to write the draft but the file is missing or empty.
1161
- // Surface as a parse_error rather than silently falling through — the
1162
- // alternative would be parsing the `DRAFT_WRITTEN` sentinel as JSON,
1163
- // which is guaranteed to fail with a confusing message.
1164
- emitReflectFailed("parse_error", "draft_missing", options.ref, {
1165
- ...(result.exitCode !== null ? { exitCode: result.exitCode } : {}),
1166
- });
1167
- return {
1168
- failure: {
1169
- schemaVersion: 2,
1170
- ok: false,
1171
- reason: "parse_error",
1172
- error: `Agent emitted DRAFT_WRITTEN but draft file is missing or empty (${lastDraftPath}). The file-write contract failed; either the agent's file tools are broken or the path was unwritable.`,
1173
- ...(options.ref ? { ref: options.ref } : {}),
1174
- engine: engineName,
1175
- exitCode: result.exitCode,
1176
- stdout: result.stdout,
1177
- ...(result.stderr ? { stderr: result.stderr } : {}),
1178
- },
1179
- };
1180
- }
1181
- if (draftFileExists && lastDraftPath) {
1182
- // Happy path: agent wrote the body to disk. Use the ref the caller
1183
- // supplied (or a placeholder when omitted — the R-3 ref-mismatch guard
1184
- // below has no effect when there is no expected ref).
1185
- const fileContent = redactSensitiveText(fs.readFileSync(lastDraftPath, "utf8"), sensitiveValues);
1186
- // Phase 6A: file-write contract carries self-rated confidence on the
1187
- // `DRAFT_WRITTEN confidence=<n>` sentinel line. Extract it so the
1188
- // file-write path is on equal footing with the JSON-stdout path for
1189
- // auto-accept gating in `akm improve`.
1190
- const draftConfidence = extractDraftConfidence(result.stdout);
1191
- return {
1192
- payload: {
1193
- ref: options.ref ?? "",
1194
- content: fileContent,
1195
- ...(draftConfidence !== undefined ? { confidence: draftConfidence } : {}),
1196
- },
1197
- };
1198
- }
1199
- try {
1200
- return { payload: parseAgentProposalPayload(result.stdout ?? "") };
653
+ /** The target's parsed ref and current content, or a refusal for a type reflect cannot rewrite. */
654
+ async function resolveReflectSource(options, stash, emitFailed) {
655
+ if (!options.ref)
656
+ return { assetContent: undefined, parsedRef: undefined };
657
+ const parsedRef = parseRefInput(options.ref);
658
+ // A secret's content is never read, whatever it looks like.
659
+ if (REFLECT_REFUSED_TYPES.has(parsedRef.type)) {
660
+ return unsupportedTypeFailure(options.ref, parsedRef.type, "secret material is never read or sent to an LLM", emitFailed);
661
+ }
662
+ let assetContent = options.assetContent;
663
+ if (assetContent === undefined) {
664
+ try {
665
+ const qualifiedRef = options.itemRef ?? options.ref;
666
+ const localFilePath = await findAssetFilePath(qualifiedRef, stash);
667
+ if (localFilePath && fs.existsSync(localFilePath)) {
668
+ assetContent = fs.readFileSync(localFilePath, "utf8");
669
+ }
670
+ else {
671
+ const entry = await lookup(parseRefInput(qualifiedRef));
672
+ if (entry?.filePath && fs.existsSync(entry.filePath))
673
+ assetContent = fs.readFileSync(entry.filePath, "utf8");
674
+ }
675
+ }
676
+ catch {
677
+ // An index miss is not fatal: the agent can still propose a fresh asset.
678
+ }
1201
679
  }
1202
- catch (err) {
1203
- // Reclassify cooldown/skip messages that arrive as stdout text instead of
1204
- // valid proposal JSON. These are legitimate skip signals, not parse failures,
1205
- // and should not pollute reflectFailedActions or recentErrors injection.
1206
- const stdoutText = result.stdout ?? "";
1207
- const isCooldownSignal = isStructuredCooldownSignal(stdoutText);
1208
- const reason = isCooldownSignal ? "cooldown" : "parse_error";
1209
- emitReflectFailed(reason, isCooldownSignal ? "stdout_cooldown_signal" : "parse_error", options.ref, {
1210
- ...(result.exitCode !== null ? { exitCode: result.exitCode } : {}),
1211
- ...(reflectLlmTelemetry(result) ?? {}),
1212
- });
1213
- return {
1214
- failure: {
1215
- schemaVersion: 2,
1216
- ok: false,
1217
- reason,
1218
- error: err instanceof Error ? err.message : String(err),
1219
- ...(options.ref ? { ref: options.ref } : {}),
1220
- engine: engineName,
1221
- exitCode: result.exitCode,
1222
- stdout: result.stdout,
1223
- ...(result.stderr ? { stderr: result.stderr } : {}),
1224
- },
1225
- };
680
+ if (!REFLECT_ALLOWED_TYPES.has(parsedRef.type) &&
681
+ (assetContent === undefined || parseFrontmatter(assetContent).frontmatter === null)) {
682
+ return unsupportedTypeFailure(options.ref, parsedRef.type, "its content is not frontmatter + markdown", emitFailed);
1226
683
  }
1227
- }
1228
- function isReflectQualityGateEnabled(activeStrategy) {
1229
- return ((activeStrategy?.processes?.reflect?.qualityGate?.enabled ?? false) ||
1230
- (activeStrategy?.processes?.distill?.qualityGate?.enabled ?? true));
1231
- }
1232
- /** Resolve the exact judge transport before generation so its credential can join the operation snapshot. */
1233
- function resolveReflectQualityJudgeRunner(config, runnerSpec, enabled, onNotices) {
1234
- if (!enabled)
1235
- return Object.freeze({ enabled: false, runner: undefined });
1236
- if (runnerIsLlm(runnerSpec))
1237
- return Object.freeze({ enabled: true, runner: runnerSpec });
1238
- const resolved = resolveImproveLlmExecution({ config, processName: "reflect_proposal_quality-judge" });
1239
- if (resolved)
1240
- onNotices(resolved.notices);
1241
- return Object.freeze({ enabled: true, runner: resolved?.runner });
1242
- }
1243
- /** Acquire through genuine preparation/lowering for all runner kinds, including SDK fallback credentials. */
1244
- function acquireReflectDispatchLease(runnerSpec, onNotices) {
1245
- const prepared = prepareInlineExecutionWithRunner({
1246
- content: "Validate reflect operation transport before dispatch.",
1247
- runner: runnerSpec,
1248
- invocationKind: "direct",
1249
- });
1250
- const lowered = lowerResolvedExecutionRequestWithRunner(prepared.request, prepared.runner);
1251
- onNotices(lowered.notices);
1252
- return acquireLoweredExecutionDispatchLease(lowered);
684
+ return { assetContent, parsedRef };
1253
685
  }
1254
686
  /**
1255
- * Resolve the single named engine for a reflect invocation (standalone --engine
1256
- * / defaults.engine, or the improve strategy's LLM-only process overlay),
1257
- * throwing on any incompatible or missing engine, and validating the unattended
1258
- * LLM requirement. Extracted verbatim from `akmReflect`.
687
+ * The single engine for this invocation: `--engine`, the improve strategy's
688
+ * LLM-only reflect process, or `defaults.engine` (announced when it falls back
689
+ * to the SDK binary). Unattended improve refuses a tool-capable engine.
1259
690
  */
1260
691
  function resolveReflectRunner(options) {
1261
692
  const config = options.config ?? loadConfig();
1262
693
  const activeStrategy = options.improveProfile ?? config.improve?.strategies?.[config.defaults?.improveStrategy ?? "default"];
1263
- let runnerSpec;
1264
- let notices = [];
694
+ const lower = (selection) => {
695
+ const prepared = resolveExecution(selection);
696
+ return buildExecution(prepared.request, prepared.runner);
697
+ };
698
+ let lowered;
1265
699
  if (options.engine) {
1266
- const prepared = prepareInlineExecution({
1267
- content: "reflect engine selection",
1268
- config,
1269
- invocationKind: "direct",
1270
- current: { engine: options.engine },
1271
- });
1272
- const lowered = lowerResolvedExecutionRequest(prepared.request, prepared.config);
1273
- runnerSpec = lowered.runner;
1274
- notices = lowered.notices;
700
+ lowered = lower({ content: "reflect engine selection", config, current: { engine: options.engine } });
1275
701
  }
1276
702
  else if (options.improveProfile) {
1277
703
  const resolved = resolveImproveLlmExecution({
@@ -1283,157 +709,64 @@ function resolveReflectRunner(options) {
1283
709
  if (!resolved) {
1284
710
  throw new ConfigError("Reflect requires an LLM engine for the active improve strategy.", "LLM_NOT_CONFIGURED", "Set defaults.llmEngine or improve.strategies.<name>.processes.reflect.engine.");
1285
711
  }
1286
- runnerSpec = resolved.runner;
1287
- notices = resolved.notices;
712
+ lowered = resolved;
1288
713
  }
1289
714
  else {
1290
715
  const { config: engineConfig, fallbackEngineName } = withEngineFallback(config);
1291
716
  const defaultEngine = engineConfig.defaults?.engine;
1292
- // Announced, never silent — same contract as the workflow freeze boundary
1293
- // and the task runner. Only this arm can select the synthesized engine.
1294
- const engineAnnouncement = fallbackAnnouncement(fallbackEngineName, defaultEngine);
1295
- if (engineAnnouncement)
1296
- warn(engineAnnouncement);
717
+ const announcement = fallbackAnnouncement(fallbackEngineName, defaultEngine);
718
+ if (announcement)
719
+ warn(announcement);
1297
720
  if (!defaultEngine) {
1298
721
  throw new ConfigError(`reflect ${NO_ENGINE_MESSAGE_SUFFIX} ${NO_ENGINE_REMEDY}`, "INVALID_CONFIG_FILE");
1299
722
  }
1300
- const prepared = prepareInlineExecution({
1301
- content: "reflect engine selection",
1302
- config,
1303
- invocationKind: "direct",
1304
- });
1305
- const lowered = lowerResolvedExecutionRequest(prepared.request, prepared.config);
1306
- runnerSpec = lowered.runner;
1307
- notices = lowered.notices;
723
+ lowered = lower({ content: "reflect engine selection", config });
1308
724
  }
725
+ const runnerSpec = lowered.runner;
1309
726
  if (options.eventSource === "improve" && !runnerIsLlm(runnerSpec)) {
1310
727
  throw new ConfigError(`Unattended improve requires an LLM engine for reflect; engine "${runnerSpec.engine ?? options.engine ?? "unknown"}" is tool-capable.`, "INVALID_CONFIG_FILE", "Set defaults.llmEngine or improve.strategies.<name>.processes.reflect.engine to an LLM engine.");
1311
728
  }
1312
729
  const engineName = runnerSpec.engine ?? options.engine;
1313
- if (!engineName) {
730
+ if (!engineName)
1314
731
  throw new ConfigError("Reflect requires a named engine.", "INVALID_CONFIG_FILE");
1315
- }
1316
- return { config, activeStrategy, runnerSpec, engineName, notices };
732
+ return { config, activeStrategy, runnerSpec, engineName, notices: lowered.notices };
1317
733
  }
1318
- function unsupportedTypeFailure(ref, type, detail, emitReflectFailed) {
1319
- emitReflectFailed("unsupported_type", "unsupported_type", ref, { type });
1320
- return {
1321
- failure: {
1322
- schemaVersion: 2,
1323
- ok: false,
1324
- reason: "unsupported_type",
1325
- error: `Reflect refused: asset type "${type}" is not supported by reflect (${detail}). Use \`akm proposal new\` or edit the file directly.`,
1326
- ref,
1327
- exitCode: null,
1328
- },
1329
- };
1330
- }
1331
- /**
1332
- * Resolve the reflect target's parsed ref + current on-disk content: enforce the
1333
- * REFLECT_ALLOWED_TYPES markdown-canonical type guard (returning a terminal
1334
- * `unsupported_type` failure), honour the `options.assetContent` test seam, else
1335
- * best-effort load via the local file path / index lookup. Extracted verbatim
1336
- * from `akmReflect`.
1337
- */
1338
- async function resolveReflectSource(options, stash, emitReflectFailed) {
1339
- let assetContent;
1340
- let parsedRef;
1341
- if (options.ref) {
1342
- parsedRef = parseRefInput(options.ref);
1343
- // 2a. Refuse `secret` before any content is read — a secret's content is
1344
- // never touched by reflect, regardless of what it happens to look like.
1345
- if (REFLECT_REFUSED_TYPES.has(parsedRef.type)) {
1346
- return unsupportedTypeFailure(options.ref, parsedRef.type, "secret material is never read or sent to an LLM", emitReflectFailed);
1347
- }
1348
- if (options.assetContent !== undefined) {
1349
- // Test seam — caller pre-loaded the source content.
1350
- assetContent = options.assetContent;
1351
- }
1352
- else {
1353
- try {
1354
- // Resolve the source by item_ref when planning supplied one, otherwise
1355
- // use the input conceptId.
1356
- const qualifiedRef = options.itemRef ?? durableImproveRef(options.ref);
1357
- const localFilePath = await findAssetFilePath(qualifiedRef, stash);
1358
- if (localFilePath && fs.existsSync(localFilePath)) {
1359
- assetContent = fs.readFileSync(localFilePath, "utf8");
1360
- }
1361
- else {
1362
- const entry = await lookup(parseRefInput(qualifiedRef));
1363
- if (entry?.filePath && fs.existsSync(entry.filePath)) {
1364
- assetContent = fs.readFileSync(entry.filePath, "utf8");
1365
- }
1366
- }
1367
- }
1368
- catch {
1369
- // Index miss is non-fatal — the agent can still propose a fresh asset.
1370
- }
1371
- }
1372
- if (!REFLECT_ALLOWED_TYPES.has(parsedRef.type)) {
1373
- if (assetContent === undefined || !isReflectableSourceShape(assetContent)) {
1374
- return unsupportedTypeFailure(options.ref, parsedRef.type, "its content is not frontmatter + markdown", emitReflectFailed);
1375
- }
1376
- }
1377
- }
1378
- return { assetContent, parsedRef };
734
+ /** Lower a runner and check its credentials, so a bad transport fails before any work. */
735
+ function preflightReflectDispatch(runnerSpec, onNotices) {
736
+ const prepared = resolveExecution({
737
+ content: "Validate reflect operation transport before dispatch.",
738
+ runner: runnerSpec,
739
+ });
740
+ const lowered = buildExecution(prepared.request, prepared.runner);
741
+ onNotices(lowered.notices);
742
+ assertRunnerCredentials(lowered.runner);
1379
743
  }
1380
744
  /**
1381
- * #952 — the flat REFLECT_CONTENT_CAP (12 000 chars) exists only to avoid
1382
- * E2BIG when the prompt travels through CLI argv (agent/SDK runners). The
1383
- * direct-LLM HTTP path never touches argv, so it can use the resolved
1384
- * engine's own context window instead. The reserve for "the rest of the
1385
- * prompt" is measured directly (not guessed): build the same prompt with
1386
- * the content cap forced to zero and use its length as the overhead, so
1387
- * feedback/standards/schema-hints/prior-draft size is accounted for
1388
- * exactly, per this call. A reflect rewrite returns a body roughly the
1389
- * size of the input, so the budget only spends HALF of the usable window
1390
- * on input content and reserves the other half for the model's own
1391
- * output — otherwise a full-context request leaves no room for a
1392
- * response. Never drops below the flat floor.
1393
- *
1394
- * Shared by the real dispatch path ({@link runReflectRefineIterations}) and
1395
- * `renderReflectPromptPreview`'s `--show-prompt` preview, so the preview
1396
- * renders the exact prompt reflect would actually send for LLM runners
1397
- * instead of always the flat-cap prompt.
745
+ * The flat 12k content cap exists for CLI argv; the HTTP runner can spend half
746
+ * its context window (after the rest of the prompt) on the asset, reserving
747
+ * the other half for the rewrite. Never below the flat floor.
1398
748
  */
1399
749
  function computeReflectContentBudgetChars(promptInput, runnerSpec) {
1400
- return runnerIsLlm(runnerSpec) && promptInput.assetContent?.trim()
1401
- ? Math.max(REFLECT_CONTENT_CAP, Math.floor(((runnerSpec.connection.contextLength ?? DEFAULT_CONTEXT_LENGTH_TOKENS) * CHARS_PER_TOKEN -
1402
- buildReflectPrompt({ ...promptInput, contentBudgetChars: 0 }).prompt.length) /
1403
- 2))
1404
- : undefined;
750
+ if (!runnerIsLlm(runnerSpec) || !promptInput.assetContent?.trim())
751
+ return undefined;
752
+ const window = (runnerSpec.connection.contextLength ?? DEFAULT_CONTEXT_LENGTH_TOKENS) * CHARS_PER_TOKEN;
753
+ const overhead = buildReflectPrompt({ ...promptInput, contentBudgetChars: 0 }).prompt.length;
754
+ return Math.max(REFLECT_CONTENT_CAP, Math.floor((window - overhead) / 2));
1405
755
  }
1406
- /**
1407
- * #952 — gather every read-only prompt-input source {@link buildReflectPromptInput}
1408
- * folds into a `ReflectPromptInput`: recent feedback, schema/lint hints, related
1409
- * lessons, previously-rejected proposals, and stash standards context.
1410
- *
1411
- * Shared by the real dispatch path (`akmReflect`'s step 4, via
1412
- * {@link runReflectRefineIterations}) and `renderReflectPromptPreview`'s
1413
- * `--show-prompt` preview, so both gather from exactly one definition instead
1414
- * of two copies that can drift out of agreement.
1415
- */
1416
- async function gatherReflectPromptSources(options, stash, parsedRef, assetContent, assetCtx) {
1417
- const feedback = readRecentFeedback(options.ref ? (options.itemRef ?? durableImproveRef(options.ref)) : undefined, options.eventsCtx);
1418
- const schemaHints = buildSchemaHints(parsedRef?.type ?? "", assetContent);
1419
- const relatedLessons = options.ref && parsedRef ? await readRelatedLessons(assetCtx, stash, options.ref, parsedRef, options.itemRef) : [];
1420
- // Reflexion-style verbal-RL: inject rejected proposals so the agent avoids
1421
- // reproducing proposals that have already been reviewed and refused.
1422
- const rejectedProposals = readRejectedProposals(stash, options.ref, options.ctx);
1423
- // Standards "rulebook" for this target — stash convention/meta facts; empty
1424
- // when none fire.
1425
- const standardsContext = resolveStandardsContext(options.ref, stash);
1426
- return { feedback, schemaHints, relatedLessons, rejectedProposals, standardsContext };
756
+ /** Every read-only prompt input, shared by dispatch and `--show-prompt`. */
757
+ async function gatherReflectPromptSources(options, stash, parsedRef, assetContent) {
758
+ return {
759
+ feedback: readRecentFeedback(options.ref ? (options.itemRef ?? options.ref) : undefined, options.eventsCtx),
760
+ schemaHints: buildSchemaHints(parsedRef?.type ?? "", assetContent),
761
+ relatedLessons: options.ref && parsedRef
762
+ ? await readRelatedLessons(stash, options.ref, parsedRef, options.itemRef, options.eventsCtx)
763
+ : [],
764
+ rejectedProposals: rejectedProposalContext(stash, options.ref, options.ctx),
765
+ standardsContext: resolveStandardsContext(options.ref, stash),
766
+ };
1427
767
  }
1428
- /**
1429
- * #952 — assemble the `ReflectPromptInput` object literal reflect actually
1430
- * sends, from gathered sources plus the per-call values (draft path, prior
1431
- * draft). Shared by the real dispatch path ({@link runReflectRefineIterations})
1432
- * and `renderReflectPromptPreview`'s `--show-prompt` preview — including
1433
- * `avoidPatterns`, which the preview previously omitted even though a live
1434
- * improve loop passes it (recent-error context, O-5 / #378).
1435
- */
1436
- function buildReflectPromptInput(args) {
768
+ /** The exact prompt reflect sends, shared by dispatch and `--show-prompt`. */
769
+ function buildReflectPromptText(args) {
1437
770
  const { options, parsedRef, assetContent, sources, runnerSpec, draftFilePath, priorDraft } = args;
1438
771
  const { feedback, schemaHints, relatedLessons, rejectedProposals, standardsContext } = sources;
1439
772
  const outputMode = runnerIsLlm(runnerSpec)
@@ -1441,7 +774,7 @@ function buildReflectPromptInput(args) {
1441
774
  ? "json_schema"
1442
775
  : "framed_markdown"
1443
776
  : undefined;
1444
- return {
777
+ const input = {
1445
778
  ...(options.ref ? { ref: options.ref } : {}),
1446
779
  ...(parsedRef?.type ? { type: parsedRef.type } : {}),
1447
780
  ...(parsedRef?.name ? { name: parsedRef.name } : {}),
@@ -1453,133 +786,100 @@ function buildReflectPromptInput(args) {
1453
786
  ...(standardsContext.trim() ? { standardsContext } : {}),
1454
787
  ...(options.avoidPatterns && options.avoidPatterns.length > 0 ? { avoidPatterns: options.avoidPatterns } : {}),
1455
788
  ...(rejectedProposals.length > 0 ? { rejectedProposals } : {}),
1456
- // R-1: inject prior draft as self-critique target on iterations > 0
1457
789
  ...(priorDraft !== undefined ? { priorDraft } : {}),
1458
- // Issue A (#reflect-pipeline file-write contract): when the runner can
1459
- // touch the filesystem, instruct the agent to write the proposal body
1460
- // to a tmp file instead of inlining it in JSON. Avoids parse failures
1461
- // on long bodies (e.g. knowledge/systems/KOKORO_USAGE_GUIDE 8.4KB).
1462
790
  ...(draftFilePath ? { draftFilePath } : {}),
1463
791
  ...(outputMode ? { outputMode } : {}),
1464
792
  };
793
+ const contentBudgetChars = computeReflectContentBudgetChars(input, runnerSpec);
794
+ const { prompt } = buildReflectPrompt({
795
+ ...input,
796
+ ...(contentBudgetChars !== undefined ? { contentBudgetChars } : {}),
797
+ });
798
+ return { prompt, ...(outputMode ? { outputMode } : {}) };
1465
799
  }
1466
800
  /**
1467
- * Run the agent with the optional Self-Refine loop (R-1 / #372): up to
1468
- * `maxRefineIters` invocations, each injecting the prior draft as self-critique
1469
- * context and exiting early on a no-op refinement. Synthesizes per-iteration
1470
- * draft paths into `draftPathsToCleanup` (mutated) and returns the final agent
1471
- * result + last draft path. Extracted verbatim from `akmReflect`.
801
+ * Dispatch with the optional self-refine loop: up to `maxRefineIters` passes,
802
+ * each critiquing the prior draft, stopping early on an unchanged draft. The
803
+ * direct-LLM repair budget is shared across passes.
1472
804
  */
1473
805
  async function runReflectRefineIterations(args) {
1474
- const { options, parsedRef, assetContent, sources, runnerSpec, lease, agentEnv, draftPathsToCleanup, onNotices } = args;
806
+ const { run, parsedRef, assetContent, sources, agentEnv, draftPaths } = args;
807
+ const { options, runnerSpec } = run;
1475
808
  const maxRefineIters = Math.max(1, options.maxRefineIters ?? 1);
1476
- // Determine whether this dispatch can honour the file-write contract.
1477
- // Agent CLI + OpenCode SDK runners both have filesystem access; the direct
1478
- // LLM HTTP runner does NOT.
1479
- const canRunnerWriteFile = runnerSupportsFileWrite(runnerSpec);
1480
- // Initialized to a sentinel; always overwritten in the first loop iteration
1481
- // (maxRefineIters is clamped to >= 1 above).
809
+ const canWriteFile = runnerSupportsFileWrite(runnerSpec);
1482
810
  let result = {};
1483
811
  let priorDraft;
1484
812
  let lastDraftPath;
1485
813
  let repairAttempts = 0;
1486
814
  for (let iter = 0; iter < maxRefineIters; iter++) {
1487
- // Synthesize a fresh tmp path per iteration so refinement passes never
1488
- // clobber an earlier draft (and so reading back is unambiguous).
1489
- const iterDraftPath = canRunnerWriteFile ? synthesizeReflectDraftPath(options.ref) : undefined;
1490
- if (iterDraftPath) {
1491
- draftPathsToCleanup.push(iterDraftPath);
1492
- lastDraftPath = iterDraftPath;
815
+ const draftFilePath = canWriteFile ? synthesizeReflectDraftPath(options.ref) : undefined;
816
+ if (draftFilePath) {
817
+ draftPaths.push(draftFilePath);
818
+ lastDraftPath = draftFilePath;
1493
819
  }
1494
- const promptInput = buildReflectPromptInput({
820
+ const { prompt, outputMode } = buildReflectPromptText({
1495
821
  options,
1496
822
  parsedRef,
1497
823
  assetContent,
1498
824
  sources,
1499
825
  runnerSpec,
1500
- draftFilePath: iterDraftPath,
826
+ draftFilePath,
1501
827
  priorDraft,
1502
828
  });
1503
- const contentBudgetChars = computeReflectContentBudgetChars(promptInput, runnerSpec);
1504
- const { prompt } = buildReflectPrompt({
1505
- ...promptInput,
1506
- ...(contentBudgetChars !== undefined ? { contentBudgetChars } : {}),
1507
- });
1508
829
  let iterResult;
1509
830
  if (runnerIsLlm(runnerSpec)) {
1510
- // LLM HTTP runners cannot honor the file-write contract, so they return
1511
- // structured output through stdout. callStructured owns preparation,
1512
- // lowering, credential materialization, and direct transport dispatch.
1513
831
  iterResult = await runReflectViaLlm({
1514
832
  prompt,
1515
833
  runner: runnerSpec,
1516
- lease,
1517
834
  ...(Object.hasOwn(options, "timeoutMs") ? { timeoutMs: options.timeoutMs } : {}),
1518
835
  ...(options.signal ? { signal: options.signal } : {}),
1519
836
  priorDraft,
1520
837
  iteration: iter,
1521
- ...(promptInput.outputMode === "json_schema"
838
+ ...(outputMode === "json_schema"
1522
839
  ? { responseSchema: options.ref ? REFLECT_JSON_SCHEMA : REFLECT_UNSCOPED_JSON_SCHEMA }
1523
840
  : {}),
1524
- outputMode: promptInput.outputMode ?? "framed_markdown",
841
+ outputMode: outputMode ?? "framed_markdown",
1525
842
  ...(options.ref ? { targetRef: options.ref } : {}),
1526
843
  allowRepair: repairAttempts === 0,
1527
844
  ...(options.chat ? { chat: options.chat } : {}),
1528
- onNotices,
845
+ onNotices: run.notices.add,
1529
846
  });
1530
847
  }
1531
848
  else {
1532
- const conversationPriorDraft = priorDraft;
1533
- const hasConversation = conversationPriorDraft !== undefined && iter > 0;
849
+ const conversation = priorDraft !== undefined && iter > 0
850
+ ? [
851
+ { role: "user", content: prompt },
852
+ { role: "assistant", content: priorDraft },
853
+ ]
854
+ : undefined;
1534
855
  const current = {
1535
856
  ...(Object.hasOwn(options, "timeoutMs") ? { timeout: options.timeoutMs } : {}),
1536
857
  ...(Object.keys(agentEnv).length > 0 ? { environment: agentEnv } : {}),
1537
858
  };
1538
- const prepared = prepareInlineExecutionWithRunner({
1539
- content: hasConversation ? REFLECT_CRITIQUE_PROMPT : (prompt ?? ""),
1540
- ...(hasConversation
1541
- ? {
1542
- conversation: [
1543
- { role: "user", content: prompt ?? "" },
1544
- { role: "assistant", content: conversationPriorDraft },
1545
- ],
1546
- }
1547
- : {}),
859
+ const prepared = resolveExecution({
860
+ content: conversation ? REFLECT_CRITIQUE_PROMPT : prompt,
861
+ ...(conversation ? { conversation } : {}),
1548
862
  runner: runnerSpec,
1549
- invocationKind: "direct",
1550
863
  ...(Object.keys(current).length > 0 ? { current } : {}),
1551
864
  });
1552
- const lowered = lowerResolvedExecutionRequestWithRunner(prepared.request, prepared.runner);
1553
- onNotices(lowered.notices);
1554
- iterResult = await dispatchLoweredExecutionRequest(lowered, {
1555
- lease,
865
+ const lowered = buildExecution(prepared.request, prepared.runner);
866
+ run.notices.add(lowered.notices);
867
+ iterResult = await runExecution(lowered, {
1556
868
  ...(options.runSdk ? { runSdk: options.runSdk } : {}),
1557
- runOptions: {
1558
- ...(options.signal ? { signal: options.signal } : {}),
1559
- ...(options.runAgentOptions ?? {}),
1560
- },
869
+ runOptions: { ...(options.signal ? { signal: options.signal } : {}), ...(options.runAgentOptions ?? {}) },
1561
870
  });
1562
871
  }
1563
- const iterTelemetry = reflectLlmTelemetry(iterResult);
1564
- if (iterTelemetry)
1565
- repairAttempts += iterTelemetry.repairAttempts;
1566
- result = iterTelemetry
1567
- ? {
1568
- ...iterResult,
1569
- parsed: {
1570
- ...iterResult.parsed,
1571
- ...iterTelemetry,
1572
- repairAttempts,
1573
- },
1574
- }
872
+ const telemetry = reflectLlmTelemetry(iterResult);
873
+ if (telemetry)
874
+ repairAttempts += telemetry.repairAttempts;
875
+ result = telemetry
876
+ ? { ...iterResult, parsed: { ...iterResult.parsed, ...telemetry, repairAttempts } }
1575
877
  : iterResult;
1576
878
  if (!result.ok)
1577
- break; // surface failure after loop
1578
- // On success, extract the draft content for the next iteration.
1579
- // If the agent returns the same content as the prior draft, stop early
1580
- // (no-op refinement) to avoid wasting tokens on identical iterations.
879
+ break;
1581
880
  if (iter < maxRefineIters - 1) {
1582
- const nextDraft = reflectLlmPriorDraft(result) ?? result.stdout ?? "";
881
+ const priorFromLlm = parsedRecord(result)?.priorDraft;
882
+ const nextDraft = typeof priorFromLlm === "string" ? priorFromLlm : (result.stdout ?? "");
1583
883
  if (priorDraft !== undefined && nextDraft === priorDraft)
1584
884
  break;
1585
885
  priorDraft = nextDraft;
@@ -1588,354 +888,343 @@ async function runReflectRefineIterations(args) {
1588
888
  return { result, lastDraftPath };
1589
889
  }
1590
890
  /**
1591
- * WI-9.10: build one `akm reflect` invocation's {@link RunContext} purely
1592
- * from values `akmReflect` has already resolved by the time it calls this
1593
- * (stash, config, runnerSpec) plus the caller-supplied seams on `options` —
1594
- * no second config load, no new db handle. reflect has no `dryRun` option
1595
- * (it never writes source assets directly, only the proposal queue — see the
1596
- * module docblock) so `dryRun` is always `false` here. reflect also has no
1597
- * `sourceRun` option; the value below mirrors the same `reflect-${Date.now()}`
1598
- * convention already used inline at proposal creation time (see
1599
- * `createInput` further down this file), as a fresh, independent token —
1600
- * nothing yet reads `ctx.sourceRun`.
891
+ * The proposal payload from a successful run: the agent's draft file
892
+ * (file-write contract, `DRAFT_WRITTEN confidence=<n>` on stdout) or the JSON
893
+ * payload on stdout.
1601
894
  */
1602
- function buildReflectRunContext(args) {
1603
- const { options, stash, config, runnerSpec } = args;
1604
- return createRunContext({
1605
- stashDir: stash,
1606
- config,
1607
- eventsCtx: options.eventsCtx ?? {},
1608
- // Not yet wired into any proposal call site this stage (mirrors
1609
- // buildImproveRunContext's proposalsCtx comment in improve.ts).
1610
- proposalsCtx: options.ctx ?? {},
1611
- chat: options.chat,
1612
- getLlmRunner: () => (runnerIsLlm(runnerSpec) ? runnerSpec : null),
1613
- sourceRun: `reflect-${Date.now()}`,
1614
- dryRun: false,
1615
- signal: options.signal,
1616
- });
895
+ function resolveReflectPayload(run, result, lastDraftPath, sensitiveValues) {
896
+ const { options } = run;
897
+ const draftFileExists = lastDraftPath !== undefined && fs.existsSync(lastDraftPath) && fs.statSync(lastDraftPath).size > 0;
898
+ const draftSignaled = /\bDRAFT_WRITTEN\b/.test(result.stdout ?? "");
899
+ if (draftSignaled && lastDraftPath && !draftFileExists) {
900
+ run.emitFailed("parse_error", "draft_missing", options.ref, exitCodeMeta(result));
901
+ return {
902
+ failure: reflectFailure(run, result, "parse_error", `Agent emitted DRAFT_WRITTEN but draft file is missing or empty (${lastDraftPath}). The file-write contract failed; either the agent's file tools are broken or the path was unwritable.`, true),
903
+ };
904
+ }
905
+ if (draftFileExists && lastDraftPath) {
906
+ const draftConfidence = extractDraftConfidence(result.stdout);
907
+ return {
908
+ payload: {
909
+ ref: options.ref ?? "",
910
+ content: redactSensitiveText(fs.readFileSync(lastDraftPath, "utf8"), sensitiveValues),
911
+ ...(draftConfidence !== undefined ? { confidence: draftConfidence } : {}),
912
+ },
913
+ };
914
+ }
915
+ try {
916
+ return { payload: parseAgentProposalPayload(result.stdout ?? "") };
917
+ }
918
+ catch (err) {
919
+ run.emitFailed("parse_error", "parse_error", options.ref, {
920
+ ...exitCodeMeta(result),
921
+ ...(reflectLlmTelemetry(result) ?? {}),
922
+ });
923
+ return {
924
+ failure: reflectFailure(run, result, "parse_error", err instanceof Error ? err.message : String(err), true),
925
+ };
926
+ }
1617
927
  }
928
+ const NOISE_SUBREASONS = {
929
+ noop: "reflect_skipped_noop",
930
+ cosmetic: "reflect_skipped_cosmetic",
931
+ "low-value": "reflect_skipped_low_value",
932
+ };
1618
933
  /**
1619
- * Build idempotent `reflect_invoked` / `reflect_completed` emitters. Invocation
1620
- * is delayed until canonical dispatch validates symbolic credentials, while
1621
- * deterministic pre-dispatch failures still close an invoke/complete pair.
1622
- *
1623
- * Fix #3 (observability 0.8.0): every failure path below MUST emit
1624
- * `reflect_completed` so observers can close the invoke/complete loop. The
1625
- * three success-side `reflect_completed` emit sites carry rich metadata
1626
- * (qualityRejected, sanitized, proposalId, etc.); the failure-side emits
1627
- * carry `{ok: false, reason}` plus the ref when known. Stable failure
1628
- * reasons line up with `AgentFailureReason`: "parse_error", "non_zero_exit",
1629
- * "cooldown", "timeout", "spawn_failed", "llm_*", plus the synthetic
1630
- * "ref_mismatch" / "enoent" / "draft_missing" subtypes for cases the agent
1631
- * surface conflates as "parse_error". Sub-reasons land in `subreason`.
934
+ * Sanitize, drop a no-op/cosmetic (and optionally low-value) change, judge the
935
+ * exact content that would be persisted, then mint. Size-flagged or
936
+ * truncation-leaking content skips the judge and waits for review.
1632
937
  */
1633
- function buildReflectEventEmitters(options) {
1634
- let invoked = false;
1635
- const emitInvoked = () => {
1636
- if (invoked)
1637
- return;
1638
- appendEvent({
1639
- eventType: "reflect_invoked",
1640
- // Key on item_ref when planning supplied one, otherwise the conceptId.
1641
- ...(options.ref ? { ref: options.itemRef ?? durableImproveRef(options.ref) } : {}),
1642
- metadata: {
1643
- ...(options.task ? { task: options.task } : {}),
1644
- ...(options.engine ? { engine: options.engine } : {}),
1645
- // Attribution tagging: stamp the eligibility lane so reflect_invoked can be
1646
- // sliced by lane downstream. See EligibilitySource.
1647
- ...(options.eligibilitySource ? { eligibilitySource: options.eligibilitySource } : {}),
1648
- },
1649
- }, options.eventsCtx);
1650
- invoked = true;
938
+ async function finalizeReflectProposal(args) {
939
+ const { run, assetContent, result, judge, feedback } = args;
940
+ const { options } = run;
941
+ const telemetry = reflectLlmTelemetry(result) ?? {};
942
+ const sanitized = sanitizeReflectPayload({ content: args.payload.content, ...(args.payload.frontmatter ? { frontmatter: args.payload.frontmatter } : {}) }, assetContent, args.payload.ref);
943
+ const payload = {
944
+ ...args.payload,
945
+ content: sanitized.content,
946
+ ...(sanitized.frontmatter ? { frontmatter: sanitized.frontmatter } : {}),
1651
947
  };
1652
- const emitFailed = (reason, subreason, ref, extra) => {
1653
- emitInvoked();
948
+ if (assetContent !== undefined) {
949
+ const changeKind = classifyReflectChange(assetContent, payload.content);
950
+ if (changeKind === "noop" ||
951
+ changeKind === "cosmetic" ||
952
+ (changeKind === "low-value" && options.lowValueFilter === true)) {
953
+ run.emitFailed("no_change", NOISE_SUBREASONS[changeKind], options.ref, { changeKind, ...telemetry });
954
+ const what = changeKind === "noop"
955
+ ? "identical to the current asset (empty diff)"
956
+ : changeKind === "low-value"
957
+ ? "a low-value prose micro-rewrite (few changed tokens, no structural changes)"
958
+ : "a cosmetic-only reformat of the current asset (whitespace/fence/YAML-folding changes)";
959
+ return reflectFailure(run, result, "no_change", `Reflect skipped: proposed content for ${payload.ref} is ${what}; no proposal created.`, false);
960
+ }
961
+ }
962
+ const flagged = Boolean(sanitized.sizeGuardRatio || sanitized.truncationMarkerLeaked);
963
+ const judged = judge.enabled && !flagged;
964
+ /** A judge refused the revision: record it for the ledger's rejection window and stop. */
965
+ const refuse = (detail, metadata, message) => {
966
+ if (options.ref) {
967
+ recordLedgerAttempt({ proposalsCtx: options.ctx, eventsCtx: options.eventsCtx }, {
968
+ stashDir: run.stash,
969
+ ref: options.itemRef ?? options.ref,
970
+ source: "reflect",
971
+ outcome: "quality_rejected",
972
+ detail,
973
+ });
974
+ }
1654
975
  appendEvent({
1655
976
  eventType: "reflect_completed",
1656
- ...(ref ? { ref } : {}),
1657
- metadata: {
1658
- source: "reflect",
1659
- ok: false,
1660
- reason,
1661
- subreason,
1662
- ...(extra ?? {}),
1663
- },
977
+ ref: payload.ref,
978
+ metadata: { source: "reflect", qualityRejected: true, ...metadata, ...telemetry },
1664
979
  }, options.eventsCtx);
980
+ return reflectFailure(run, result, "quality_rejected", message, false);
1665
981
  };
1666
- return { emitInvoked, emitFailed };
1667
- }
1668
- function cleanupReflectDrafts(paths) {
1669
- for (const draftPath of paths) {
1670
- try {
1671
- if (fs.existsSync(draftPath))
1672
- fs.unlinkSync(draftPath);
1673
- }
1674
- catch {
1675
- // Draft cleanup is best-effort; the proposal result remains authoritative.
982
+ if (judged) {
983
+ const verdict = await runReflectQualityJudge(run.config, payload.content, assetContent ?? "", feedback, options.chat, {
984
+ runnerSelectionFrozen: true,
985
+ ...(judge.runner ? { llmRunner: judge.runner } : {}),
986
+ ...(Object.hasOwn(options, "timeoutMs") ? { timeoutMs: options.timeoutMs } : {}),
987
+ ...(options.signal ? { signal: options.signal } : {}),
988
+ onNotices: run.notices.add,
989
+ });
990
+ if (!verdict.pass) {
991
+ return refuse(verdict.reason, {
992
+ qualityScore: verdict.score,
993
+ qualityReason: verdict.reason,
994
+ ...(verdict.criteria ? { qualityCriteria: verdict.criteria } : {}),
995
+ }, `Reflect proposal quality gate rejected: score=${verdict.score}, reason="${verdict.reason}"`);
1676
996
  }
1677
997
  }
1678
- }
1679
- function validateReflectPayloadRef(args) {
1680
- const { payload, result, options, engineName, emitReflectFailed, executionNotices } = args;
1681
- if (!options.ref)
1682
- return undefined;
1683
- try {
1684
- const expectedParsed = parseRefInput(options.ref);
1685
- const actualParsed = parseRefInput(payload.ref);
1686
- if (expectedParsed.type === actualParsed.type && expectedParsed.name === actualParsed.name)
1687
- return undefined;
1688
- emitReflectFailed("parse_error", "ref_mismatch", options.ref, {
1689
- expectedRef: options.ref,
1690
- actualRef: payload.ref,
1691
- ...(result.exitCode !== null ? { exitCode: result.exitCode } : {}),
1692
- ...(reflectLlmTelemetry(result) ?? {}),
998
+ // #722: a rewrite of an existing asset must not grade lower on its own retrieval queries.
999
+ if (judged && judge.runner && assetContent !== undefined) {
1000
+ const retrieval = await runRetrievalRegressionGate({
1001
+ ref: payload.ref,
1002
+ before: assetContent,
1003
+ after: payload.content,
1004
+ queries: loadRetrievalQueries({ proposalsCtx: options.ctx, eventsCtx: options.eventsCtx }, payload.ref),
1005
+ runner: judge.runner,
1006
+ ...(options.chat ? { chat: options.chat } : {}),
1007
+ ...(Object.hasOwn(options, "timeoutMs") ? { timeoutMs: options.timeoutMs } : {}),
1008
+ ...(options.signal ? { signal: options.signal } : {}),
1009
+ onNotices: run.notices.add,
1693
1010
  });
1694
- return {
1695
- schemaVersion: 2,
1696
- ok: false,
1697
- reason: "parse_error",
1698
- error: `Agent retargeted proposal: expected ref "${options.ref}" but got "${payload.ref}". Proposal rejected to prevent silent ref hallucination.`,
1699
- ref: options.ref,
1700
- engine: engineName,
1701
- exitCode: result.exitCode,
1702
- stdout: result.stdout,
1703
- ...(result.stderr ? { stderr: result.stderr } : {}),
1704
- ...reflectNoticeFields(executionNotices),
1705
- };
1706
- }
1707
- catch {
1708
- // Malformed refs are rejected downstream by proposal validation.
1709
- return undefined;
1011
+ if (!retrieval.pass) {
1012
+ return refuse(retrieval.reason, {
1013
+ retrievalRegression: true,
1014
+ retrievalQueries: retrieval.queries,
1015
+ ...(retrieval.oldMean !== undefined ? { retrievalGradeBefore: retrieval.oldMean } : {}),
1016
+ ...(retrieval.newMean !== undefined ? { retrievalGradeAfter: retrieval.newMean } : {}),
1017
+ }, `Reflect proposal refused: ${retrieval.reason}`);
1018
+ }
1710
1019
  }
1020
+ // A lesson reflect wrote is marked so a later reflect on the same skill does
1021
+ // not read it back as independent evidence.
1022
+ const frontmatter = {
1023
+ ...(payload.frontmatter ?? {}),
1024
+ ...(lenientRefType(payload.ref) === "lesson" ? { derived_from_reflect: true } : {}),
1025
+ };
1026
+ const reviewReasons = [
1027
+ ...(judge.skippedNoJudge ? ["no-judge-configured"] : []),
1028
+ ...(sanitized.sizeGuardRatio ? ["reflect-size-ratio"] : []),
1029
+ ...(sanitized.truncationMarkerLeaked ? ["reflect-truncation-leak"] : []),
1030
+ ];
1031
+ const proposal = mintProposal(run.stash, options.ctx, {
1032
+ ref: payload.ref,
1033
+ ...(options.target ? { target: options.target } : {}),
1034
+ source: "reflect",
1035
+ sourceRun: `reflect-${Date.now()}`,
1036
+ payload: { content: payload.content, ...(Object.keys(frontmatter).length > 0 ? { frontmatter } : {}) },
1037
+ ...(typeof payload.confidence === "number" ? { confidence: payload.confidence } : {}),
1038
+ ...(options.eligibilitySource ? { eligibilitySource: options.eligibilitySource } : {}),
1039
+ ...(options.itemRef ? { attemptedRefs: [options.itemRef] } : {}),
1040
+ }, reviewReasons.length > 0
1041
+ ? {
1042
+ review: {
1043
+ reason: reviewReasons.join("+"),
1044
+ gate: "reflect",
1045
+ ...(sanitized.sizeGuardRatio ? { measured: Math.round(sanitized.sizeGuardRatio.ratio * 100) } : {}),
1046
+ },
1047
+ }
1048
+ : { judged });
1049
+ appendEvent({
1050
+ eventType: "reflect_completed",
1051
+ ref: proposal.ref,
1052
+ metadata: {
1053
+ proposalId: proposal.id,
1054
+ source: "reflect",
1055
+ engine: run.engineName,
1056
+ ...(judge.skippedNoJudge ? { qualityGateSkippedNoJudge: true } : {}),
1057
+ ...(sanitized.sizeGuardRatio
1058
+ ? { sizeGuardRatio: sanitized.sizeGuardRatio.code, sizeGuardRatioValue: sanitized.sizeGuardRatio.ratio }
1059
+ : {}),
1060
+ ...(sanitized.truncationMarkerLeaked ? { truncationMarkerLeaked: true } : {}),
1061
+ ...telemetry,
1062
+ },
1063
+ }, options.eventsCtx);
1064
+ return {
1065
+ schemaVersion: 2,
1066
+ ok: true,
1067
+ proposal,
1068
+ ref: proposal.ref,
1069
+ engine: run.engineName,
1070
+ durationMs: result.durationMs,
1071
+ ...run.notices.fields(),
1072
+ };
1711
1073
  }
1712
1074
  /**
1713
- * #952 — render the composed reflect prompt for exactly one asset with no
1714
- * engine dispatch. Reuses every read-only step `akmReflect` performs before
1715
- * {@link buildReflectPrompt} (source resolution, runner resolution, feedback /
1716
- * schema-hint / related-lesson / rejected-proposal gathering) and stops right
1717
- * there: no dispatch lease is acquired, no request is sent, and — because the
1718
- * `emitReflectFailed` callback passed to {@link resolveReflectSource} here is
1719
- * a no-op — no `reflect_invoked`/`reflect_completed` event is appended either.
1720
- *
1721
- * `akm improve <ref> --show-prompt` (`improve-cli.ts`) is the CLI surface: a
1722
- * field operator uses it to see the exact prompt reflect would send, in
1723
- * seconds, without running a full improve cycle or needing a reachable
1724
- * engine.
1075
+ * `akm improve <ref> --show-prompt`: the exact prompt reflect would send for
1076
+ * one asset. Read-only: no credential, no dispatch, no event.
1725
1077
  */
1726
1078
  export async function renderReflectPromptPreview(options) {
1727
1079
  if (!options.ref) {
1728
1080
  throw new UsageError("renderReflectPromptPreview requires options.ref.", "INVALID_FLAG_VALUE");
1729
1081
  }
1730
1082
  const ref = options.ref;
1731
- const stash = resolveRunStashDir(options.stashDir);
1732
- const sourceResolved = await resolveReflectSource(options, stash, () => {
1733
- // No event emitted: this is a read-only preview, not a real invocation.
1734
- });
1735
- if ("failure" in sourceResolved) {
1736
- const { failure } = sourceResolved;
1083
+ const stash = options.stashDir ?? resolveStashDir();
1084
+ const source = await resolveReflectSource(options, stash, () => { });
1085
+ if ("failure" in source) {
1086
+ const { failure } = source;
1737
1087
  throw new UsageError((!failure.ok && failure.error) || `Reflect cannot preview ref "${ref}".`, "INVALID_FLAG_VALUE");
1738
1088
  }
1739
- const { assetContent, parsedRef } = sourceResolved;
1740
1089
  const { runnerSpec, engineName } = resolveReflectRunner(options);
1741
- const ctx = buildReflectRunContext({ options, stash, config: options.config ?? loadConfig(), runnerSpec });
1742
- const assetCtx = ctx.withFreshAssetMemo();
1743
- const sources = await gatherReflectPromptSources(options, stash, parsedRef, assetContent, assetCtx);
1744
- const canRunnerWriteFile = runnerSupportsFileWrite(runnerSpec);
1745
- // Same tmp-path synthesis a real dispatch would use (Issue A) — never
1746
- // written to, since this preview never runs the agent.
1747
- const draftFilePath = canRunnerWriteFile ? synthesizeReflectDraftPath(ref) : undefined;
1748
- const previewPromptInput = buildReflectPromptInput({
1090
+ const sources = await gatherReflectPromptSources(options, stash, source.parsedRef, source.assetContent);
1091
+ const { prompt } = buildReflectPromptText({
1749
1092
  options,
1750
- parsedRef,
1751
- assetContent,
1093
+ parsedRef: source.parsedRef,
1094
+ assetContent: source.assetContent,
1752
1095
  sources,
1753
1096
  runnerSpec,
1754
- draftFilePath,
1097
+ // The same tmp-path shape a dispatch would use; never written.
1098
+ draftFilePath: runnerSupportsFileWrite(runnerSpec) ? synthesizeReflectDraftPath(ref) : undefined,
1755
1099
  priorDraft: undefined,
1756
1100
  });
1757
- // #952 — mirror the real dispatch path's context-aware content budget (see
1758
- // computeReflectContentBudgetChars) so the preview shows the exact prompt
1759
- // reflect would send: an LLM engine with a large context window gets the
1760
- // full asset with no truncation marker, not the flat 12 000-char cap.
1761
- const contentBudgetChars = computeReflectContentBudgetChars(previewPromptInput, runnerSpec);
1762
- const { prompt } = buildReflectPrompt({
1763
- ...previewPromptInput,
1764
- ...(contentBudgetChars !== undefined ? { contentBudgetChars } : {}),
1765
- });
1766
1101
  return { ref, prompt, engine: engineName, engineKind: runnerSpec.kind };
1767
1102
  }
1768
1103
  export async function akmReflect(options = {}) {
1769
- const stash = resolveRunStashDir(options.stashDir);
1770
- // Build lazy event emitters. The invocation row is committed only after the
1771
- // canonical dispatch has validated symbolic credentials; deterministic
1772
- // pre-dispatch skips still emit it through emitReflectFailed.
1773
- const { emitInvoked: emitReflectInvoked, emitFailed: emitReflectFailed } = buildReflectEventEmitters(options);
1774
- // 2. Resolve target asset content (if a ref is supplied).
1775
- const sourceResolved = await resolveReflectSource(options, stash, emitReflectFailed);
1776
- if ("failure" in sourceResolved)
1777
- return sourceResolved.failure;
1778
- const { assetContent, parsedRef } = sourceResolved;
1779
- // 3. Resolve exactly one named engine. Standalone reflect uses --engine or
1780
- // defaults.engine; improve resolves its LLM-only strategy/process overlay.
1781
- // An incompatible explicit engine is an error and never falls through.
1104
+ const stash = options.stashDir ?? resolveStashDir();
1105
+ const { emitInvoked, emitFailed } = reflectEmitters(options);
1106
+ const source = await resolveReflectSource(options, stash, emitFailed);
1107
+ if ("failure" in source)
1108
+ return source.failure;
1109
+ const { assetContent, parsedRef } = source;
1782
1110
  const { config, activeStrategy, runnerSpec, engineName, notices: resolutionNotices } = resolveReflectRunner(options);
1783
- const executionNotices = new Map();
1784
- collectLoweringNotices(executionNotices, resolutionNotices);
1785
- const collectExecutionNotices = (notices) => collectLoweringNotices(executionNotices, notices);
1786
- let qualityJudgeSelection = resolveReflectQualityJudgeRunner(config, runnerSpec, isReflectQualityGateEnabled(activeStrategy), collectExecutionNotices);
1787
- const qualityGateSkippedNoJudge = qualityJudgeSelection.enabled && !qualityJudgeSelection.runner;
1788
- if (qualityGateSkippedNoJudge) {
1789
- warnOnce("reflect-quality-gate-no-judge", "Reflect proposal quality gate has no LLM configured to judge proposals (set defaults.llmEngine, or improve.strategies.<name>.processes.reflect.qualityGate.engine). Skipping the gate for this run; the proposal is queued for human review instead.");
1790
- qualityJudgeSelection = Object.freeze({ enabled: false, runner: undefined });
1111
+ const notices = noticeSet();
1112
+ notices.add(resolutionNotices);
1113
+ const run = { options, stash, config, runnerSpec, engineName, notices, emitInvoked, emitFailed };
1114
+ // Judge selection is frozen before dispatch so a missing judge credential fails first.
1115
+ const judgeWanted = (activeStrategy?.processes?.reflect?.qualityGate?.enabled ?? false) ||
1116
+ (activeStrategy?.processes?.distill?.qualityGate?.enabled ?? true);
1117
+ let judgeRunner;
1118
+ if (judgeWanted) {
1119
+ if (runnerIsLlm(runnerSpec)) {
1120
+ judgeRunner = runnerSpec;
1121
+ }
1122
+ else {
1123
+ const resolved = resolveImproveLlmExecution({ config, processName: "reflect_proposal_quality-judge" });
1124
+ if (resolved)
1125
+ notices.add(resolved.notices);
1126
+ judgeRunner = resolved?.runner;
1127
+ }
1791
1128
  }
1792
- const qualityJudgeRunner = qualityJudgeSelection.runner;
1793
- let generationLease;
1794
- let qualityJudgeLease;
1129
+ const skippedNoJudge = judgeWanted && !judgeRunner;
1130
+ if (skippedNoJudge) {
1131
+ warnOnce("reflect-quality-gate-no-judge", "Reflect proposal quality gate has no LLM configured to judge proposals (set defaults.llmEngine). Skipping the gate for this run; the proposal is queued for human review instead.");
1132
+ }
1133
+ preflightReflectDispatch(runnerSpec, notices.add);
1134
+ if (judgeRunner && judgeRunner !== runnerSpec)
1135
+ preflightReflectDispatch(judgeRunner, notices.add);
1136
+ const sources = await gatherReflectPromptSources(options, stash, parsedRef, assetContent);
1137
+ const agentEnv = options.eventSource === "improve" ? { AKM_EVENT_SOURCE: "improve" } : {};
1138
+ const sensitiveValues = collectDispatchSensitiveValues(runnerSpec, {
1139
+ ...(Object.keys(agentEnv).length > 0 ? { env: agentEnv } : {}),
1140
+ ...(options.runAgentOptions ?? {}),
1141
+ });
1142
+ const draftPaths = [];
1143
+ let result;
1144
+ let payload;
1795
1145
  try {
1796
- generationLease = acquireReflectDispatchLease(runnerSpec, collectExecutionNotices);
1797
- qualityJudgeLease =
1798
- qualityJudgeRunner === runnerSpec
1799
- ? generationLease
1800
- : qualityJudgeRunner
1801
- ? acquireReflectDispatchLease(qualityJudgeRunner, collectExecutionNotices)
1802
- : undefined;
1803
- // WI-9.10: RunContext, built only once config/runnerSpec exist so engine
1804
- // resolution's existing error-priority ordering is undisturbed (see
1805
- // buildReflectRunContext's docblock). D6: assetCtx is a fresh,
1806
- // per-invocation memo — readRelatedLessons below is its genuine
1807
- // content-read consumer.
1808
- const ctx = buildReflectRunContext({ options, stash, config, runnerSpec });
1809
- const assetCtx = ctx.withFreshAssetMemo();
1810
- // 4. Build the shared prompt inputs — feedback, hints, lessons, rejected
1811
- // proposals. These are stable across refinement iterations; only the
1812
- // `priorDraft` field changes per-iteration (R-1 / #372).
1813
- const sources = await gatherReflectPromptSources(options, stash, parsedRef, assetContent, assetCtx);
1814
- // 5. Spawn the agent — with the optional Self-Refine loop (R-1 / #372),
1815
- // extracted to {@link runReflectRefineIterations}.
1816
- const agentEnv = options.eventSource === "improve" ? { AKM_EVENT_SOURCE: "improve" } : {};
1817
- const sensitiveValues = collectDispatchSensitiveValues(runnerSpec, {
1818
- ...(Object.keys(agentEnv).length > 0 ? { env: agentEnv } : {}),
1819
- ...(options.runAgentOptions ?? {}),
1820
- });
1821
- const draftPathsToCleanup = [];
1822
- // `result` / `lastDraftPath` / `payload` are populated inside the try. Hoisted
1823
- // here so the post-try sections (R-3 ref guard, sanitizer, quality gate,
1824
- // createProposal) can use them after the drafts have been cleaned up.
1825
- let result = {};
1826
- let lastDraftPath;
1827
- let payload;
1828
- try {
1829
- const iterated = await runReflectRefineIterations({
1830
- options,
1831
- parsedRef,
1832
- assetContent,
1833
- sources,
1834
- runnerSpec,
1835
- lease: generationLease,
1836
- agentEnv,
1837
- draftPathsToCleanup,
1838
- onNotices: collectExecutionNotices,
1839
- });
1840
- emitReflectInvoked();
1841
- result = iterated.result;
1842
- lastDraftPath = iterated.lastDraftPath;
1843
- const finalResult = result;
1844
- if (!finalResult.ok) {
1845
- // B3: ENOENT / not-found gives an actionable hint.
1846
- if (isEnoentFailure(finalResult)) {
1847
- emitReflectFailed("spawn_failed", "enoent", options.ref, {
1848
- ...(finalResult.exitCode !== undefined ? { exitCode: finalResult.exitCode } : {}),
1849
- });
1850
- return {
1851
- ...failureEnvelope(finalResult, options.ref, engineName),
1852
- error: enoentHintMessage(runnerIsLlm(runnerSpec) ? engineName : runnerSpec.profile.bin),
1853
- ...reflectNoticeFields(executionNotices),
1854
- };
1855
- }
1856
- const envelope = failureEnvelope(finalResult, options.ref, engineName);
1857
- emitReflectFailed(envelope.reason, envelope.reason === "parse_error" ? "parse_error" : "agent_crash", options.ref, {
1858
- ...(envelope.exitCode !== null ? { exitCode: envelope.exitCode } : {}),
1859
- ...(reflectLlmTelemetry(finalResult) ?? {}),
1146
+ const iterated = await runReflectRefineIterations({ run, parsedRef, assetContent, sources, agentEnv, draftPaths });
1147
+ emitInvoked();
1148
+ result = iterated.result;
1149
+ if (!result.ok) {
1150
+ if (isEnoentFailure(result)) {
1151
+ emitFailed("spawn_failed", "enoent", options.ref, {
1152
+ ...(result.exitCode !== undefined ? { exitCode: result.exitCode } : {}),
1860
1153
  });
1861
- return { ...envelope, ...reflectNoticeFields(executionNotices) };
1154
+ return {
1155
+ ...baseFailureFields(result),
1156
+ schemaVersion: 2,
1157
+ ...(options.ref ? { ref: options.ref } : {}),
1158
+ engine: engineName,
1159
+ error: enoentHintMessage(runnerIsLlm(runnerSpec) ? engineName : runnerSpec.profile.bin),
1160
+ ...notices.fields(),
1161
+ };
1862
1162
  }
1863
- // Re-alias to `result` for the downstream code that references it.
1864
- result = finalResult;
1865
- const resolved = resolveReflectPayload({
1866
- result,
1867
- lastDraftPath,
1868
- sensitiveValues,
1869
- options,
1870
- engineName,
1871
- emitReflectFailed,
1872
- });
1873
- if ("failure" in resolved) {
1874
- return { ...resolved.failure, ...reflectNoticeFields(executionNotices) };
1875
- }
1876
- payload = resolved.payload;
1877
- }
1878
- catch (error) {
1879
- if (!(error instanceof ConfigError))
1880
- emitReflectInvoked();
1881
- throw error;
1882
- }
1883
- finally {
1884
- // Always remove tmp draft files — success, failure, or exception. Returns
1885
- // inside the try above trigger this block before the function exits. Code
1886
- // after this point uses the already-loaded `payload` and never touches the
1887
- // draft paths.
1888
- cleanupReflectDrafts(draftPathsToCleanup);
1889
- }
1890
- const unsafeContent = generatedContentRejection(payload.content, redactSensitiveText(payload.content, sensitiveValues));
1891
- if (unsafeContent) {
1892
- emitReflectFailed("parse_error", "parse_error", options.ref, {
1893
- ...(result.exitCode !== null ? { exitCode: result.exitCode } : {}),
1894
- });
1895
- return {
1163
+ const envelope = {
1164
+ ...baseFailureFields(result),
1896
1165
  schemaVersion: 2,
1897
- ok: false,
1898
- reason: "parse_error",
1899
- error: unsafeContent,
1900
1166
  ...(options.ref ? { ref: options.ref } : {}),
1901
1167
  engine: engineName,
1902
- exitCode: result.exitCode,
1903
- ...reflectNoticeFields(executionNotices),
1904
1168
  };
1169
+ emitFailed(envelope.reason, envelope.reason === "parse_error" ? "parse_error" : "agent_crash", options.ref, {
1170
+ ...(envelope.exitCode !== null ? { exitCode: envelope.exitCode } : {}),
1171
+ ...(reflectLlmTelemetry(result) ?? {}),
1172
+ });
1173
+ return { ...envelope, ...notices.fields() };
1905
1174
  }
1906
- const refFailure = validateReflectPayloadRef({
1907
- payload,
1908
- result,
1909
- options,
1910
- engineName,
1911
- emitReflectFailed,
1912
- executionNotices,
1913
- });
1914
- if (refFailure)
1915
- return refFailure;
1916
- const finalized = await finalizeReflectProposal({
1917
- payload,
1918
- assetContent,
1919
- result,
1920
- options,
1921
- engineName,
1922
- config,
1923
- qualityGateEnabled: qualityJudgeSelection.enabled,
1924
- qualityGateSkippedNoJudge,
1925
- qualityJudgeRunner,
1926
- qualityJudgeLease,
1927
- feedback: sources.feedback,
1928
- stash,
1929
- emitReflectFailed,
1930
- onNotices: collectExecutionNotices,
1931
- });
1932
- return { ...finalized, ...reflectNoticeFields(executionNotices) };
1175
+ const resolved = resolveReflectPayload(run, result, iterated.lastDraftPath, sensitiveValues);
1176
+ if ("failure" in resolved)
1177
+ return resolved.failure;
1178
+ payload = resolved.payload;
1179
+ }
1180
+ catch (error) {
1181
+ if (!(error instanceof ConfigError))
1182
+ emitInvoked();
1183
+ throw error;
1933
1184
  }
1934
1185
  finally {
1935
- if (qualityJudgeLease && qualityJudgeLease !== generationLease) {
1936
- disposeLoweredExecutionDispatchLease(qualityJudgeLease);
1186
+ for (const draftPath of draftPaths) {
1187
+ try {
1188
+ if (fs.existsSync(draftPath))
1189
+ fs.unlinkSync(draftPath);
1190
+ }
1191
+ catch {
1192
+ // best-effort
1193
+ }
1937
1194
  }
1938
- if (generationLease)
1939
- disposeLoweredExecutionDispatchLease(generationLease);
1940
1195
  }
1196
+ const unsafeContent = generatedContentRejection(payload.content, redactSensitiveText(payload.content, sensitiveValues));
1197
+ if (unsafeContent) {
1198
+ emitFailed("parse_error", "parse_error", options.ref, exitCodeMeta(result));
1199
+ return reflectFailure(run, result, "parse_error", unsafeContent, false);
1200
+ }
1201
+ // A retargeted proposal is refused (malformed refs are left to proposal validation).
1202
+ if (options.ref) {
1203
+ let retargeted = false;
1204
+ try {
1205
+ const expected = parseRefInput(options.ref);
1206
+ const actual = parseRefInput(payload.ref);
1207
+ retargeted = expected.type !== actual.type || expected.name !== actual.name;
1208
+ }
1209
+ catch {
1210
+ retargeted = false;
1211
+ }
1212
+ if (retargeted) {
1213
+ emitFailed("parse_error", "ref_mismatch", options.ref, {
1214
+ expectedRef: options.ref,
1215
+ actualRef: payload.ref,
1216
+ ...exitCodeMeta(result),
1217
+ ...(reflectLlmTelemetry(result) ?? {}),
1218
+ });
1219
+ return reflectFailure(run, result, "parse_error", `Agent retargeted proposal: expected ref "${options.ref}" but got "${payload.ref}". Proposal rejected to prevent silent ref hallucination.`, true);
1220
+ }
1221
+ }
1222
+ return finalizeReflectProposal({
1223
+ run,
1224
+ payload,
1225
+ assetContent,
1226
+ result,
1227
+ judge: { enabled: judgeWanted && !skippedNoJudge, skippedNoJudge, runner: judgeRunner },
1228
+ feedback: sources.feedback,
1229
+ });
1941
1230
  }