akm-cli 0.9.17-alpha.2 → 0.9.17-alpha.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (343) hide show
  1. package/CHANGELOG.md +756 -0
  2. package/dist/akm +94 -196
  3. package/dist/cli/shared.js +6 -2
  4. package/dist/cli.js +22 -9
  5. package/dist/commands/agent/agent-dispatch.js +1 -1
  6. package/dist/commands/command/command-execution.js +24 -62
  7. package/dist/commands/feedback-cli.js +0 -1
  8. package/dist/commands/health/accept-rate.js +2 -2
  9. package/dist/commands/health/checks.js +30 -75
  10. package/dist/commands/health/config-skew.js +38 -0
  11. package/dist/commands/health/egress.js +54 -0
  12. package/dist/commands/health/html-report.js +0 -38
  13. package/dist/commands/health/improve-metrics.js +123 -562
  14. package/dist/commands/health/plugin-staleness.js +53 -3
  15. package/dist/commands/health/renderers.js +12 -4
  16. package/dist/commands/health/report-view-model.js +11 -106
  17. package/dist/commands/health/types-improve.js +4 -19
  18. package/dist/commands/health/windows.js +64 -73
  19. package/dist/commands/health.js +122 -143
  20. package/dist/commands/improve/consolidate/chunking.js +25 -100
  21. package/dist/commands/improve/consolidate/sanitize.js +54 -149
  22. package/dist/commands/improve/consolidate.js +538 -1075
  23. package/dist/commands/improve/content-hash.js +16 -24
  24. package/dist/commands/improve/distill/content-repair.js +18 -100
  25. package/dist/commands/improve/distill-guards.js +20 -81
  26. package/dist/commands/improve/distill-promotion-policy.js +23 -243
  27. package/dist/commands/improve/distill.js +608 -1075
  28. package/dist/commands/improve/eligibility.js +126 -400
  29. package/dist/commands/improve/execution.js +3 -5
  30. package/dist/commands/improve/extract.js +487 -1046
  31. package/dist/commands/improve/feedback-valence.js +0 -25
  32. package/dist/commands/improve/improve-cli.js +29 -166
  33. package/dist/commands/improve/improve-result-file.js +10 -66
  34. package/dist/commands/improve/improve-strategies.js +12 -7
  35. package/dist/commands/improve/improve-usage-report.js +18 -64
  36. package/dist/commands/improve/improve.js +443 -1063
  37. package/dist/commands/improve/ledger.js +114 -0
  38. package/dist/commands/improve/locks.js +2 -8
  39. package/dist/commands/improve/loop-stages.js +459 -1172
  40. package/dist/commands/improve/memory/derived-ref.js +12 -77
  41. package/dist/commands/improve/memory/memory-belief.js +14 -118
  42. package/dist/commands/improve/memory/memory-improve.js +4 -3
  43. package/dist/commands/improve/outcome-loop.js +28 -156
  44. package/dist/commands/improve/planner.js +5 -10
  45. package/dist/commands/improve/preparation.js +851 -2339
  46. package/dist/commands/improve/proactive-maintenance.js +34 -101
  47. package/dist/commands/improve/reflect-noise.js +104 -280
  48. package/dist/commands/improve/reflect.js +621 -1367
  49. package/dist/commands/improve/salience.js +46 -232
  50. package/dist/commands/improve/session-asset.js +19 -100
  51. package/dist/commands/improve/stage.js +323 -0
  52. package/dist/commands/proposal/drain.js +251 -644
  53. package/dist/commands/proposal/proposal-cli.js +3 -18
  54. package/dist/commands/proposal/proposal-types.js +20 -41
  55. package/dist/commands/proposal/proposal.js +1 -2
  56. package/dist/commands/proposal/propose.js +134 -160
  57. package/dist/commands/proposal/repository.js +502 -1487
  58. package/dist/commands/proposal/validators/proposal-quality-validators.js +71 -174
  59. package/dist/commands/proposal/validators/proposal-validators.js +1 -1
  60. package/dist/commands/proposal/validators/proposals.js +13 -89
  61. package/dist/commands/read/curate.js +63 -413
  62. package/dist/commands/read/search-cli.js +16 -33
  63. package/dist/commands/read/search.js +17 -23
  64. package/dist/commands/read/show.js +2 -13
  65. package/dist/commands/sources/bundle-cli.js +25 -2
  66. package/dist/commands/sources/bundle-config-ops.js +7 -0
  67. package/dist/commands/sources/dangerous-env-audit.js +1 -2
  68. package/dist/commands/sources/info.js +2 -11
  69. package/dist/commands/sources/installed-stashes.js +197 -746
  70. package/dist/commands/sources/schema-repair.js +98 -129
  71. package/dist/commands/sources/source-add.js +62 -12
  72. package/dist/commands/sources/stash-cli.js +1 -1
  73. package/dist/commands/tasks/explain.js +10 -13
  74. package/dist/commands/tasks/tasks-cli.js +9 -8
  75. package/dist/commands/tasks/tasks.js +326 -930
  76. package/dist/commands/tasks/validate.js +42 -21
  77. package/dist/commands/workflow/plan.js +22 -29
  78. package/dist/commands/workflow-cli.js +4 -4
  79. package/dist/core/adapter/adapters/akm-adapter.js +0 -1
  80. package/dist/core/adapter/adapters/akm-lint.js +2 -3
  81. package/dist/core/adapter/adapters/akm-metadata.js +11 -12
  82. package/dist/core/adapter/adapters/akm-workflow-adapter.js +1 -1
  83. package/dist/core/adapter/execution-source.js +17 -29
  84. package/dist/core/asset/resolve-ref.js +1 -1
  85. package/dist/core/bundle-id.js +42 -5
  86. package/dist/core/bundle-rename.js +291 -0
  87. package/dist/core/config/config-io.js +1 -2
  88. package/dist/core/config/config-schema.js +1 -33
  89. package/dist/core/config/config-walker.js +1 -1
  90. package/dist/core/config/config.js +163 -68
  91. package/dist/core/config/legacy-source-shape-shim.js +38 -9
  92. package/dist/core/config/schema/embedding.js +20 -5
  93. package/dist/core/config/schema/engines.js +5 -0
  94. package/dist/core/config/schema/execution.js +1 -1
  95. package/dist/core/config/schema/experimental.js +1 -1
  96. package/dist/core/config/schema/improve-processes.js +21 -95
  97. package/dist/core/config/schema/improve.js +4 -42
  98. package/dist/core/config/schema/scheduler.js +12 -12
  99. package/dist/core/config/schema/search.js +6 -22
  100. package/dist/core/env-secret-ref.js +0 -1
  101. package/dist/core/errors.js +8 -9
  102. package/dist/core/file-lock.js +76 -173
  103. package/dist/core/logs-db.js +2 -2
  104. package/dist/core/paths.js +0 -27
  105. package/dist/core/redaction.js +109 -2
  106. package/dist/core/run-lock.js +2 -5
  107. package/dist/core/spawn-env.js +1 -1
  108. package/dist/core/state/migrations.js +108 -61
  109. package/dist/core/state-db-scope.js +2 -4
  110. package/dist/core/state-db.js +126 -692
  111. package/dist/core/type-presentation.js +1 -9
  112. package/dist/core/write-source.js +293 -1012
  113. package/dist/execution/input-contract.js +1 -1
  114. package/dist/execution/resolved-request.js +135 -689
  115. package/dist/execution/source.js +63 -257
  116. package/dist/execution/target-ref.js +1 -1
  117. package/dist/indexer/bundle-identity-guard.js +2 -2
  118. package/dist/indexer/db/graph-db.js +106 -46
  119. package/dist/indexer/ensure-index.js +44 -85
  120. package/dist/indexer/graph/graph-extraction.js +340 -562
  121. package/dist/indexer/graph/graph-related.js +130 -0
  122. package/dist/indexer/index-rebuild-lock.js +3 -11
  123. package/dist/indexer/index-writer-lock.js +8 -17
  124. package/dist/indexer/index-written-assets.js +139 -151
  125. package/dist/indexer/indexer.js +524 -846
  126. package/dist/indexer/materialize-embeddings.js +60 -397
  127. package/dist/indexer/passes/memory-inference.js +81 -90
  128. package/dist/indexer/passes/metadata.js +132 -200
  129. package/dist/indexer/read-preflight.js +0 -7
  130. package/dist/indexer/scan/doc-to-entry.js +1 -3
  131. package/dist/indexer/scan/drain-dir.js +1 -1
  132. package/dist/indexer/search/db-search.js +181 -590
  133. package/dist/indexer/search/fts-query.js +30 -41
  134. package/dist/indexer/search/ranking.js +28 -154
  135. package/dist/indexer/search/search-attribution.js +12 -32
  136. package/dist/indexer/search/search-fields.js +11 -15
  137. package/dist/indexer/search/search-hit-enrichers.js +54 -85
  138. package/dist/indexer/search/search-source.js +1 -4
  139. package/dist/indexer/usage/usage-events.js +2 -7
  140. package/dist/integrations/agent/engine-fallback.js +23 -40
  141. package/dist/integrations/agent/engine-resolution.js +93 -183
  142. package/dist/integrations/agent/execution.js +507 -0
  143. package/dist/integrations/agent/model-map.js +28 -156
  144. package/dist/integrations/agent/request-lowering.js +66 -141
  145. package/dist/integrations/agent/runner-dispatch.js +143 -321
  146. package/dist/integrations/agent/runner.js +54 -14
  147. package/dist/integrations/lockfile.js +53 -101
  148. package/dist/llm/embedders/deterministic.js +2 -3
  149. package/dist/llm/embedders/profile.js +71 -0
  150. package/dist/llm/embedders/remote.js +10 -15
  151. package/dist/llm/graph-extract.js +3 -12
  152. package/dist/llm/index-passes.js +3 -5
  153. package/dist/llm/memory-infer.js +1 -2
  154. package/dist/llm/metadata-enhance.js +1 -2
  155. package/dist/llm/structured-call.js +5 -24
  156. package/dist/output/generic-render.js +23 -11
  157. package/dist/output/html-render.js +13 -10
  158. package/dist/output/render-registry.js +3 -32
  159. package/dist/output/shapes/helpers.js +2 -34
  160. package/dist/output/shapes/passthrough.js +1 -9
  161. package/dist/{indexer/search/ranking-types.js → output/text/bundle-rename.js} +4 -1
  162. package/dist/output/text/command-format.js +60 -23
  163. package/dist/output/text/helpers.js +1 -1
  164. package/dist/output/text/migrate.js +5 -14
  165. package/dist/output/text/proposal-format.js +1 -2
  166. package/dist/output/text/workflow-format.js +0 -32
  167. package/dist/output/text.js +2 -0
  168. package/dist/registry/factory.js +4 -19
  169. package/dist/registry/network.js +66 -220
  170. package/dist/registry/providers/index.js +0 -2
  171. package/dist/registry/providers/skills-sh.js +3 -14
  172. package/dist/registry/providers/static-index.js +24 -26
  173. package/dist/registry/resolve.js +55 -131
  174. package/dist/scripts/akm-migrate-node.js +43937 -93313
  175. package/dist/scripts/akm-migrate.js +43697 -93071
  176. package/dist/setup/registry-stash-loader.js +4 -13
  177. package/dist/setup/semantic-assets.js +3 -44
  178. package/dist/setup/setup.js +1 -1
  179. package/dist/setup/steps/tasks.js +25 -15
  180. package/dist/sources/provider-factory.js +17 -18
  181. package/dist/sources/providers/filesystem.js +2 -3
  182. package/dist/sources/providers/git-install.js +7 -1
  183. package/dist/sources/providers/git-provider.js +0 -3
  184. package/dist/sources/providers/git-stash.js +0 -17
  185. package/dist/sources/providers/npm.js +2 -4
  186. package/dist/sources/providers/provider-utils.js +5 -10
  187. package/dist/sources/providers/website.js +0 -2
  188. package/dist/sources/snapshot-fetchers/website-ingest.js +1 -1
  189. package/dist/sources/website-url.js +2 -2
  190. package/dist/storage/database.js +9 -35
  191. package/dist/storage/repositories/improve-ledger-repository.js +168 -0
  192. package/dist/storage/repositories/index-connection.js +34 -70
  193. package/dist/storage/repositories/index-entries-repository.js +69 -111
  194. package/dist/storage/repositories/index-entry-mapper.js +1 -2
  195. package/dist/storage/repositories/index-entry-schema.js +83 -269
  196. package/dist/storage/repositories/index-fts-repository.js +86 -256
  197. package/dist/storage/repositories/index-llm-cache-repository.js +17 -0
  198. package/dist/storage/repositories/index-meta-repository.js +6 -4
  199. package/dist/storage/repositories/index-schema.js +192 -220
  200. package/dist/storage/repositories/index-utility-repository.js +8 -29
  201. package/dist/storage/repositories/index-vec-repository.js +133 -414
  202. package/dist/storage/repositories/outcome-repository.js +2 -1
  203. package/dist/storage/repositories/proposals-repository.js +35 -0
  204. package/dist/storage/repositories/registry-index-cache-repository.js +100 -0
  205. package/dist/storage/repositories/task-history-repository.js +26 -4
  206. package/dist/storage/repositories/workflow-runs-repository.js +53 -244
  207. package/dist/storage/sqlite-migrations.js +136 -0
  208. package/dist/storage/sqlite-pragmas.js +11 -9
  209. package/dist/storage/sqlite-transaction.js +170 -0
  210. package/dist/storage/state-db-integrity.js +34 -27
  211. package/dist/tasks/activation-config.js +134 -62
  212. package/dist/tasks/backends/cron.js +129 -277
  213. package/dist/tasks/backends/exec-utils.js +2 -5
  214. package/dist/tasks/backends/launchd.js +125 -745
  215. package/dist/tasks/backends/schtasks.js +101 -620
  216. package/dist/tasks/prepare/prepare-support.js +5 -15
  217. package/dist/tasks/prepare/prepare.js +0 -2
  218. package/dist/tasks/resolve-akm-bin.js +20 -79
  219. package/dist/tasks/run/attempt-lifecycle.js +0 -1
  220. package/dist/tasks/scheduler-binding.js +18 -238
  221. package/dist/tasks/scheduler-invocation.js +52 -52
  222. package/dist/tasks/scheduler-lock.js +53 -0
  223. package/dist/tasks/scheduler-sync.js +363 -679
  224. package/dist/tasks/source/parse-task-source.js +160 -10
  225. package/dist/tasks/source/task-source-v3-frozen.js +3 -4
  226. package/dist/tasks/source/task-to-v4.js +2 -2
  227. package/dist/workflows/authoring/authoring.js +3 -12
  228. package/dist/workflows/compile.js +211 -0
  229. package/dist/workflows/concurrency-policy.js +13 -74
  230. package/dist/workflows/exec/child-invocation.js +3 -17
  231. package/dist/workflows/exec/child-workflow.js +32 -141
  232. package/dist/workflows/exec/dispatch-redaction.js +13 -53
  233. package/dist/workflows/exec/environment.js +98 -0
  234. package/dist/workflows/exec/exec-unit.js +33 -140
  235. package/dist/workflows/exec/frozen-judge.js +7 -59
  236. package/dist/workflows/exec/native-executor.js +82 -341
  237. package/dist/workflows/exec/param-secrets.js +29 -47
  238. package/dist/workflows/exec/run-workflow.js +154 -387
  239. package/dist/workflows/exec/scheduler.js +9 -36
  240. package/dist/workflows/exec/step-work.js +127 -430
  241. package/dist/workflows/exec/unit-dispatch.js +11 -63
  242. package/dist/workflows/exec/unit-writer.js +8 -52
  243. package/dist/workflows/exec/worktree.js +39 -273
  244. package/dist/workflows/freeze/child-output-references.js +4 -15
  245. package/dist/workflows/freeze/environment.js +99 -92
  246. package/dist/workflows/freeze/freeze.js +172 -0
  247. package/dist/workflows/freeze/step-values.js +19 -21
  248. package/dist/workflows/freeze/targets/child-workflow.js +23 -92
  249. package/dist/workflows/freeze/targets/command.js +10 -33
  250. package/dist/workflows/freeze/targets/script.js +5 -12
  251. package/dist/workflows/freeze/targets/shell.js +3 -6
  252. package/dist/workflows/freeze/targets/task.js +25 -80
  253. package/dist/workflows/freeze/task-bindings.js +20 -67
  254. package/dist/workflows/{source-ir/github-yaml.js → github-yaml.js} +88 -206
  255. package/dist/workflows/ir/params.js +6 -51
  256. package/dist/workflows/ir/plan-hash.js +2 -34
  257. package/dist/workflows/parser.js +140 -43
  258. package/dist/{commands/improve/consolidate/types.js → workflows/plan.js} +2 -1
  259. package/dist/workflows/renderer.js +36 -69
  260. package/dist/workflows/resource-limits.js +12 -120
  261. package/dist/workflows/runtime/agent-identity.js +8 -40
  262. package/dist/workflows/runtime/run-outputs.js +3 -6
  263. package/dist/workflows/runtime/run-plan.js +316 -0
  264. package/dist/workflows/runtime/runs.js +48 -200
  265. package/dist/workflows/runtime/workflow-asset-loader.js +24 -57
  266. package/dist/workflows/{source-ir/semantics.js → source-semantics.js} +16 -20
  267. package/dist/workflows/validate-summary.js +2 -7
  268. package/docs/integration/bundling-akm.md +49 -42
  269. package/docs/migration/README.md +1 -0
  270. package/docs/migration/release-notes/0.9.17.md +41 -0
  271. package/docs/migration/v0.9.1-to-v0.9.2.md +19 -7
  272. package/docs/reference/cli.md +182 -125
  273. package/docs/reference/configuration.md +49 -56
  274. package/docs/reference/data-and-telemetry.md +19 -20
  275. package/docs/reference/tasks.md +86 -38
  276. package/docs/reference/workflow-schema.md +14 -18
  277. package/docs/reference/workflows.md +6 -9
  278. package/package.json +1 -1
  279. package/schemas/akm-config.json +87 -406
  280. package/dist/commands/health/advisories.js +0 -150
  281. package/dist/commands/health/metrics.js +0 -329
  282. package/dist/commands/health/surfaces.js +0 -102
  283. package/dist/commands/improve/anti-collapse.js +0 -83
  284. package/dist/commands/improve/collapse-detector.js +0 -432
  285. package/dist/commands/improve/consolidate/eligibility.js +0 -48
  286. package/dist/commands/improve/consolidate/merge.js +0 -146
  287. package/dist/commands/improve/distill/promote-memory.js +0 -329
  288. package/dist/commands/improve/distill/quality-gate.js +0 -500
  289. package/dist/commands/improve/memory/memory-contradiction-detect.js +0 -291
  290. package/dist/commands/improve/proposal-envelope.js +0 -31
  291. package/dist/commands/improve/run-context.js +0 -123
  292. package/dist/commands/improve/shared.js +0 -21
  293. package/dist/commands/improve/source-identity.js +0 -28
  294. package/dist/commands/improve/triage.js +0 -96
  295. package/dist/commands/proposal/drain-policies.js +0 -151
  296. package/dist/commands/sources/update-transaction.js +0 -220
  297. package/dist/core/action-contributors.js +0 -28
  298. package/dist/core/config/config-version-shim.js +0 -101
  299. package/dist/core/config/retired-experimental-keys-shim.js +0 -62
  300. package/dist/core/fs-txn.js +0 -405
  301. package/dist/core/lexical-score.js +0 -25
  302. package/dist/core/maintenance-barrier.js +0 -167
  303. package/dist/execution/executable-identity.js +0 -105
  304. package/dist/execution/guarded-source.js +0 -427
  305. package/dist/indexer/graph/graph-boost.js +0 -427
  306. package/dist/indexer/graph/graph-dedup.js +0 -95
  307. package/dist/indexer/search/name-match.js +0 -35
  308. package/dist/indexer/search/ranking-contributors.js +0 -515
  309. package/dist/indexer/walk/project-context.js +0 -192
  310. package/dist/integrations/agent/execution-cascade.js +0 -566
  311. package/dist/integrations/agent/execution-definitions.js +0 -202
  312. package/dist/integrations/agent/execution-lowering.js +0 -841
  313. package/dist/integrations/agent/execution-preparation.js +0 -98
  314. package/dist/integrations/agent/inline-execution.js +0 -74
  315. package/dist/registry/create-provider-registry.js +0 -29
  316. package/dist/registry/pinned-request-helper.js +0 -247
  317. package/dist/registry/pinned-transport.js +0 -717
  318. package/dist/sources/providers/index.js +0 -14
  319. package/dist/storage/engines/sqlite-migrations.js +0 -271
  320. package/dist/storage/repositories/canaries-repository.js +0 -107
  321. package/dist/storage/repositories/embedding-salvage-repository.js +0 -184
  322. package/dist/storage/repositories/registry-cache.js +0 -113
  323. package/dist/tasks/scheduler-sync-preview.js +0 -52
  324. package/dist/workflows/freeze/resolve-steps.js +0 -86
  325. package/dist/workflows/freeze/source-freeze.js +0 -64
  326. package/dist/workflows/ir/compile.js +0 -321
  327. package/dist/workflows/ir/environment-v4.js +0 -330
  328. package/dist/workflows/ir/freeze-v4.js +0 -153
  329. package/dist/workflows/ir/schema-v4.js +0 -745
  330. package/dist/workflows/ir/schema.js +0 -354
  331. package/dist/workflows/program/schema.js +0 -78
  332. package/dist/workflows/runtime/checkin.js +0 -57
  333. package/dist/workflows/runtime/plan-classifier.js +0 -196
  334. package/dist/workflows/runtime/unit-checkin.js +0 -45
  335. package/dist/workflows/runtime/unit-phases.js +0 -20
  336. package/dist/workflows/schema.js +0 -4
  337. package/dist/workflows/source-ir/compile.js +0 -200
  338. package/dist/workflows/source-ir/program.js +0 -50
  339. package/dist/workflows/source-ir/result.js +0 -26
  340. package/dist/workflows/source-ir/schema.js +0 -786
  341. package/dist/workflows/source-ir/triggers.js +0 -79
  342. package/dist/workflows/source-ir/uses.js +0 -40
  343. package/dist/workflows/validator.js +0 -60
@@ -2,24 +2,13 @@
2
2
  // License, v. 2.0. If a copy of the MPL was not distributed with this
3
3
  // file, You can obtain one at https://mozilla.org/MPL/2.0/.
4
4
  /**
5
- * `akm reflect [ref]` — proposal-producing agent command (#226).
5
+ * `akm reflect [ref]` — ask an engine for a revised asset and queue it as a
6
+ * proposal (`source: "reflect"`). Reflect never writes an asset: the proposal
7
+ * queue is the only path, `akm proposal accept` the bridge.
6
8
  *
7
- * Pipeline:
8
- *
9
- * 1. Emit `reflect_invoked` event at command entry (always, even on failure).
10
- * 2. If `ref` is provided, look the asset up via the FTS index and read its
11
- * content. Pull recent feedback (`feedback` events for that ref) and
12
- * lesson-lint findings to surface as schema hints.
13
- * 3. Build the prompt via {@link buildReflectPrompt}.
14
- * 4. Prepare, authorize, lower, and dispatch the frozen engine selection.
15
- * 5. Parse the agent's stdout into a {@link AgentProposalPayload}.
16
- * 6. Insert into the proposal queue via {@link createProposal} with
17
- * `source: "reflect"`.
18
- *
19
- * Failures are surfaced as structured envelopes carrying an
20
- * {@link AgentFailureReason} discriminant. Reflect NEVER calls
21
- * `writeAssetToSource` directly — the proposal queue is the only path to
22
- * a committed asset, and the `accept` flow is the bridge.
9
+ * Every invocation closes with one `reflect_completed` event; `reflect_invoked`
10
+ * is emitted once the dispatch has validated its credentials (deterministic
11
+ * pre-dispatch refusals still emit both).
23
12
  */
24
13
  import fs from "node:fs";
25
14
  import os from "node:os";
@@ -28,6 +17,7 @@ import { assembleAssetFromString, serializeFrontmatter } from "../../core/asset/
28
17
  import { parseFrontmatter } from "../../core/asset/frontmatter.js";
29
18
  import { conceptIdFromTypeName, parseRefInput } from "../../core/asset/resolve-ref.js";
30
19
  import { DESCRIPTION_MAX_CHARS, requiresDescription } from "../../core/authoring-rules.js";
20
+ import { resolveStashDir } from "../../core/common.js";
31
21
  import { loadConfig } from "../../core/config/config.js";
32
22
  import { generatedContentRejection, stripReflectPromptScaffolding } from "../../core/content-safety.js";
33
23
  import { ConfigError, UsageError } from "../../core/errors.js";
@@ -40,78 +30,45 @@ import { warn, warnOnce } from "../../core/warn.js";
40
30
  import { lookup } from "../../indexer/indexer.js";
41
31
  import { DEFAULT_LLM_TIMEOUT_MS } from "../../integrations/agent/config.js";
42
32
  import { fallbackAnnouncement, NO_ENGINE_MESSAGE_SUFFIX, NO_ENGINE_REMEDY, withEngineFallback, } from "../../integrations/agent/engine-fallback.js";
43
- import { acquireLoweredExecutionDispatchLease, dispatchLoweredExecutionRequest, disposeLoweredExecutionDispatchLease, lowerResolvedExecutionRequest, lowerResolvedExecutionRequestWithRunner, } from "../../integrations/agent/execution-lowering.js";
44
- import { prepareInlineExecution, prepareInlineExecutionWithRunner } from "../../integrations/agent/inline-execution.js";
33
+ import { buildExecution, resolveExecution } from "../../integrations/agent/execution.js";
45
34
  import { buildReflectOutputRepairPrompt, buildReflectPrompt, extractDraftConfidence, parseAgentProposalPayload, REFLECT_CONTENT_CAP, REFLECT_TRUNCATION_MARKER, } from "../../integrations/agent/prompts.js";
46
35
  import { runnerIsLlm, runnerSupportsFileWrite } from "../../integrations/agent/runner.js";
47
- import { collectDispatchSensitiveValues } from "../../integrations/agent/runner-dispatch.js";
36
+ import { assertRunnerCredentials, collectDispatchSensitiveValues, runExecution, } from "../../integrations/agent/runner-dispatch.js";
48
37
  import { isJsonSchemaKnownUnsupported, LlmCallError } from "../../llm/client.js";
49
- import { callStructured } from "../../llm/structured-call.js";
50
38
  import { baseFailureFields, enoentHintMessage, isEnoentFailure } from "../agent/agent-support.js";
51
- import { isStaleTargetRejection } from "../proposal/proposal-types.js";
52
- import { isProposalSkipped, listProposalsReadOnly, recordGateDecision, } from "../proposal/repository.js";
53
39
  import { checkReflectSize, isValidDescription } from "../proposal/validators/proposal-quality-validators.js";
54
40
  import { CHARS_PER_TOKEN, DEFAULT_CONTEXT_LENGTH_TOKENS } from "./consolidate/chunking.js";
55
41
  import { deriveLessonRef } from "./distill.js";
56
- import { runReflectQualityJudge } from "./distill/quality-gate.js";
57
42
  import { findAssetFilePath } from "./eligibility.js";
58
43
  import { resolveImproveLlmExecution } from "./execution.js";
59
- import { emitProposal } from "./proposal-envelope.js";
60
- import { classifyReflectChange } from "./reflect-noise.js";
61
- import { createRunContext, resolveRunStashDir } from "./run-context.js";
62
- import { MAX_REJECTED_PROPOSALS } from "./shared.js";
63
- import { durableImproveRef, improveStateReadRefs } from "./source-identity.js";
64
- function collectLoweringNotices(target, notices) {
65
- for (const notice of notices)
66
- target.set(JSON.stringify(notice), notice);
67
- }
68
- function reflectNoticeFields(notices) {
69
- return notices.size > 0 ? { notices: Object.freeze([...notices.values()]) } : {};
70
- }
44
+ import { recordLedgerAttempt } from "./ledger.js";
45
+ import { classifyReflectChange, splitFrontmatter } from "./reflect-noise.js";
46
+ import { callStage, mintProposal, noticeSet, rejectedProposalContext, runReflectQualityJudge, } from "./stage.js";
71
47
  const MAX_FEEDBACK_LINES = 10;
72
48
  const MAX_GLOBAL_FEEDBACK_LINES = 20;
73
- /**
74
- * Pull recent `feedback` events from events.jsonl. When `ref` is present we
75
- * scope to that asset; otherwise we surface the most recent feedback across
76
- * all assets so `akm reflect` can operate in a general "review recent
77
- * signals" mode. Best-effort — a missing or empty events stream returns `[]`.
78
- */
79
- export function readOnlyEventsContext(ctx) {
49
+ function readOnlyEventsContext(ctx) {
80
50
  return ctx?.db ? ctx : { ...(ctx ?? {}), readOnly: true };
81
51
  }
52
+ /** Recent `feedback` lines for `ref` (or across all assets without one). Best-effort. */
82
53
  function readRecentFeedback(ref, eventsCtx) {
83
54
  try {
84
55
  const events = readEvents({ type: "feedback", ...(ref ? { ref } : {}) }, readOnlyEventsContext(eventsCtx)).events;
85
- const lines = [];
86
- const limit = ref ? MAX_FEEDBACK_LINES : MAX_GLOBAL_FEEDBACK_LINES;
87
- for (const event of events.slice(-limit)) {
88
- const md = (event.metadata ?? {});
56
+ return events.slice(-(ref ? MAX_FEEDBACK_LINES : MAX_GLOBAL_FEEDBACK_LINES)).map((event) => {
57
+ const md = event.metadata ?? {};
89
58
  const signal = typeof md.signal === "string" ? md.signal : "?";
90
59
  const note = typeof md.reason === "string" ? md.reason : typeof md.note === "string" ? md.note : "";
91
60
  const details = note ? `[${signal}] ${note}` : `[${signal}]`;
92
- lines.push(!ref && event.ref ? `${event.ref} ${details}` : details);
93
- }
94
- return lines;
61
+ return !ref && event.ref ? `${event.ref} ${details}` : details;
62
+ });
95
63
  }
96
64
  catch {
97
65
  return [];
98
66
  }
99
67
  }
100
68
  /**
101
- * Asset types that reflect is allowed to operate on.
102
- *
103
- * Reflect's canonical output shape is `frontmatter + markdown body`. Running it
104
- * against types whose on-disk form is NOT markdown (executable scripts, env files
105
- * env files, YAML tasks) blindly prepends `---\n…\n---\n` to the asset and
106
- * breaks the runtime contract — for example a `.ts` script with a YAML preamble
107
- * is a TypeScript syntax error.
108
- *
109
- * Whitelisting (rather than blacklisting) keeps the door closed by default as
110
- * new asset types are registered. To allow a custom registered type, extend
111
- * this set explicitly.
112
- *
113
- * Observed regression: proposal `8737ab63` (May 2026) prepended frontmatter to
114
- * a `.ts` script file via reflect. This whitelist prevents that.
69
+ * Types reflect may rewrite: its output is frontmatter + markdown, which would
70
+ * break a script or env file. Another type is allowed only when its current
71
+ * content already has that shape; secrets are never read.
115
72
  */
116
73
  export const REFLECT_ALLOWED_TYPES = new Set([
117
74
  "knowledge",
@@ -123,138 +80,68 @@ export const REFLECT_ALLOWED_TYPES = new Set([
123
80
  "workflow",
124
81
  ]);
125
82
  const REFLECT_REFUSED_TYPES = new Set(["secret"]);
126
- function isReflectableSourceShape(content) {
127
- return parseFrontmatter(content).frontmatter !== null;
128
- }
129
- /**
130
- * Identity / structural frontmatter fields the LLM is NEVER allowed to change.
131
- *
132
- * Renaming `name` on a skill silently breaks ref resolution because the ref is
133
- * derived from the on-disk path. Similar reasoning for `ref`, `id`, `slug`,
134
- * and `type`. The post-processor below restores any of these fields if the
135
- * LLM tried to rewrite them.
136
- *
137
- * Observed regression: proposal `26941510` (May 2026) renamed
138
- * `skills/openpalm-stack-diagnostics`'s `name` field to `"diagnostic-checklist"`.
139
- */
83
+ /** Identity fields the model may never change (a renamed `name` breaks ref resolution). */
140
84
  const PROTECTED_FRONTMATTER_FIELDS = new Set(["name", "ref", "id", "slug", "type"]);
141
85
  /**
142
- * Read the last 1–3 archived rejected proposals for a given ref from the
143
- * proposal store. Returns `[]` when the proposals store is absent (not yet
144
- * created) or the ref is undefined — `listProposalsReadOnly` already handles
145
- * that case; a genuine read failure propagates instead of being swallowed,
146
- * since silently dropping this Reflexion-style context risks re-proposing
147
- * content that was already rejected (arXiv:2303.11366).
148
- */
149
- function readRejectedProposals(stash, ref, proposalsCtx) {
150
- if (!ref)
151
- return [];
152
- // Exclude the drain's stale-target auto-rejects (STALE, R20): those are a
153
- // procedural refusal (the target changed after mint), not a judgement on
154
- // the content, and would mislead this Reflexion-style "don't repeat this"
155
- // context.
156
- return listProposalsReadOnly(stash, { ref, status: "rejected", includeArchive: true }, proposalsCtx)
157
- .filter((p) => !isStaleTargetRejection(p))
158
- .sort((a, b) => new Date(b.updatedAt ?? 0).getTime() - new Date(a.updatedAt ?? 0).getTime())
159
- .slice(0, MAX_REJECTED_PROPOSALS)
160
- .map((p) => ({
161
- ref: p.ref,
162
- reason: p.review?.reason ?? "no reason given",
163
- // #legacy: `changes` is empty for pre-existing rows (storedToChanges),
164
- // which makes `proposalContent` throw before reflect dispatch even
165
- // runs. `payload.content` is populated for every row regardless, so
166
- // read the preview from there instead.
167
- contentPreview: p.payload.content.slice(0, 500),
168
- }));
169
- }
170
- /**
171
- * Synthesize a tmp draft-file path for the agent/sdk file-write contract.
172
- *
173
- * Mirrors the draft-path synthesis in `src/commands/proposal/propose.ts` —
174
- * when the runner is agent-CLI or the OpenCode SDK, we instruct the agent to
175
- * write the proposal body directly to this file instead of inlining it in
176
- * JSON on stdout. This bypasses two
177
- * known failure modes for long assets: (a) ARG_MAX truncation on prompt
178
- * round-trips through fenced JSON, and (b) embedded-JSON parser brittleness
179
- * on multi-KB bodies (e.g. the `knowledge/systems/KOKORO_USAGE_GUIDE` 8.4KB
180
- * payload that produced 4/5 `parse_error` in May 2026 reflect validation).
181
- *
182
- * The path lives under {@link os.tmpdir} and embeds the (sanitized) ref +
183
- * timestamp + random suffix so concurrent reflect calls cannot collide.
184
- *
185
- * The LLM HTTP runner cannot use this path because chat-completion transport
186
- * has no filesystem access.
86
+ * A fresh tmp path per iteration for the agent/SDK file-write contract (long
87
+ * bodies are written to a file instead of fenced JSON on stdout). The direct
88
+ * LLM runner has no filesystem and never gets one.
187
89
  */
188
90
  function synthesizeReflectDraftPath(ref) {
189
91
  const safeRef = (ref ?? "no-ref").replace(/[^a-z0-9_-]/gi, "_");
190
92
  const rand = Math.random().toString(36).slice(2, 8);
191
93
  return path.join(os.tmpdir(), `akm-reflect-${safeRef}-${Date.now()}-${rand}.md`);
192
94
  }
193
- /**
194
- * Heuristic check that the agent honoured the file-write contract.
195
- * The contract instructs the agent to emit a single `DRAFT_WRITTEN` line on
196
- * stdout when it has finished writing the draft file. Some agents print
197
- * additional log lines; we match anywhere in the captured stdout.
198
- */
199
- function stdoutSignalsDraftWritten(stdout) {
200
- if (!stdout)
201
- return false;
202
- return /\bDRAFT_WRITTEN\b/.test(stdout);
203
- }
204
- /**
205
- * Build schema/lint hints for the prompt. For lesson refs, run the lesson
206
- * lint over the current content and surface any findings — they are a
207
- * concrete starting point for the agent's revision.
208
- */
95
+ /** Lesson lint findings for the prompt: a concrete starting point for the revision. */
209
96
  function buildSchemaHints(type, content) {
210
- if (!content)
97
+ if (!content || type !== "lesson")
211
98
  return [];
212
- if (type !== "lesson")
213
- return [];
214
- const report = lintLessonContent(content, "reflect");
215
- return report.findings.map((f) => `[${f.kind}] ${f.message}`);
216
- }
217
- function hasRelatedSkillSource(content, skillRef) {
218
- const parsed = parseFrontmatter(content);
219
- const sources = parsed.data.sources;
220
- return Array.isArray(sources) && sources.some((source) => typeof source === "string" && source.trim() === skillRef);
99
+ return lintLessonContent(content, "reflect").findings.map((f) => `[${f.kind}] ${f.message}`);
221
100
  }
222
- async function readRelatedLessons(ctx, stash, ref, parsedRef, itemRef) {
101
+ /**
102
+ * Lessons related to a skill: its derived lesson, lessons distilled from it,
103
+ * and lessons citing it in `sources`. Without independent feedback on the skill,
104
+ * lessons reflect itself produced are dropped so its own output is not fed
105
+ * back as evidence.
106
+ */
107
+ async function readRelatedLessons(stash, ref, parsedRef, itemRef, eventsCtx) {
223
108
  if (parsedRef.type !== "skill")
224
109
  return [];
110
+ const cache = new Map();
111
+ const read = (filePath) => {
112
+ const key = path.resolve(filePath);
113
+ const cached = cache.get(key) ?? fs.readFileSync(filePath, "utf8");
114
+ cache.set(key, cached);
115
+ return cached;
116
+ };
225
117
  const related = new Map();
226
118
  const derivedLessonRef = deriveLessonRef(ref);
227
119
  const candidateRefs = new Set([derivedLessonRef]);
228
120
  const derivedLessonPath = path.join(stash, "lessons", `${parseRefInput(derivedLessonRef).name}.md`);
229
121
  if (fs.existsSync(derivedLessonPath)) {
230
- // WI-9.10: genuine content read — routed through the per-invocation asset
231
- // memo (D6). No write to this same path happens later in this invocation,
232
- // so memoizing is safe (see run-context.ts's D6 seam docblock).
233
- related.set(derivedLessonRef, { ref: derivedLessonRef, content: ctx.readAsset(derivedLessonPath) });
122
+ related.set(derivedLessonRef, { ref: derivedLessonRef, content: read(derivedLessonPath) });
234
123
  }
235
124
  try {
236
- // Match events using the candidate's single durable state key.
237
- const distillInvokedKeys = new Set(improveStateReadRefs(ref, itemRef));
238
- const feedbackEvents = readEvents({ type: "distill_invoked" }, readOnlyEventsContext(ctx.eventsCtx)).events.filter((event) => event.ref !== undefined && distillInvokedKeys.has(event.ref));
239
- for (const event of feedbackEvents) {
125
+ const keys = new Set([itemRef ?? ref]);
126
+ for (const event of readEvents({ type: "distill_invoked" }, readOnlyEventsContext(eventsCtx)).events) {
127
+ if (event.ref === undefined || !keys.has(event.ref))
128
+ continue;
240
129
  const proposalRef = typeof event.metadata?.proposalRef === "string" ? event.metadata.proposalRef : undefined;
241
130
  if (proposalRef && lenientRefType(proposalRef) === "lesson")
242
131
  candidateRefs.add(proposalRef);
243
132
  }
244
133
  }
245
134
  catch {
246
- // Best effort only.
135
+ // best-effort
247
136
  }
248
137
  for (const candidateRef of candidateRefs) {
249
138
  try {
250
- const filePath = await findAssetFilePath(durableImproveRef(candidateRef), stash);
251
- if (!filePath || !fs.existsSync(filePath))
252
- continue;
253
- const content = ctx.readAsset(filePath);
254
- related.set(candidateRef, { ref: candidateRef, content });
139
+ const filePath = await findAssetFilePath(candidateRef, stash);
140
+ if (filePath && fs.existsSync(filePath))
141
+ related.set(candidateRef, { ref: candidateRef, content: read(filePath) });
255
142
  }
256
143
  catch {
257
- // Index miss is non-fatal.
144
+ // An index miss is not fatal.
258
145
  }
259
146
  }
260
147
  try {
@@ -263,83 +150,40 @@ async function readRelatedLessons(ctx, stash, ref, parsedRef, itemRef) {
263
150
  for (const fileName of fs.readdirSync(lessonsDir)) {
264
151
  if (!fileName.endsWith(".md"))
265
152
  continue;
266
- const content = ctx.readAsset(path.join(lessonsDir, fileName));
267
- if (!hasRelatedSkillSource(content, ref))
153
+ const content = read(path.join(lessonsDir, fileName));
154
+ const sources = parseFrontmatter(content).data.sources;
155
+ if (!Array.isArray(sources) || !sources.some((s) => typeof s === "string" && s.trim() === ref))
268
156
  continue;
269
- const lessonName = fileName.slice(0, -3);
270
- const lessonRef = conceptIdFromTypeName("lesson", lessonName);
271
- if (!related.has(lessonRef)) {
157
+ const lessonRef = conceptIdFromTypeName("lesson", fileName.slice(0, -3));
158
+ if (!related.has(lessonRef))
272
159
  related.set(lessonRef, { ref: lessonRef, content });
273
- }
274
160
  }
275
161
  }
276
162
  }
277
163
  catch {
278
- // Best effort only.
164
+ // best-effort
279
165
  }
280
- // R-4 / #373: Filter out lessons with `derived_from_reflect: true` unless
281
- // independent feedback exists for the skill. This prevents the echo-chamber
282
- // risk where reflect-output lessons feed back into the next reflect pass as
283
- // "independent" evidence, amplifying their own prior outputs over time.
284
- //
285
- // ExpeL arXiv:2308.10144: rules need differential evidence from independent
286
- // sources (success vs failure traces). A lesson that only ever appeared from
287
- // reflect-internal signals has no such differential signal.
288
- //
289
- // "Independent feedback" = any usage_events "feedback" events for the skill
290
- // ref itself, indicating a human or external system rated the skill.
291
- let hasIndependentFeedback = false;
166
+ let hasIndependentFeedback = true;
292
167
  try {
293
- const feedbackEventsForSkill = readEvents({ type: "feedback", ref }, readOnlyEventsContext(ctx.eventsCtx)).events;
294
- hasIndependentFeedback = feedbackEventsForSkill.length > 0;
168
+ hasIndependentFeedback = readEvents({ type: "feedback", ref }, readOnlyEventsContext(eventsCtx)).events.length > 0;
295
169
  }
296
170
  catch {
297
- // Best effort — if we can't check, allow all lessons through.
298
- hasIndependentFeedback = true;
171
+ // Unknown: keep every lesson.
299
172
  }
300
173
  if (!hasIndependentFeedback) {
301
- // No independent feedback: exclude all reflect-derived lessons to prevent
302
- // echo-chamber amplification.
303
- for (const [lessonRef, lesson] of related.entries()) {
174
+ for (const [lessonRef, lesson] of related) {
304
175
  try {
305
- const lessonFm = parseFrontmatter(lesson.content);
306
- if (lessonFm.data.derived_from_reflect === true) {
176
+ if (parseFrontmatter(lesson.content).data.derived_from_reflect === true)
307
177
  related.delete(lessonRef);
308
- }
309
178
  }
310
179
  catch {
311
- // If we can't parse the frontmatter, keep the lesson (safe default).
180
+ // Unparseable frontmatter: keep it.
312
181
  }
313
182
  }
314
183
  }
315
184
  return [...related.values()];
316
185
  }
317
- /**
318
- * Returns true only when `stdout` is a recognised AKM proposal-skip signal.
319
- *
320
- * Accepted forms are structured JSON: `{ skipped: true }` or
321
- * `{ reason: "<known-skip-reason>" }`.
322
- */
323
- function isStructuredCooldownSignal(stdout) {
324
- try {
325
- const parsed = JSON.parse(stdout.trim());
326
- if (parsed?.skipped === true)
327
- return true;
328
- if (typeof parsed?.reason === "string" && ["fingerprint_match", "rejection_backoff"].includes(parsed.reason))
329
- return true;
330
- }
331
- catch {
332
- // Non-JSON stdout is never a structured cooldown signal.
333
- }
334
- return false;
335
- }
336
- /**
337
- * Best-effort asset type for a maybe-ref string, in the 0.9.0 `[bundle//]conceptId`
338
- * grammar (`""` when it does not parse). Replaces the pre-0.9.0 `ref.split(":")[0]`
339
- * type-extraction, which yielded the whole conceptId (`lessons/my-lesson`) instead
340
- * of the type once refs stopped carrying a `type:` prefix (ref-grammar decision
341
- * D-R3). Lenient by design — the callers degrade gracefully on an empty type.
342
- */
186
+ /** The asset type of a maybe-ref, or `""` when it does not parse. */
343
187
  function lenientRefType(ref) {
344
188
  if (!ref)
345
189
  return "";
@@ -351,81 +195,43 @@ function lenientRefType(ref) {
351
195
  }
352
196
  }
353
197
  /**
354
- * Split a markdown blob into `[frontmatterText, bodyText]`.
355
- *
356
- * Returns `[null, raw]` when the blob does not start with a frontmatter block.
357
- */
358
- function splitFrontmatter(raw) {
359
- const m = raw.match(/^---\r?\n([\s\S]*?)\r?\n---\r?\n?([\s\S]*)$/);
360
- if (!m)
361
- return { fmText: null, body: raw };
362
- return { fmText: m[1], body: m[2] };
363
- }
364
- /**
365
- * Strip an LLM-appended duplicate frontmatter block from a body string.
366
- *
367
- * When the LLM echoes the original source file verbatim after its rewrite,
368
- * the resulting body contains a second `---...---` YAML block. We detect it
369
- * by requiring BOTH a balanced fence (opening + closing `---`) AND YAML-like
370
- * `key: value` content inside, so legitimate Markdown thematic breaks and
371
- * code-fence examples are never truncated.
198
+ * Cut a duplicate frontmatter block the model appended after its rewrite.
199
+ * Requires a balanced fence AND `key:` lines so thematic breaks survive.
372
200
  */
373
201
  function stripAppendedFrontmatter(body) {
374
- const fencePattern = /\n---\r?\n([\s\S]*?)\n---\r?\n/;
375
- const match = body.match(fencePattern);
376
- if (!match)
377
- return body;
378
- // Only strip when the captured block looks like YAML frontmatter.
379
- if (!/^\w[\w-]*:/m.test(match[1]))
202
+ const match = body.match(/\n---\r?\n([\s\S]*?)\n---\r?\n/);
203
+ if (!match || !/^\w[\w-]*:/m.test(match[1]))
380
204
  return body;
381
205
  return body.slice(0, body.indexOf(match[0])).replace(/\s+$/, "");
382
206
  }
383
207
  /**
384
- * #636 — deterministically derive a valid `description` from an asset's existing
385
- * metadata when one is missing. Sources, in priority order: the `title:`
386
- * frontmatter field, the first `# Heading` in the (proposed or source) body, and
387
- * the first sentence of the opening body paragraph. The candidate is normalized
388
- * (whitespace collapsed, trailing punctuation/markdown stripped, clamped to the
389
- * description max) and only returned if it PASSES `isValidDescription` — so this
390
- * never produces a heading-fragment, truncated, or otherwise gate-failing value.
391
- * Returns `undefined` when nothing usable can be derived (caller leaves the
392
- * proposal as-is rather than fabricating prose).
393
- *
394
- * This is intentionally deterministic and lives in the reflect proposal-build
395
- * path — it does NOT touch the validators or the promote-time repair.
208
+ * A description derived from existing metadata (title, first heading, first
209
+ * prose sentence) that passes `isValidDescription`, or `undefined`. Never
210
+ * free-form invention.
396
211
  */
397
212
  function deriveDescriptionFromAsset(title, proposedBody, sourceBody, targetRef) {
398
- // Each candidate is tagged with its kind. A title or `# Heading` is a bare
399
- // fragment ("Paged.js — Named Page") that reads poorly as a description even
400
- // when it is long enough to pass the length gate, so for those we prefer the
401
- // padded sentence form. A prose sentence is already a sentence, so it is used
402
- // as-is (padding it would double-wrap an already-complete sentence).
403
213
  const candidates = [];
404
- // 1. title: frontmatter
405
214
  if (typeof title === "string" && title.trim())
406
215
  candidates.push({ text: title.trim(), kind: "fragment" });
407
- // 2. first `# Heading` (proposed body first, then source body)
408
216
  for (const body of [proposedBody, sourceBody]) {
409
- const headingMatch = body.match(/^#{1,6}\s+(.+?)\s*$/m);
410
- if (headingMatch?.[1])
411
- candidates.push({ text: headingMatch[1].trim(), kind: "fragment" });
217
+ const heading = body.match(/^#{1,6}\s+(.+?)\s*$/m)?.[1];
218
+ if (heading)
219
+ candidates.push({ text: heading.trim(), kind: "fragment" });
412
220
  }
413
- // 3. first sentence of the opening prose paragraph (skip headings, fences,
414
- // list markers, blockquotes — those are not prose).
415
221
  for (const body of [proposedBody, sourceBody]) {
416
- const firstSentence = firstProseSentence(body);
417
- if (firstSentence)
418
- candidates.push({ text: firstSentence, kind: "prose" });
222
+ const sentence = firstProseSentence(body);
223
+ if (sentence)
224
+ candidates.push({ text: sentence, kind: "prose" });
419
225
  }
420
226
  for (const { text, kind } of candidates) {
421
- const normalized = normalizeDescriptionCandidate(text);
227
+ const normalized = text
228
+ .replace(/`/g, "")
229
+ .replace(/^[#>*\-\s]+/, "")
230
+ .replace(/\s+/g, " ")
231
+ .trim();
422
232
  if (!normalized)
423
233
  continue;
424
- // For a title/heading fragment, try the padded sentence form FIRST so the
425
- // result reads as a sentence rather than a bare fragment — a short but valid
426
- // title like "Paged.js — Named Page" (21 chars) would otherwise be returned
427
- // verbatim. Fall back to the bare form only if the padded form fails the
428
- // gate. A prose candidate is already a sentence, so it is used as-is.
234
+ // A bare title/heading reads poorly as a description: prefer the sentence form.
429
235
  const variants = kind === "fragment" ? [`Reference notes on ${normalized}.`, normalized] : [normalized];
430
236
  for (const v of variants) {
431
237
  const clamped = v.length > DESCRIPTION_MAX_CHARS ? v.slice(0, DESCRIPTION_MAX_CHARS).trimEnd() : v;
@@ -435,51 +241,21 @@ function deriveDescriptionFromAsset(title, proposedBody, sourceBody, targetRef)
435
241
  }
436
242
  return undefined;
437
243
  }
438
- /** Extract the first prose sentence from a markdown body, or `""` if none. */
439
244
  function firstProseSentence(body) {
440
245
  for (const rawLine of body.split(/\r?\n/)) {
441
246
  const line = rawLine.trim();
442
- if (!line)
443
- continue;
444
- if (/^(#{1,6}\s|```|~~~|[-*+]\s|\d+\.\s|>|\||<!--)/.test(line))
247
+ if (!line || /^(#{1,6}\s|```|~~~|[-*+]\s|\d+\.\s|>|\||<!--)/.test(line))
445
248
  continue;
446
- const sentenceMatch = line.match(/^(.+?[.!?])(\s|$)/);
447
- return (sentenceMatch?.[1] ?? line).trim();
249
+ return (line.match(/^(.+?[.!?])(\s|$)/)?.[1] ?? line).trim();
448
250
  }
449
251
  return "";
450
252
  }
451
- /** Normalize a description candidate: strip markdown markers, collapse space. */
452
- function normalizeDescriptionCandidate(raw) {
453
- return raw
454
- .replace(/`/g, "")
455
- .replace(/^[#>*\-\s]+/, "")
456
- .replace(/\s+/g, " ")
457
- .trim();
458
- }
459
253
  /**
460
- * Reflect post-processor — enforces the safety rails described at the top of
461
- * this file:
462
- *
463
- * 1. Restore the source frontmatter so reflect never strips load-bearing
464
- * `description`, `when_to_use`, `tags`, etc. The LLM is only allowed to
465
- * change the markdown body. Frontmatter fields proposed by the LLM are
466
- * treated as a *merge on top* of the source — concrete field renames /
467
- * identity changes (`name`, `ref`, `id`, `slug`, `type`) are reverted.
468
- * 2. Reject responses that shrink or expand the body past the configured
469
- * ratio thresholds, when the source body is large enough to be reliable.
470
- * 3. Drop any leading `---` frontmatter block the LLM produced inside the
471
- * body — the prompt asks it to emit body only, and a stray YAML preamble
472
- * on top of an executable-typed asset is dangerous.
473
- *
474
- * Caller branches:
475
- * - On `reject`: surface as a failure with the reported reason.
476
- * - Otherwise: substitute `content` (and optional `frontmatter`) into the
477
- * proposal payload.
478
- *
479
- * Source-less / new-asset case (`sourceContent === undefined`): we still strip
480
- * the LLM's frontmatter block from `content` and re-emit a clean block built
481
- * from `payload.frontmatter` so identity fields can be enforced. Size guard
482
- * is skipped because there is no source to compare against.
254
+ * Reflect's content rails: the source frontmatter is restored and the model's
255
+ * frontmatter merged on top except identity fields; a stray or appended
256
+ * frontmatter block and echoed run-only guidance are stripped; a missing
257
+ * required description is derived deterministically; a body outside the size
258
+ * ratios or echoing the truncation notice is flagged for review.
483
259
  */
484
260
  export function sanitizeReflectPayload(payload, sourceContent, targetRef) {
485
261
  const warnings = [];
@@ -488,13 +264,9 @@ export function sanitizeReflectPayload(payload, sourceContent, targetRef) {
488
264
  : { fmText: null, body: "" };
489
265
  const sourceFm = sourceFmText !== null ? parseFrontmatter(sourceContent ?? "").data : {};
490
266
  const { fmText: llmFmText, body: rawLlmBody } = splitFrontmatter(payload.content);
491
- if (llmFmText !== null) {
492
- warnings.push("LLM emitted frontmatter in content; stripped and merged through identity guard.");
493
- }
494
- // Parse the LLM-emitted frontmatter (if any) so we can merge its non-identity
495
- // keys into the source frontmatter.
496
267
  let llmFm = {};
497
268
  if (llmFmText !== null) {
269
+ warnings.push("LLM emitted frontmatter in content; stripped and merged through identity guard.");
498
270
  try {
499
271
  llmFm = parseFrontmatter(payload.content).data;
500
272
  }
@@ -502,107 +274,60 @@ export function sanitizeReflectPayload(payload, sourceContent, targetRef) {
502
274
  llmFm = {};
503
275
  }
504
276
  }
505
- // Also accept the explicit `frontmatter` field on the payload.
506
- if (payload.frontmatter && typeof payload.frontmatter === "object") {
277
+ if (payload.frontmatter && typeof payload.frontmatter === "object")
507
278
  llmFm = { ...llmFm, ...payload.frontmatter };
508
- }
509
- // Strip protected identity fields from any LLM-supplied frontmatter — they
510
- // must come from the source asset, never from the LLM.
511
279
  for (const field of PROTECTED_FRONTMATTER_FIELDS) {
512
280
  if (field in llmFm && llmFm[field] !== sourceFm[field]) {
513
281
  warnings.push(`LLM attempted to change protected frontmatter field "${field}"; restored from source.`);
514
282
  delete llmFm[field];
515
283
  }
516
284
  }
517
- // Build the effective frontmatter: source overlaid with sanitized LLM fields.
518
- // Source fields always win on identity keys.
519
285
  const mergedFm = { ...sourceFm, ...llmFm };
520
- for (const field of PROTECTED_FRONTMATTER_FIELDS) {
521
- if (field in sourceFm) {
286
+ for (const field of PROTECTED_FRONTMATTER_FIELDS)
287
+ if (field in sourceFm)
522
288
  mergedFm[field] = sourceFm[field];
523
- }
524
- }
525
- const withoutAppendedFrontmatter = stripAppendedFrontmatter(rawLlmBody.replace(/^\s+/, ""));
526
- const promptScaffolding = stripReflectPromptScaffolding(withoutAppendedFrontmatter);
527
- const cleanedBody = promptScaffolding.content;
528
- if (promptScaffolding.stripped) {
289
+ const scaffolding = stripReflectPromptScaffolding(stripAppendedFrontmatter(rawLlmBody.replace(/^\s+/, "")));
290
+ const cleanedBody = scaffolding.content;
291
+ if (scaffolding.stripped) {
529
292
  warnings.push('Removed echoed run-only "Avoid These Patterns" guidance from the proposed asset body (#963).');
530
293
  }
531
- // #636 — deterministic description fallback (reflect-side belt-and-suspenders).
532
- // If the type requires a `description` and the merged frontmatter is still
533
- // MISSING one (source had none AND the model didn't author one), derive a
534
- // description DETERMINISTICALLY from the existing `title:` frontmatter or the
535
- // first `# Heading` / opening body sentence — never free-form invention. This
536
- // runs in the reflect proposal-build path, BEFORE the proposal is created, so
537
- // the validator/promote path is left untouched (no gate fabricates content).
538
- //
539
- // Scope is the issue's target: a source asset that ALREADY carries frontmatter
540
- // (e.g. scraped docs: `source`/`title`/`scraped`) but has a MISSING/empty
541
- // `description`. We deliberately do NOT fire when:
542
- // - the source has no frontmatter block at all (injecting one would be a
543
- // structural change and would defeat the #580 no-op/cosmetic noise gate
544
- // for a pure body echo), or
545
- // - a present-but-otherwise-invalid description exists (too short, a heading
546
- // fragment) — overwriting authored content is out of scope; the prompt
547
- // instruction handles improving it instead.
294
+ // Only a source that already has frontmatter but no description gets one:
295
+ // injecting a whole block, or overwriting an authored one, is out of scope.
548
296
  const refType = lenientRefType(targetRef);
549
- const mergedDesc = mergedFm.description;
550
- const descIsMissing = typeof mergedDesc !== "string" || mergedDesc.trim().length === 0;
297
+ const desc = mergedFm.description;
551
298
  const sourceHadFrontmatter = sourceFmText !== null && Object.keys(sourceFm).length > 0;
552
- if (refType && requiresDescription(refType) && descIsMissing && sourceHadFrontmatter) {
299
+ if (refType &&
300
+ requiresDescription(refType) &&
301
+ (typeof desc !== "string" || desc.trim().length === 0) &&
302
+ sourceHadFrontmatter) {
553
303
  const derived = deriveDescriptionFromAsset(mergedFm.title, cleanedBody, sourceBody, targetRef);
554
304
  if (derived) {
555
305
  mergedFm.description = derived;
556
306
  warnings.push("Synthesized a deterministic `description` from title/heading (#636) — source and proposal lacked one.");
557
307
  }
558
308
  }
559
- // Size guard — only when source body is meaningfully large. The pure
560
- // predicate lives in `core/proposal-quality-validators` so the same check
561
- // also runs inside `runProposalValidators` on `proposal accept`.
562
- const sizeOutcome = checkReflectSize(sourceBody, cleanedBody);
309
+ const size = checkReflectSize(sourceBody, cleanedBody);
563
310
  let sizeGuardRatio;
564
- if (!sizeOutcome.ok) {
565
- const pct = (sizeOutcome.ratio * 100).toFixed(0);
566
- const limit = sizeOutcome.code === "EXCESSIVE_SHRINKAGE" ? "minimum 50%" : "maximum 250%";
567
- const cause = sizeOutcome.code === "EXCESSIVE_SHRINKAGE"
568
- ? "Concrete content was likely deleted."
569
- : "Speculative material was likely added.";
570
- warnings.push(`${sizeOutcome.code} — proposed body is ${pct}% of source (${limit}) for ref ${targetRef}. ${cause} Flagged for review.`);
571
- sizeGuardRatio = { code: sizeOutcome.code, ratio: sizeOutcome.ratio };
311
+ if (!size.ok) {
312
+ const shrink = size.code === "EXCESSIVE_SHRINKAGE";
313
+ warnings.push(`${size.code} — proposed body is ${(size.ratio * 100).toFixed(0)}% of source (${shrink ? "minimum 50%" : "maximum 250%"}) for ref ${targetRef}. ${shrink ? "Concrete content was likely deleted." : "Speculative material was likely added."} Flagged for review.`);
314
+ sizeGuardRatio = { code: size.code, ratio: size.ratio };
572
315
  }
573
- // Truncation-marker leak (#952) — a model that saw a capped/truncated
574
- // asset sometimes echoes the "[truncated ...]" notice verbatim into its
575
- // rewrite instead of proposing real content for the missing tail. The
576
- // body-length ratio check above does not reliably catch this (a leaked
577
- // marker can still fall inside the 50%-250% band). Flag and defer to
578
- // human review — same "degrade with a warning" rung as the size guard,
579
- // not a new hard reject.
580
316
  const truncationMarkerLeaked = cleanedBody.includes(REFLECT_TRUNCATION_MARKER);
581
317
  if (truncationMarkerLeaked) {
582
318
  warnings.push(`Proposed body for ref ${targetRef} contains the truncation-notice text the model was shown for a capped source asset ("${REFLECT_TRUNCATION_MARKER}"). The model likely echoed the notice instead of writing real content. Flagged for review.`);
583
319
  }
584
- // Reassemble final content: merged frontmatter + cleaned body.
585
- // When there is no frontmatter at all (no source fm and no LLM fm), emit body
586
- // only so we don't add a stray `---` to e.g. a script asset that bypassed the
587
- // type guard via a custom registration.
320
+ // No frontmatter at all stays body-only, never gaining a stray `---`.
588
321
  const hasFrontmatter = Object.keys(mergedFm).length > 0;
589
- const reassembled = hasFrontmatter
590
- ? assembleAssetFromString(serializeFrontmatter(mergedFm), cleanedBody)
591
- : cleanedBody;
592
322
  return {
593
- content: reassembled,
323
+ content: hasFrontmatter ? assembleAssetFromString(serializeFrontmatter(mergedFm), cleanedBody) : cleanedBody,
594
324
  ...(hasFrontmatter ? { frontmatter: mergedFm } : {}),
595
325
  warnings,
596
326
  ...(sizeGuardRatio ? { sizeGuardRatio } : {}),
597
327
  ...(truncationMarkerLeaked ? { truncationMarkerLeaked } : {}),
598
328
  };
599
329
  }
600
- /**
601
- * JSON Schema for structured reflect output. Passed to `chatCompletion` when
602
- * {@link wantsJsonSchemaOutput} selects `outputMode: "json_schema"`, so the
603
- * model returns a strict JSON object containing only the target-scoped
604
- * fields AKM cannot derive.
605
- */
330
+ // ── Direct-LLM output contract ───────────────────────────────────────────────
606
331
  const REFLECT_FRONTMATTER_PATCH_JSON_SCHEMA = {
607
332
  type: "object",
608
333
  required: ["description", "when_to_use"],
@@ -612,23 +337,19 @@ const REFLECT_FRONTMATTER_PATCH_JSON_SCHEMA = {
612
337
  when_to_use: { type: ["string", "null"] },
613
338
  },
614
339
  };
340
+ const REFLECT_CONFIDENCE_SCHEMA = {
341
+ type: "number",
342
+ minimum: 0,
343
+ maximum: 1,
344
+ description: "Self-reported quality confidence in [0, 1]. Persisted on the proposal for reviewers and the triage judge to read during adjudication.",
345
+ };
615
346
  export const REFLECT_JSON_SCHEMA = {
616
347
  type: "object",
617
348
  required: ["content", "confidence", "frontmatterPatch"],
618
349
  additionalProperties: false,
619
350
  properties: {
620
351
  content: { type: "string", description: "Complete improved markdown body without YAML frontmatter." },
621
- // Phase 6A (Advantage D6a): self-reported confidence in [0, 1]. When the
622
- // LLM is well-calibrated, scores at or above the configured threshold
623
- // (default 0.8) drive auto-accept in `akm improve`. Out-of-range or
624
- // non-finite values are rejected by direct-output extraction. Agent and SDK
625
- // confidence remains optional on their separate existing contracts.
626
- confidence: {
627
- type: "number",
628
- minimum: 0,
629
- maximum: 1,
630
- description: "Self-reported quality confidence in [0, 1]. Persisted on the proposal for reviewers and the triage judge to read during adjudication.",
631
- },
352
+ confidence: REFLECT_CONFIDENCE_SCHEMA,
632
353
  frontmatterPatch: REFLECT_FRONTMATTER_PATCH_JSON_SCHEMA,
633
354
  },
634
355
  };
@@ -639,44 +360,32 @@ const REFLECT_UNSCOPED_JSON_SCHEMA = {
639
360
  properties: {
640
361
  ref: { type: "string", description: "Selected asset ref as a subdir-qualified conceptId." },
641
362
  content: { type: "string", description: "Complete improved markdown body without YAML frontmatter." },
642
- confidence: {
643
- type: "number",
644
- minimum: 0,
645
- maximum: 1,
646
- description: "Self-reported quality confidence in [0, 1].",
647
- },
363
+ confidence: { type: "number", minimum: 0, maximum: 1, description: "Self-reported quality confidence in [0, 1]." },
648
364
  frontmatterPatch: REFLECT_FRONTMATTER_PATCH_JSON_SCHEMA,
649
365
  },
650
366
  };
651
367
  /**
652
- * Whether to frame the reflect prompt for structured JSON output on this
653
- * connection. Optimistic by default — `chatCompletion` attempts
654
- * `response_format: json_schema` fresh on every call and falls back once on
655
- * a 4xx, so there is no persisted verdict to consult here. `false` only when
656
- * a human/workflow explicitly disabled it, or a real call already proved
657
- * this connection rejects it earlier in the same process.
368
+ * Frame for JSON Schema unless the connection disabled it or already proved
369
+ * this process that it rejects it (the transport retries plain text on a 4xx).
658
370
  */
659
371
  function wantsJsonSchemaOutput(connection) {
660
372
  return connection.supportsJsonSchema !== false && !isJsonSchemaKnownUnsupported(connection);
661
373
  }
662
- /** Critique prompt injected between prior draft and refinement request (Self-Refine loop). */
374
+ /** Injected between the prior draft and the refinement request (self-refine). */
663
375
  const REFLECT_CRITIQUE_PROMPT = "Your previous proposal is shown above. Review it critically and provide an improved version that is more specific, actionable, and avoids any issues with the previous attempt. Return only the improved response using the output contract from the original prompt.";
376
+ function parsedRecord(result) {
377
+ return result.parsed && typeof result.parsed === "object" && !Array.isArray(result.parsed)
378
+ ? result.parsed
379
+ : undefined;
380
+ }
664
381
  function reflectLlmTelemetry(result) {
665
- if (!result.parsed || typeof result.parsed !== "object" || Array.isArray(result.parsed))
666
- return undefined;
667
- const parsed = result.parsed;
668
- if (parsed.outputMode !== "json_schema" && parsed.outputMode !== "framed_markdown")
382
+ const parsed = parsedRecord(result);
383
+ if (!parsed || (parsed.outputMode !== "json_schema" && parsed.outputMode !== "framed_markdown"))
669
384
  return undefined;
670
385
  if (typeof parsed.repairAttempts !== "number")
671
386
  return undefined;
672
387
  return { outputMode: parsed.outputMode, repairAttempts: parsed.repairAttempts };
673
388
  }
674
- function reflectLlmPriorDraft(result) {
675
- if (!result.parsed || typeof result.parsed !== "object" || Array.isArray(result.parsed))
676
- return undefined;
677
- const priorDraft = result.parsed.priorDraft;
678
- return typeof priorDraft === "string" ? priorDraft : undefined;
679
- }
680
389
  function parseReflectConfidence(value) {
681
390
  if (typeof value !== "number" || !Number.isFinite(value) || value < 0 || value > 1) {
682
391
  throw new Error('direct reflect response missing required number field "confidence" in [0, 1]');
@@ -744,88 +453,57 @@ function parseFramedReflectOutput(raw, targetRef) {
744
453
  throw new Error("direct reflect response missing terminal AKM_REFLECT_CONTENT_END marker");
745
454
  }
746
455
  const headerLines = normalized.slice(0, beginIndex).trim().split("\n").filter(Boolean);
747
- const confidenceLine = headerLines.find((line) => line.startsWith("AKM_REFLECT_CONFIDENCE:"));
748
- const refLine = headerLines.find((line) => line.startsWith("AKM_REFLECT_REF:"));
749
- const patchLine = headerLines.find((line) => line.startsWith("AKM_REFLECT_FRONTMATTER_PATCH:"));
750
- const expectedHeaderLines = targetRef ? 2 : 3;
456
+ const header = (prefix) => headerLines.find((line) => line.startsWith(prefix));
457
+ const confidenceLine = header("AKM_REFLECT_CONFIDENCE:");
458
+ const refLine = header("AKM_REFLECT_REF:");
459
+ const patchLine = header("AKM_REFLECT_FRONTMATTER_PATCH:");
751
460
  const invalidRefLine = targetRef ? refLine !== undefined : refLine === undefined;
752
- if (headerLines.length !== expectedHeaderLines || !confidenceLine || !patchLine || invalidRefLine) {
461
+ if (headerLines.length !== (targetRef ? 2 : 3) || !confidenceLine || !patchLine || invalidRefLine) {
753
462
  throw new Error("direct reflect response contained invalid frame metadata");
754
463
  }
755
464
  const confidenceText = confidenceLine.slice("AKM_REFLECT_CONFIDENCE:".length).trim();
756
465
  if (!/^(?:0(?:\.\d+)?|1(?:\.0+)?)$/.test(confidenceText)) {
757
466
  throw new Error("direct reflect frame confidence must be a decimal number in [0, 1]");
758
467
  }
759
- const confidence = parseReflectConfidence(Number(confidenceText));
760
468
  const ref = targetRef ?? refLine?.slice("AKM_REFLECT_REF:".length).trim() ?? "";
761
469
  if (!ref)
762
470
  throw new Error("direct reflect response contained an empty AKM_REFLECT_REF value");
763
471
  const content = normalized.slice(contentStart, endIndex);
764
472
  if (!content.trim())
765
473
  throw new Error("direct reflect response contained empty framed content");
766
- const patchText = patchLine.slice("AKM_REFLECT_FRONTMATTER_PATCH:".length).trim();
767
474
  let parsedPatch;
768
475
  try {
769
- parsedPatch = JSON.parse(patchText);
476
+ parsedPatch = JSON.parse(patchLine.slice("AKM_REFLECT_FRONTMATTER_PATCH:".length).trim());
770
477
  }
771
478
  catch {
772
479
  throw new Error("direct reflect response contained invalid frontmatter patch JSON");
773
480
  }
774
481
  const frontmatter = parseReflectFrontmatterPatch(parsedPatch);
482
+ const confidence = parseReflectConfidence(Number(confidenceText));
775
483
  return { ref, content, confidence, ...(frontmatter ? { frontmatter } : {}) };
776
484
  }
777
- function parseDirectReflectOutput(raw, mode, targetRef) {
778
- return mode === "json_schema" ? parseSchemaReflectOutput(raw, targetRef) : parseFramedReflectOutput(raw, targetRef);
779
- }
780
485
  /**
781
- * Run a single reflect iteration directly via the LLM API (v2 config path).
782
- *
783
- * Returns an {@link AgentRunResult}-shaped object so it can slot into the same
784
- * dispatch loop as agent-based runners. Production calls extract the selected
785
- * direct-LLM contract and normalize it to proposal JSON in `stdout`. Errors
786
- * are captured into the result rather than thrown.
486
+ * One reflect iteration through the direct LLM runner, as an agent-shaped
487
+ * result (errors captured, never thrown except configuration). An unparseable
488
+ * response gets one repair turn within the original deadline.
787
489
  */
788
490
  export async function runReflectViaLlm(opts) {
789
491
  const start = Date.now();
790
492
  let repairAttempts = 0;
791
- const _connection = opts.runner.connection;
792
- const messages = [{ role: "user", content: opts.prompt ?? "" }];
793
493
  const configuredTimeout = Object.hasOwn(opts, "timeoutMs")
794
494
  ? (opts.timeoutMs ?? null)
795
495
  : Object.hasOwn(opts.runner, "timeoutMs")
796
496
  ? (opts.runner.timeoutMs ?? null)
797
497
  : DEFAULT_LLM_TIMEOUT_MS;
798
498
  const deadline = typeof configuredTimeout === "number" ? start + configuredTimeout : undefined;
499
+ const messages = [{ role: "user", content: opts.prompt ?? "" }];
799
500
  if (opts.priorDraft !== undefined && opts.iteration > 0) {
800
- messages.push({ role: "assistant", content: opts.priorDraft });
801
- messages.push({ role: "user", content: REFLECT_CRITIQUE_PROMPT });
501
+ messages.push({ role: "assistant", content: opts.priorDraft }, { role: "user", content: REFLECT_CRITIQUE_PROMPT });
802
502
  }
803
- const call = async (callMessages, repairTimeoutMs) => callStructured({
804
- feature: "reflect_proposal",
805
- runner: opts.runner,
806
- ...(opts.lease ? { lease: opts.lease } : {}),
807
- messages: callMessages,
808
- request: {
809
- ...(repairTimeoutMs !== undefined
810
- ? { timeoutMs: repairTimeoutMs }
811
- : Object.hasOwn(opts, "timeoutMs")
812
- ? { timeoutMs: opts.timeoutMs }
813
- : {}),
814
- ...(opts.signal ? { signal: opts.signal } : {}),
815
- ...(opts.responseSchema !== undefined ? { responseSchema: opts.responseSchema } : {}),
816
- ...(opts.maxTokens !== undefined ? { maxTokens: opts.maxTokens } : {}),
817
- // Reflect requires a machine-readable payload. Visible chain-of-thought
818
- // can consume the output cap before the model reaches the envelope.
819
- enableThinking: false,
820
- ...(opts.chat ? { chat: opts.chat } : {}),
821
- },
822
- ...(opts.onNotices ? { onNotices: opts.onNotices } : {}),
823
- parse: (raw) => raw ?? "",
824
- // Unreachable on the ungated path (errors propagate to the catch below).
825
- onError: () => "",
826
- fallback: "",
827
- });
828
- const failure = (err, reason, repairAttempts, stdout = "", exitCode = 1) => {
503
+ const parse = (raw) => opts.outputMode === "json_schema"
504
+ ? parseSchemaReflectOutput(raw, opts.targetRef)
505
+ : parseFramedReflectOutput(raw, opts.targetRef);
506
+ const failure = (err, reason, stdout = "", exitCode = 1) => {
829
507
  const msg = err instanceof Error ? err.message : String(err);
830
508
  return {
831
509
  ok: false,
@@ -838,6 +516,34 @@ export async function runReflectViaLlm(opts) {
838
516
  parsed: { outputMode: opts.outputMode, repairAttempts },
839
517
  };
840
518
  };
519
+ const call = async (callMessages, repairTimeoutMs) => {
520
+ const outcome = await callStage({
521
+ feature: "reflect_proposal",
522
+ runner: opts.runner,
523
+ prompt: callMessages.at(-1)?.content ?? "",
524
+ history: callMessages.slice(0, -1),
525
+ request: {
526
+ ...(repairTimeoutMs !== undefined
527
+ ? { timeoutMs: repairTimeoutMs }
528
+ : Object.hasOwn(opts, "timeoutMs")
529
+ ? { timeoutMs: opts.timeoutMs }
530
+ : {}),
531
+ ...(opts.signal ? { signal: opts.signal } : {}),
532
+ ...(opts.responseSchema !== undefined ? { responseSchema: opts.responseSchema } : {}),
533
+ ...(opts.maxTokens !== undefined ? { maxTokens: opts.maxTokens } : {}),
534
+ // Visible chain-of-thought can exhaust the output before the envelope.
535
+ enableThinking: false,
536
+ ...(opts.chat ? { chat: opts.chat } : {}),
537
+ },
538
+ ...(opts.onNotices ? { onNotices: opts.onNotices } : {}),
539
+ });
540
+ if (!outcome.ok) {
541
+ throw outcome.reason === "timeout"
542
+ ? new LlmCallError(outcome.error ?? "timeout", "timeout")
543
+ : new Error(outcome.error ?? "LLM call failed");
544
+ }
545
+ return outcome.raw;
546
+ };
841
547
  try {
842
548
  if (opts.signal?.aborted)
843
549
  throw new Error("Reflect request aborted");
@@ -845,33 +551,25 @@ export async function runReflectViaLlm(opts) {
845
551
  let payload;
846
552
  let acceptedOutput = stdout;
847
553
  try {
848
- payload = parseDirectReflectOutput(stdout, opts.outputMode, opts.targetRef);
554
+ payload = parse(stdout);
849
555
  }
850
556
  catch (err) {
851
557
  if (opts.allowRepair === false)
852
- return failure(err, "parse_error", 0, stdout, 0);
558
+ return failure(err, "parse_error", stdout, 0);
853
559
  if (opts.signal?.aborted)
854
- return failure(new Error("Reflect request aborted"), "aborted", 0, stdout);
560
+ return failure(new Error("Reflect request aborted"), "aborted", stdout);
855
561
  const remaining = deadline === undefined ? undefined : deadline - Date.now();
856
562
  if (remaining !== undefined && remaining <= 0) {
857
- return failure(new LlmCallError("Reflect request timed out before output repair", "timeout"), "timeout", 0, stdout);
563
+ return failure(new LlmCallError("Reflect request timed out before output repair", "timeout"), "timeout", stdout);
858
564
  }
859
565
  repairAttempts = 1;
860
- const repairMessages = [
861
- ...messages,
862
- { role: "assistant", content: stdout },
863
- {
864
- role: "user",
865
- content: buildReflectOutputRepairPrompt(opts.outputMode, opts.targetRef !== undefined),
866
- },
867
- ];
868
- const repaired = await call(repairMessages, remaining);
869
- acceptedOutput = repaired;
566
+ const repairPrompt = buildReflectOutputRepairPrompt(opts.outputMode, opts.targetRef !== undefined);
567
+ acceptedOutput = await call([...messages, { role: "assistant", content: stdout }, { role: "user", content: repairPrompt }], remaining);
870
568
  try {
871
- payload = parseDirectReflectOutput(repaired, opts.outputMode, opts.targetRef);
569
+ payload = parse(acceptedOutput);
872
570
  }
873
- catch (err) {
874
- return failure(err, "parse_error", repairAttempts, repaired, 0);
571
+ catch (repairErr) {
572
+ return failure(repairErr, "parse_error", acceptedOutput, 0);
875
573
  }
876
574
  }
877
575
  return {
@@ -891,398 +589,114 @@ export async function runReflectViaLlm(opts) {
891
589
  : err instanceof LlmCallError && err.code === "timeout"
892
590
  ? "timeout"
893
591
  : "non_zero_exit";
894
- return failure(err, reason, repairAttempts);
592
+ return failure(err, reason);
895
593
  }
896
594
  }
897
- function failureEnvelope(result, ref, engine, fallbackReason = "non_zero_exit") {
898
- return {
899
- ...baseFailureFields(result, fallbackReason),
900
- schemaVersion: 2,
901
- ...(ref ? { ref } : {}),
902
- ...(engine ? { engine } : {}),
903
- };
904
- }
905
- /**
906
- * Reflect content-preservation + proposal creation: restore/reset protected
907
- * frontmatter and reject unsafe body-size ratios (sanitizeReflectPayload), the
908
- * #580 noise gate, the optional quality judge, then create the proposal (with
909
- * the R-4/#373 lesson provenance stamp) and emit `reflect_completed`. Extracted
910
- * verbatim from `akmReflect`; every reject/skip envelope and event is
911
- * byte-identical.
912
- */
913
- async function finalizeReflectProposal(args) {
914
- const { assetContent, result, options, engineName, config, qualityGateEnabled, qualityGateSkippedNoJudge, qualityJudgeRunner, qualityJudgeLease, feedback, stash, emitReflectFailed, onNotices, } = args;
915
- let payload = args.payload;
916
- const outputTelemetry = reflectLlmTelemetry(result);
917
- // 7. Reflect content-preservation rails:
918
- // - Restore source frontmatter so reflect can never strip indexable
919
- // fields (`description`, `when_to_use`, `tags`, ...).
920
- // - Reset protected identity fields (`name`, `ref`, `id`, `slug`,
921
- // `type`) the LLM tried to change.
922
- // - Reject proposals that shrink/expand the body past safe ratios.
923
- //
924
- // See REFLECT_ALLOWED_TYPES / sanitizeReflectPayload for the underlying
925
- // hypotheses + observed regressions (`8737ab63`, `26941510`, and the
926
- // catastrophic-shrinkage cases from the May 2026 review).
927
- const sanitizeOutcome = sanitizeReflectPayload({ content: payload.content, ...(payload.frontmatter ? { frontmatter: payload.frontmatter } : {}) }, assetContent, payload.ref);
928
- if (sanitizeOutcome.reject) {
595
+ /** The lazy `reflect_invoked` + failure-side `reflect_completed` emitters. */
596
+ function reflectEmitters(options) {
597
+ let invoked = false;
598
+ const emitInvoked = () => {
599
+ if (invoked)
600
+ return;
929
601
  appendEvent({
930
- eventType: "reflect_completed",
931
- ref: payload.ref,
602
+ eventType: "reflect_invoked",
603
+ ...(options.ref ? { ref: options.itemRef ?? options.ref } : {}),
932
604
  metadata: {
933
- source: "reflect",
934
- sanitized: true,
935
- rejected: true,
936
- rejectReason: sanitizeOutcome.reject.error,
937
- ...(sanitizeOutcome.warnings.length > 0 ? { sanitizerWarnings: sanitizeOutcome.warnings } : {}),
938
- ...(outputTelemetry ?? {}),
605
+ ...(options.task ? { task: options.task } : {}),
606
+ ...(options.engine ? { engine: options.engine } : {}),
607
+ ...(options.eligibilitySource ? { eligibilitySource: options.eligibilitySource } : {}),
939
608
  },
940
609
  }, options.eventsCtx);
941
- return {
942
- schemaVersion: 2,
943
- ok: false,
944
- reason: sanitizeOutcome.reject.reason,
945
- error: sanitizeOutcome.reject.error,
946
- ...(options.ref ? { ref: options.ref } : {}),
947
- engine: engineName,
948
- exitCode: result.exitCode,
949
- };
950
- }
951
- payload = {
952
- ...payload,
953
- content: sanitizeOutcome.content,
954
- ...(sanitizeOutcome.frontmatter ? { frontmatter: sanitizeOutcome.frontmatter } : {}),
610
+ invoked = true;
955
611
  };
956
- // 7c. Noise gate (#580): never queue a proposal whose sanitized content is
957
- // identical to the current asset (empty diff) or differs only cosmetically
958
- // (whitespace reflow, code-fence language hints, YAML scalar re-folding).
959
- // Pure deterministic text comparison — see `reflect-noise.ts`. Skipped when
960
- // there is no source asset (new-asset proposals have nothing to diff against).
961
- if (assetContent !== undefined) {
962
- const changeKind = classifyReflectChange(assetContent, payload.content);
963
- // 'low-value' is config-gated (#639). DEFAULT OFF — absent = byte-identical
964
- // pre-#639 behaviour (low-value treated the same as substantive). Resolved
965
- // by the caller from the active improve strategy's
966
- // `processes.reflect.lowValueFilter.enabled` and passed via options, so the
967
- // running strategy decides.
968
- const lowValueFilterEnabled = options.lowValueFilter === true;
969
- const isDeferred = changeKind === "noop" || changeKind === "cosmetic" || (changeKind === "low-value" && lowValueFilterEnabled);
970
- if (isDeferred) {
971
- const subreason = changeKind === "noop"
972
- ? "reflect_skipped_noop"
973
- : changeKind === "low-value"
974
- ? "reflect_skipped_low_value"
975
- : "reflect_skipped_cosmetic";
976
- emitReflectFailed("no_change", subreason, options.ref, { changeKind, ...(outputTelemetry ?? {}) });
977
- return {
978
- schemaVersion: 2,
979
- ok: false,
980
- reason: "no_change",
981
- error: changeKind === "noop"
982
- ? `Reflect skipped: proposed content for ${payload.ref} is identical to the current asset (empty diff); no proposal created.`
983
- : changeKind === "low-value"
984
- ? `Reflect skipped: proposed content for ${payload.ref} is a low-value prose micro-rewrite (few changed tokens, no structural changes); no proposal created.`
985
- : `Reflect skipped: proposed content for ${payload.ref} is a cosmetic-only reformat of the current asset (whitespace/fence/YAML-folding changes); no proposal created.`,
986
- ...(options.ref ? { ref: options.ref } : {}),
987
- engine: engineName,
988
- exitCode: result.exitCode,
989
- };
990
- }
991
- }
992
- // 7c. Judge the exact sanitized content that can be persisted. Fail closed
993
- // on cancellation, transport failure, malformed output, or an invalid score.
994
- // Skipped when the size guard or the truncation-marker leak already fired —
995
- // that content is deferred to human review regardless of what the judge says.
996
- if (qualityGateEnabled && !sanitizeOutcome.sizeGuardRatio && !sanitizeOutcome.truncationMarkerLeaked) {
997
- const judgeResult = await runReflectQualityJudge(config, payload.content, assetContent ?? "", feedback, options.chat, {
998
- runnerSelectionFrozen: true,
999
- ...(qualityJudgeRunner ? { llmRunner: qualityJudgeRunner } : {}),
1000
- ...(qualityJudgeLease ? { lease: qualityJudgeLease } : {}),
1001
- ...(Object.hasOwn(options, "timeoutMs") ? { timeoutMs: options.timeoutMs } : {}),
1002
- ...(options.signal ? { signal: options.signal } : {}),
1003
- onNotices,
1004
- });
1005
- if (!judgeResult.pass) {
1006
- appendEvent({
1007
- eventType: "reflect_completed",
1008
- ref: payload.ref,
1009
- metadata: {
1010
- source: "reflect",
1011
- qualityRejected: true,
1012
- qualityScore: judgeResult.score,
1013
- qualityReason: judgeResult.reason,
1014
- ...(judgeResult.criteria ? { qualityCriteria: judgeResult.criteria } : {}),
1015
- ...(outputTelemetry ?? {}),
1016
- },
1017
- }, options.eventsCtx);
1018
- return {
1019
- schemaVersion: 2,
1020
- ok: false,
1021
- reason: "quality_rejected",
1022
- error: `Reflect proposal quality gate rejected: score=${judgeResult.score}, reason="${judgeResult.reason}"`,
1023
- ...(options.ref ? { ref: options.ref } : {}),
1024
- engine: engineName,
1025
- exitCode: result.exitCode,
1026
- };
1027
- }
1028
- }
1029
- return createReflectProposal({
1030
- payload,
1031
- options,
1032
- stash,
1033
- engineName,
1034
- durationMs: result.durationMs,
1035
- emitReflectFailed,
1036
- outputTelemetry,
1037
- qualityGateSkippedNoJudge,
1038
- sizeGuardRatio: sanitizeOutcome.sizeGuardRatio,
1039
- truncationMarkerLeaked: sanitizeOutcome.truncationMarkerLeaked,
1040
- });
612
+ const emitFailed = (reason, subreason, ref, extra) => {
613
+ emitInvoked();
614
+ appendEvent({
615
+ eventType: "reflect_completed",
616
+ ...(ref ? { ref } : {}),
617
+ metadata: { source: "reflect", ok: false, reason, subreason, ...(extra ?? {}) },
618
+ }, options.eventsCtx);
619
+ };
620
+ return { emitInvoked, emitFailed };
1041
621
  }
1042
- /**
1043
- * Create the reflect proposal from sanitized+judged payload: stamp the R-4/#373
1044
- * lesson provenance marker, call `createProposal`, and emit the terminal
1045
- * `reflect_completed` (or a cooldown skip envelope). Extracted verbatim from
1046
- * `akmReflect`'s finalize tail.
1047
- */
1048
- function createReflectProposal(args) {
1049
- const { payload, options, stash, engineName, durationMs, emitReflectFailed, outputTelemetry, qualityGateSkippedNoJudge, sizeGuardRatio, truncationMarkerLeaked, } = args;
1050
- // 8. Create the proposal. The proposal queue is the ONLY thing reflect
1051
- // writes — promotion to a real asset is gated by `akm proposal accept`.
1052
- //
1053
- // R-4 / #373: Stamp `derived_from_reflect: true` in the frontmatter of any
1054
- // lesson proposal generated by reflect. This provenance marker lets
1055
- // `readRelatedLessons` exclude echo-chamber lessons (lessons that originate
1056
- // from prior reflect runs on the same skill) unless independent feedback
1057
- // evidence exists. ExpeL arXiv:2308.10144 — reject rules without success/
1058
- // failure differential from independent evidence.
1059
- const isLessonProposal = (() => {
1060
- try {
1061
- return parseRefInput(payload.ref).type === "lesson";
1062
- }
1063
- catch {
1064
- return false;
1065
- }
1066
- })();
1067
- const basePayloadFrontmatter = payload.frontmatter ?? {};
1068
- const payloadFrontmatterWithProvenance = isLessonProposal
1069
- ? { ...basePayloadFrontmatter, derived_from_reflect: true }
1070
- : basePayloadFrontmatter;
1071
- const createInput = {
1072
- ref: payload.ref,
1073
- ...(options.target ? { target: options.target } : {}),
1074
- source: "reflect",
1075
- sourceRun: `reflect-${Date.now()}`,
1076
- payload: {
1077
- content: payload.content,
1078
- ...(Object.keys(payloadFrontmatterWithProvenance).length > 0
1079
- ? { frontmatter: payloadFrontmatterWithProvenance }
1080
- : {}),
1081
- },
1082
- // Phase 6A: forward LLM-reported confidence into the proposal record.
1083
- // `parseAgentProposalPayload` already clamps to [0, 1] and drops non-
1084
- // finite values; `createProposal` runs its own sanitizer as a safety net.
1085
- ...(typeof payload.confidence === "number" ? { confidence: payload.confidence } : {}),
1086
- // Attribution tagging: persist the eligibility lane on the proposal so it
1087
- // survives to accept/reject/revert time even across runs. See EligibilitySource.
1088
- ...(options.eligibilitySource ? { eligibilitySource: options.eligibilitySource } : {}),
1089
- // §23.6 fingerprint model-id term (WI-6.4): the engine that generated
1090
- // this draft (reflect resolves engines, not bare model ids).
1091
- modelId: engineName,
622
+ /** A post-dispatch failure envelope (with the run's notices). */
623
+ function reflectFailure(run, result, reason, error, withOutput) {
624
+ return {
625
+ schemaVersion: 2,
626
+ ok: false,
627
+ reason,
628
+ error,
629
+ ...(run.options.ref ? { ref: run.options.ref } : {}),
630
+ engine: run.engineName,
631
+ exitCode: result.exitCode,
632
+ ...(withOutput ? { stdout: result.stdout, ...(result.stderr ? { stderr: result.stderr } : {}) } : {}),
633
+ ...run.notices.fields(),
1092
634
  };
1093
- const proposalResult = emitProposal({ stashDir: stash, proposalsCtx: options.ctx }, createInput);
1094
- if (isProposalSkipped(proposalResult)) {
1095
- // Dedup/cooldown guard fired — surface as a "cooldown" reason (not "parse_error")
1096
- // so the improve orchestrator can distinguish legitimate skips from real failures
1097
- // and exclude them from recentErrors/avoidPatterns injection.
1098
- emitReflectFailed("cooldown", "proposal_skipped", options.ref, {
1099
- proposalSkipReason: proposalResult.reason,
1100
- ...(outputTelemetry ?? {}),
1101
- });
1102
- return {
635
+ }
636
+ function exitCodeMeta(result) {
637
+ return result.exitCode !== null ? { exitCode: result.exitCode } : {};
638
+ }
639
+ function unsupportedTypeFailure(ref, type, detail, emitFailed) {
640
+ emitFailed("unsupported_type", "unsupported_type", ref, { type });
641
+ return {
642
+ failure: {
1103
643
  schemaVersion: 2,
1104
644
  ok: false,
1105
- reason: "cooldown",
1106
- error: `Proposal skipped (${proposalResult.reason}): ${proposalResult.message}`,
1107
- ...(options.ref ? { ref: options.ref } : {}),
1108
- engine: engineName,
645
+ reason: "unsupported_type",
646
+ error: `Reflect refused: asset type "${type}" is not supported by reflect (${detail}). Use \`akm proposal new\` or edit the file directly.`,
647
+ ref,
1109
648
  exitCode: null,
1110
- };
1111
- }
1112
- let proposal = proposalResult;
1113
- const reviewReasons = [];
1114
- if (qualityGateSkippedNoJudge)
1115
- reviewReasons.push("no-judge-configured");
1116
- if (sizeGuardRatio)
1117
- reviewReasons.push("reflect-size-ratio");
1118
- if (truncationMarkerLeaked)
1119
- reviewReasons.push("reflect-truncation-leak");
1120
- if (reviewReasons.length > 0) {
1121
- proposal =
1122
- recordGateDecision(stash, proposal.id, {
1123
- outcome: "deferred",
1124
- reason: reviewReasons.join("+"),
1125
- gate: "reflect",
1126
- ...(sizeGuardRatio ? { measured: Math.round(sizeGuardRatio.ratio * 100) } : {}),
1127
- }, options.ctx) ?? proposal;
1128
- }
1129
- appendEvent({
1130
- eventType: "reflect_completed",
1131
- ref: proposal.ref,
1132
- metadata: {
1133
- proposalId: proposal.id,
1134
- source: "reflect",
1135
- engine: engineName,
1136
- ...(qualityGateSkippedNoJudge ? { qualityGateSkippedNoJudge: true } : {}),
1137
- ...(sizeGuardRatio ? { sizeGuardRatio: sizeGuardRatio.code, sizeGuardRatioValue: sizeGuardRatio.ratio } : {}),
1138
- ...(truncationMarkerLeaked ? { truncationMarkerLeaked: true } : {}),
1139
- ...(outputTelemetry ?? {}),
1140
649
  },
1141
- }, options.eventsCtx);
1142
- return {
1143
- schemaVersion: 2,
1144
- ok: true,
1145
- proposal,
1146
- ref: proposal.ref,
1147
- engine: engineName,
1148
- durationMs,
1149
650
  };
1150
651
  }
1151
- /**
1152
- * Resolve the agent's proposal payload from a successful run: the file-write
1153
- * contract path (read `lastDraftPath`, extract self-rated confidence) or the
1154
- * JSON-stdout path used by direct LLM runners. Returns the payload or a terminal
1155
- * failure envelope.
1156
- */
1157
- function resolveReflectPayload(args) {
1158
- const { result, lastDraftPath, sensitiveValues, options, engineName, emitReflectFailed } = args;
1159
- // 6. Resolve the proposal content.
1160
- //
1161
- // Path A (file-write contract — preferred for agent/sdk runners on long
1162
- // assets): the agent wrote the body to `lastDraftPath` and printed
1163
- // `DRAFT_WRITTEN` on stdout. Load the body from disk and synthesize a
1164
- // payload. The `EXCESSIVE_EXPANSION`/schema-shape gates downstream still
1165
- // apply — they validate content, not transport.
1166
- //
1167
- // Path B (JSON stdout): the direct LLM runner cannot honour file-write.
1168
- const draftFileExists = lastDraftPath !== undefined && fs.existsSync(lastDraftPath) && fs.statSync(lastDraftPath).size > 0;
1169
- const draftSignaled = stdoutSignalsDraftWritten(result.stdout);
1170
- if (draftSignaled && lastDraftPath && !draftFileExists) {
1171
- // Agent claimed to write the draft but the file is missing or empty.
1172
- // Surface as a parse_error rather than silently falling through — the
1173
- // alternative would be parsing the `DRAFT_WRITTEN` sentinel as JSON,
1174
- // which is guaranteed to fail with a confusing message.
1175
- emitReflectFailed("parse_error", "draft_missing", options.ref, {
1176
- ...(result.exitCode !== null ? { exitCode: result.exitCode } : {}),
1177
- });
1178
- return {
1179
- failure: {
1180
- schemaVersion: 2,
1181
- ok: false,
1182
- reason: "parse_error",
1183
- error: `Agent emitted DRAFT_WRITTEN but draft file is missing or empty (${lastDraftPath}). The file-write contract failed; either the agent's file tools are broken or the path was unwritable.`,
1184
- ...(options.ref ? { ref: options.ref } : {}),
1185
- engine: engineName,
1186
- exitCode: result.exitCode,
1187
- stdout: result.stdout,
1188
- ...(result.stderr ? { stderr: result.stderr } : {}),
1189
- },
1190
- };
1191
- }
1192
- if (draftFileExists && lastDraftPath) {
1193
- // Happy path: agent wrote the body to disk. Use the ref the caller
1194
- // supplied (or a placeholder when omitted — the R-3 ref-mismatch guard
1195
- // below has no effect when there is no expected ref).
1196
- const fileContent = redactSensitiveText(fs.readFileSync(lastDraftPath, "utf8"), sensitiveValues);
1197
- // Phase 6A: file-write contract carries self-rated confidence on the
1198
- // `DRAFT_WRITTEN confidence=<n>` sentinel line. Extract it so the
1199
- // file-write path is on equal footing with the JSON-stdout path for
1200
- // auto-accept gating in `akm improve`.
1201
- const draftConfidence = extractDraftConfidence(result.stdout);
1202
- return {
1203
- payload: {
1204
- ref: options.ref ?? "",
1205
- content: fileContent,
1206
- ...(draftConfidence !== undefined ? { confidence: draftConfidence } : {}),
1207
- },
1208
- };
1209
- }
1210
- try {
1211
- return { payload: parseAgentProposalPayload(result.stdout ?? "") };
652
+ /** The target's parsed ref and current content, or a refusal for a type reflect cannot rewrite. */
653
+ async function resolveReflectSource(options, stash, emitFailed) {
654
+ if (!options.ref)
655
+ return { assetContent: undefined, parsedRef: undefined };
656
+ const parsedRef = parseRefInput(options.ref);
657
+ // A secret's content is never read, whatever it looks like.
658
+ if (REFLECT_REFUSED_TYPES.has(parsedRef.type)) {
659
+ return unsupportedTypeFailure(options.ref, parsedRef.type, "secret material is never read or sent to an LLM", emitFailed);
660
+ }
661
+ let assetContent = options.assetContent;
662
+ if (assetContent === undefined) {
663
+ try {
664
+ const qualifiedRef = options.itemRef ?? options.ref;
665
+ const localFilePath = await findAssetFilePath(qualifiedRef, stash);
666
+ if (localFilePath && fs.existsSync(localFilePath)) {
667
+ assetContent = fs.readFileSync(localFilePath, "utf8");
668
+ }
669
+ else {
670
+ const entry = await lookup(parseRefInput(qualifiedRef));
671
+ if (entry?.filePath && fs.existsSync(entry.filePath))
672
+ assetContent = fs.readFileSync(entry.filePath, "utf8");
673
+ }
674
+ }
675
+ catch {
676
+ // An index miss is not fatal: the agent can still propose a fresh asset.
677
+ }
1212
678
  }
1213
- catch (err) {
1214
- // Reclassify cooldown/skip messages that arrive as stdout text instead of
1215
- // valid proposal JSON. These are legitimate skip signals, not parse failures,
1216
- // and should not pollute reflectFailedActions or recentErrors injection.
1217
- const stdoutText = result.stdout ?? "";
1218
- const isCooldownSignal = isStructuredCooldownSignal(stdoutText);
1219
- const reason = isCooldownSignal ? "cooldown" : "parse_error";
1220
- emitReflectFailed(reason, isCooldownSignal ? "stdout_cooldown_signal" : "parse_error", options.ref, {
1221
- ...(result.exitCode !== null ? { exitCode: result.exitCode } : {}),
1222
- ...(reflectLlmTelemetry(result) ?? {}),
1223
- });
1224
- return {
1225
- failure: {
1226
- schemaVersion: 2,
1227
- ok: false,
1228
- reason,
1229
- error: err instanceof Error ? err.message : String(err),
1230
- ...(options.ref ? { ref: options.ref } : {}),
1231
- engine: engineName,
1232
- exitCode: result.exitCode,
1233
- stdout: result.stdout,
1234
- ...(result.stderr ? { stderr: result.stderr } : {}),
1235
- },
1236
- };
679
+ if (!REFLECT_ALLOWED_TYPES.has(parsedRef.type) &&
680
+ (assetContent === undefined || parseFrontmatter(assetContent).frontmatter === null)) {
681
+ return unsupportedTypeFailure(options.ref, parsedRef.type, "its content is not frontmatter + markdown", emitFailed);
1237
682
  }
1238
- }
1239
- function isReflectQualityGateEnabled(activeStrategy) {
1240
- return ((activeStrategy?.processes?.reflect?.qualityGate?.enabled ?? false) ||
1241
- (activeStrategy?.processes?.distill?.qualityGate?.enabled ?? true));
1242
- }
1243
- /** Resolve the exact judge transport before generation so its credential can join the operation snapshot. */
1244
- function resolveReflectQualityJudgeRunner(config, runnerSpec, enabled, onNotices) {
1245
- if (!enabled)
1246
- return Object.freeze({ enabled: false, runner: undefined });
1247
- if (runnerIsLlm(runnerSpec))
1248
- return Object.freeze({ enabled: true, runner: runnerSpec });
1249
- const resolved = resolveImproveLlmExecution({ config, processName: "reflect_proposal_quality-judge" });
1250
- if (resolved)
1251
- onNotices(resolved.notices);
1252
- return Object.freeze({ enabled: true, runner: resolved?.runner });
1253
- }
1254
- /** Acquire through genuine preparation/lowering for all runner kinds, including SDK fallback credentials. */
1255
- function acquireReflectDispatchLease(runnerSpec, onNotices) {
1256
- const prepared = prepareInlineExecutionWithRunner({
1257
- content: "Validate reflect operation transport before dispatch.",
1258
- runner: runnerSpec,
1259
- invocationKind: "direct",
1260
- });
1261
- const lowered = lowerResolvedExecutionRequestWithRunner(prepared.request, prepared.runner);
1262
- onNotices(lowered.notices);
1263
- return acquireLoweredExecutionDispatchLease(lowered);
683
+ return { assetContent, parsedRef };
1264
684
  }
1265
685
  /**
1266
- * Resolve the single named engine for a reflect invocation (standalone --engine
1267
- * / defaults.engine, or the improve strategy's LLM-only process overlay),
1268
- * throwing on any incompatible or missing engine, and validating the unattended
1269
- * LLM requirement. Extracted verbatim from `akmReflect`.
686
+ * The single engine for this invocation: `--engine`, the improve strategy's
687
+ * LLM-only reflect process, or `defaults.engine` (announced when it falls back
688
+ * to the SDK binary). Unattended improve refuses a tool-capable engine.
1270
689
  */
1271
690
  function resolveReflectRunner(options) {
1272
691
  const config = options.config ?? loadConfig();
1273
692
  const activeStrategy = options.improveProfile ?? config.improve?.strategies?.[config.defaults?.improveStrategy ?? "default"];
1274
- let runnerSpec;
1275
- let notices = [];
693
+ const lower = (selection) => {
694
+ const prepared = resolveExecution(selection);
695
+ return buildExecution(prepared.request, prepared.runner);
696
+ };
697
+ let lowered;
1276
698
  if (options.engine) {
1277
- const prepared = prepareInlineExecution({
1278
- content: "reflect engine selection",
1279
- config,
1280
- invocationKind: "direct",
1281
- current: { engine: options.engine },
1282
- });
1283
- const lowered = lowerResolvedExecutionRequest(prepared.request, prepared.config);
1284
- runnerSpec = lowered.runner;
1285
- notices = lowered.notices;
699
+ lowered = lower({ content: "reflect engine selection", config, current: { engine: options.engine } });
1286
700
  }
1287
701
  else if (options.improveProfile) {
1288
702
  const resolved = resolveImproveLlmExecution({
@@ -1294,157 +708,64 @@ function resolveReflectRunner(options) {
1294
708
  if (!resolved) {
1295
709
  throw new ConfigError("Reflect requires an LLM engine for the active improve strategy.", "LLM_NOT_CONFIGURED", "Set defaults.llmEngine or improve.strategies.<name>.processes.reflect.engine.");
1296
710
  }
1297
- runnerSpec = resolved.runner;
1298
- notices = resolved.notices;
711
+ lowered = resolved;
1299
712
  }
1300
713
  else {
1301
714
  const { config: engineConfig, fallbackEngineName } = withEngineFallback(config);
1302
715
  const defaultEngine = engineConfig.defaults?.engine;
1303
- // Announced, never silent — same contract as the workflow freeze boundary
1304
- // and the task runner. Only this arm can select the synthesized engine.
1305
- const engineAnnouncement = fallbackAnnouncement(fallbackEngineName, defaultEngine);
1306
- if (engineAnnouncement)
1307
- warn(engineAnnouncement);
716
+ const announcement = fallbackAnnouncement(fallbackEngineName, defaultEngine);
717
+ if (announcement)
718
+ warn(announcement);
1308
719
  if (!defaultEngine) {
1309
720
  throw new ConfigError(`reflect ${NO_ENGINE_MESSAGE_SUFFIX} ${NO_ENGINE_REMEDY}`, "INVALID_CONFIG_FILE");
1310
721
  }
1311
- const prepared = prepareInlineExecution({
1312
- content: "reflect engine selection",
1313
- config,
1314
- invocationKind: "direct",
1315
- });
1316
- const lowered = lowerResolvedExecutionRequest(prepared.request, prepared.config);
1317
- runnerSpec = lowered.runner;
1318
- notices = lowered.notices;
722
+ lowered = lower({ content: "reflect engine selection", config });
1319
723
  }
724
+ const runnerSpec = lowered.runner;
1320
725
  if (options.eventSource === "improve" && !runnerIsLlm(runnerSpec)) {
1321
726
  throw new ConfigError(`Unattended improve requires an LLM engine for reflect; engine "${runnerSpec.engine ?? options.engine ?? "unknown"}" is tool-capable.`, "INVALID_CONFIG_FILE", "Set defaults.llmEngine or improve.strategies.<name>.processes.reflect.engine to an LLM engine.");
1322
727
  }
1323
728
  const engineName = runnerSpec.engine ?? options.engine;
1324
- if (!engineName) {
729
+ if (!engineName)
1325
730
  throw new ConfigError("Reflect requires a named engine.", "INVALID_CONFIG_FILE");
1326
- }
1327
- return { config, activeStrategy, runnerSpec, engineName, notices };
731
+ return { config, activeStrategy, runnerSpec, engineName, notices: lowered.notices };
1328
732
  }
1329
- function unsupportedTypeFailure(ref, type, detail, emitReflectFailed) {
1330
- emitReflectFailed("unsupported_type", "unsupported_type", ref, { type });
1331
- return {
1332
- failure: {
1333
- schemaVersion: 2,
1334
- ok: false,
1335
- reason: "unsupported_type",
1336
- error: `Reflect refused: asset type "${type}" is not supported by reflect (${detail}). Use \`akm proposal new\` or edit the file directly.`,
1337
- ref,
1338
- exitCode: null,
1339
- },
1340
- };
1341
- }
1342
- /**
1343
- * Resolve the reflect target's parsed ref + current on-disk content: enforce the
1344
- * REFLECT_ALLOWED_TYPES markdown-canonical type guard (returning a terminal
1345
- * `unsupported_type` failure), honour the `options.assetContent` test seam, else
1346
- * best-effort load via the local file path / index lookup. Extracted verbatim
1347
- * from `akmReflect`.
1348
- */
1349
- async function resolveReflectSource(options, stash, emitReflectFailed) {
1350
- let assetContent;
1351
- let parsedRef;
1352
- if (options.ref) {
1353
- parsedRef = parseRefInput(options.ref);
1354
- // 2a. Refuse `secret` before any content is read — a secret's content is
1355
- // never touched by reflect, regardless of what it happens to look like.
1356
- if (REFLECT_REFUSED_TYPES.has(parsedRef.type)) {
1357
- return unsupportedTypeFailure(options.ref, parsedRef.type, "secret material is never read or sent to an LLM", emitReflectFailed);
1358
- }
1359
- if (options.assetContent !== undefined) {
1360
- // Test seam — caller pre-loaded the source content.
1361
- assetContent = options.assetContent;
1362
- }
1363
- else {
1364
- try {
1365
- // Resolve the source by item_ref when planning supplied one, otherwise
1366
- // use the input conceptId.
1367
- const qualifiedRef = options.itemRef ?? durableImproveRef(options.ref);
1368
- const localFilePath = await findAssetFilePath(qualifiedRef, stash);
1369
- if (localFilePath && fs.existsSync(localFilePath)) {
1370
- assetContent = fs.readFileSync(localFilePath, "utf8");
1371
- }
1372
- else {
1373
- const entry = await lookup(parseRefInput(qualifiedRef));
1374
- if (entry?.filePath && fs.existsSync(entry.filePath)) {
1375
- assetContent = fs.readFileSync(entry.filePath, "utf8");
1376
- }
1377
- }
1378
- }
1379
- catch {
1380
- // Index miss is non-fatal — the agent can still propose a fresh asset.
1381
- }
1382
- }
1383
- if (!REFLECT_ALLOWED_TYPES.has(parsedRef.type)) {
1384
- if (assetContent === undefined || !isReflectableSourceShape(assetContent)) {
1385
- return unsupportedTypeFailure(options.ref, parsedRef.type, "its content is not frontmatter + markdown", emitReflectFailed);
1386
- }
1387
- }
1388
- }
1389
- return { assetContent, parsedRef };
733
+ /** Lower a runner and check its credentials, so a bad transport fails before any work. */
734
+ function preflightReflectDispatch(runnerSpec, onNotices) {
735
+ const prepared = resolveExecution({
736
+ content: "Validate reflect operation transport before dispatch.",
737
+ runner: runnerSpec,
738
+ });
739
+ const lowered = buildExecution(prepared.request, prepared.runner);
740
+ onNotices(lowered.notices);
741
+ assertRunnerCredentials(lowered.runner);
1390
742
  }
1391
743
  /**
1392
- * #952 — the flat REFLECT_CONTENT_CAP (12 000 chars) exists only to avoid
1393
- * E2BIG when the prompt travels through CLI argv (agent/SDK runners). The
1394
- * direct-LLM HTTP path never touches argv, so it can use the resolved
1395
- * engine's own context window instead. The reserve for "the rest of the
1396
- * prompt" is measured directly (not guessed): build the same prompt with
1397
- * the content cap forced to zero and use its length as the overhead, so
1398
- * feedback/standards/schema-hints/prior-draft size is accounted for
1399
- * exactly, per this call. A reflect rewrite returns a body roughly the
1400
- * size of the input, so the budget only spends HALF of the usable window
1401
- * on input content and reserves the other half for the model's own
1402
- * output — otherwise a full-context request leaves no room for a
1403
- * response. Never drops below the flat floor.
1404
- *
1405
- * Shared by the real dispatch path ({@link runReflectRefineIterations}) and
1406
- * `renderReflectPromptPreview`'s `--show-prompt` preview, so the preview
1407
- * renders the exact prompt reflect would actually send for LLM runners
1408
- * instead of always the flat-cap prompt.
744
+ * The flat 12k content cap exists for CLI argv; the HTTP runner can spend half
745
+ * its context window (after the rest of the prompt) on the asset, reserving
746
+ * the other half for the rewrite. Never below the flat floor.
1409
747
  */
1410
748
  function computeReflectContentBudgetChars(promptInput, runnerSpec) {
1411
- return runnerIsLlm(runnerSpec) && promptInput.assetContent?.trim()
1412
- ? Math.max(REFLECT_CONTENT_CAP, Math.floor(((runnerSpec.connection.contextLength ?? DEFAULT_CONTEXT_LENGTH_TOKENS) * CHARS_PER_TOKEN -
1413
- buildReflectPrompt({ ...promptInput, contentBudgetChars: 0 }).prompt.length) /
1414
- 2))
1415
- : undefined;
749
+ if (!runnerIsLlm(runnerSpec) || !promptInput.assetContent?.trim())
750
+ return undefined;
751
+ const window = (runnerSpec.connection.contextLength ?? DEFAULT_CONTEXT_LENGTH_TOKENS) * CHARS_PER_TOKEN;
752
+ const overhead = buildReflectPrompt({ ...promptInput, contentBudgetChars: 0 }).prompt.length;
753
+ return Math.max(REFLECT_CONTENT_CAP, Math.floor((window - overhead) / 2));
1416
754
  }
1417
- /**
1418
- * #952 — gather every read-only prompt-input source {@link buildReflectPromptInput}
1419
- * folds into a `ReflectPromptInput`: recent feedback, schema/lint hints, related
1420
- * lessons, previously-rejected proposals, and stash standards context.
1421
- *
1422
- * Shared by the real dispatch path (`akmReflect`'s step 4, via
1423
- * {@link runReflectRefineIterations}) and `renderReflectPromptPreview`'s
1424
- * `--show-prompt` preview, so both gather from exactly one definition instead
1425
- * of two copies that can drift out of agreement.
1426
- */
1427
- async function gatherReflectPromptSources(options, stash, parsedRef, assetContent, assetCtx) {
1428
- const feedback = readRecentFeedback(options.ref ? (options.itemRef ?? durableImproveRef(options.ref)) : undefined, options.eventsCtx);
1429
- const schemaHints = buildSchemaHints(parsedRef?.type ?? "", assetContent);
1430
- const relatedLessons = options.ref && parsedRef ? await readRelatedLessons(assetCtx, stash, options.ref, parsedRef, options.itemRef) : [];
1431
- // Reflexion-style verbal-RL: inject rejected proposals so the agent avoids
1432
- // reproducing proposals that have already been reviewed and refused.
1433
- const rejectedProposals = readRejectedProposals(stash, options.ref, options.ctx);
1434
- // Standards "rulebook" for this target — stash convention/meta facts; empty
1435
- // when none fire.
1436
- const standardsContext = resolveStandardsContext(options.ref, stash);
1437
- return { feedback, schemaHints, relatedLessons, rejectedProposals, standardsContext };
755
+ /** Every read-only prompt input, shared by dispatch and `--show-prompt`. */
756
+ async function gatherReflectPromptSources(options, stash, parsedRef, assetContent) {
757
+ return {
758
+ feedback: readRecentFeedback(options.ref ? (options.itemRef ?? options.ref) : undefined, options.eventsCtx),
759
+ schemaHints: buildSchemaHints(parsedRef?.type ?? "", assetContent),
760
+ relatedLessons: options.ref && parsedRef
761
+ ? await readRelatedLessons(stash, options.ref, parsedRef, options.itemRef, options.eventsCtx)
762
+ : [],
763
+ rejectedProposals: rejectedProposalContext(stash, options.ref, options.ctx),
764
+ standardsContext: resolveStandardsContext(options.ref, stash),
765
+ };
1438
766
  }
1439
- /**
1440
- * #952 — assemble the `ReflectPromptInput` object literal reflect actually
1441
- * sends, from gathered sources plus the per-call values (draft path, prior
1442
- * draft). Shared by the real dispatch path ({@link runReflectRefineIterations})
1443
- * and `renderReflectPromptPreview`'s `--show-prompt` preview — including
1444
- * `avoidPatterns`, which the preview previously omitted even though a live
1445
- * improve loop passes it (recent-error context, O-5 / #378).
1446
- */
1447
- function buildReflectPromptInput(args) {
767
+ /** The exact prompt reflect sends, shared by dispatch and `--show-prompt`. */
768
+ function buildReflectPromptText(args) {
1448
769
  const { options, parsedRef, assetContent, sources, runnerSpec, draftFilePath, priorDraft } = args;
1449
770
  const { feedback, schemaHints, relatedLessons, rejectedProposals, standardsContext } = sources;
1450
771
  const outputMode = runnerIsLlm(runnerSpec)
@@ -1452,7 +773,7 @@ function buildReflectPromptInput(args) {
1452
773
  ? "json_schema"
1453
774
  : "framed_markdown"
1454
775
  : undefined;
1455
- return {
776
+ const input = {
1456
777
  ...(options.ref ? { ref: options.ref } : {}),
1457
778
  ...(parsedRef?.type ? { type: parsedRef.type } : {}),
1458
779
  ...(parsedRef?.name ? { name: parsedRef.name } : {}),
@@ -1464,133 +785,100 @@ function buildReflectPromptInput(args) {
1464
785
  ...(standardsContext.trim() ? { standardsContext } : {}),
1465
786
  ...(options.avoidPatterns && options.avoidPatterns.length > 0 ? { avoidPatterns: options.avoidPatterns } : {}),
1466
787
  ...(rejectedProposals.length > 0 ? { rejectedProposals } : {}),
1467
- // R-1: inject prior draft as self-critique target on iterations > 0
1468
788
  ...(priorDraft !== undefined ? { priorDraft } : {}),
1469
- // Issue A (#reflect-pipeline file-write contract): when the runner can
1470
- // touch the filesystem, instruct the agent to write the proposal body
1471
- // to a tmp file instead of inlining it in JSON. Avoids parse failures
1472
- // on long bodies (e.g. knowledge/systems/KOKORO_USAGE_GUIDE 8.4KB).
1473
789
  ...(draftFilePath ? { draftFilePath } : {}),
1474
790
  ...(outputMode ? { outputMode } : {}),
1475
791
  };
792
+ const contentBudgetChars = computeReflectContentBudgetChars(input, runnerSpec);
793
+ const { prompt } = buildReflectPrompt({
794
+ ...input,
795
+ ...(contentBudgetChars !== undefined ? { contentBudgetChars } : {}),
796
+ });
797
+ return { prompt, ...(outputMode ? { outputMode } : {}) };
1476
798
  }
1477
799
  /**
1478
- * Run the agent with the optional Self-Refine loop (R-1 / #372): up to
1479
- * `maxRefineIters` invocations, each injecting the prior draft as self-critique
1480
- * context and exiting early on a no-op refinement. Synthesizes per-iteration
1481
- * draft paths into `draftPathsToCleanup` (mutated) and returns the final agent
1482
- * result + last draft path. Extracted verbatim from `akmReflect`.
800
+ * Dispatch with the optional self-refine loop: up to `maxRefineIters` passes,
801
+ * each critiquing the prior draft, stopping early on an unchanged draft. The
802
+ * direct-LLM repair budget is shared across passes.
1483
803
  */
1484
804
  async function runReflectRefineIterations(args) {
1485
- const { options, parsedRef, assetContent, sources, runnerSpec, lease, agentEnv, draftPathsToCleanup, onNotices } = args;
805
+ const { run, parsedRef, assetContent, sources, agentEnv, draftPaths } = args;
806
+ const { options, runnerSpec } = run;
1486
807
  const maxRefineIters = Math.max(1, options.maxRefineIters ?? 1);
1487
- // Determine whether this dispatch can honour the file-write contract.
1488
- // Agent CLI + OpenCode SDK runners both have filesystem access; the direct
1489
- // LLM HTTP runner does NOT.
1490
- const canRunnerWriteFile = runnerSupportsFileWrite(runnerSpec);
1491
- // Initialized to a sentinel; always overwritten in the first loop iteration
1492
- // (maxRefineIters is clamped to >= 1 above).
808
+ const canWriteFile = runnerSupportsFileWrite(runnerSpec);
1493
809
  let result = {};
1494
810
  let priorDraft;
1495
811
  let lastDraftPath;
1496
812
  let repairAttempts = 0;
1497
813
  for (let iter = 0; iter < maxRefineIters; iter++) {
1498
- // Synthesize a fresh tmp path per iteration so refinement passes never
1499
- // clobber an earlier draft (and so reading back is unambiguous).
1500
- const iterDraftPath = canRunnerWriteFile ? synthesizeReflectDraftPath(options.ref) : undefined;
1501
- if (iterDraftPath) {
1502
- draftPathsToCleanup.push(iterDraftPath);
1503
- lastDraftPath = iterDraftPath;
814
+ const draftFilePath = canWriteFile ? synthesizeReflectDraftPath(options.ref) : undefined;
815
+ if (draftFilePath) {
816
+ draftPaths.push(draftFilePath);
817
+ lastDraftPath = draftFilePath;
1504
818
  }
1505
- const promptInput = buildReflectPromptInput({
819
+ const { prompt, outputMode } = buildReflectPromptText({
1506
820
  options,
1507
821
  parsedRef,
1508
822
  assetContent,
1509
823
  sources,
1510
824
  runnerSpec,
1511
- draftFilePath: iterDraftPath,
825
+ draftFilePath,
1512
826
  priorDraft,
1513
827
  });
1514
- const contentBudgetChars = computeReflectContentBudgetChars(promptInput, runnerSpec);
1515
- const { prompt } = buildReflectPrompt({
1516
- ...promptInput,
1517
- ...(contentBudgetChars !== undefined ? { contentBudgetChars } : {}),
1518
- });
1519
828
  let iterResult;
1520
829
  if (runnerIsLlm(runnerSpec)) {
1521
- // LLM HTTP runners cannot honor the file-write contract, so they return
1522
- // structured output through stdout. callStructured owns preparation,
1523
- // lowering, credential materialization, and direct transport dispatch.
1524
830
  iterResult = await runReflectViaLlm({
1525
831
  prompt,
1526
832
  runner: runnerSpec,
1527
- lease,
1528
833
  ...(Object.hasOwn(options, "timeoutMs") ? { timeoutMs: options.timeoutMs } : {}),
1529
834
  ...(options.signal ? { signal: options.signal } : {}),
1530
835
  priorDraft,
1531
836
  iteration: iter,
1532
- ...(promptInput.outputMode === "json_schema"
837
+ ...(outputMode === "json_schema"
1533
838
  ? { responseSchema: options.ref ? REFLECT_JSON_SCHEMA : REFLECT_UNSCOPED_JSON_SCHEMA }
1534
839
  : {}),
1535
- outputMode: promptInput.outputMode ?? "framed_markdown",
840
+ outputMode: outputMode ?? "framed_markdown",
1536
841
  ...(options.ref ? { targetRef: options.ref } : {}),
1537
842
  allowRepair: repairAttempts === 0,
1538
843
  ...(options.chat ? { chat: options.chat } : {}),
1539
- onNotices,
844
+ onNotices: run.notices.add,
1540
845
  });
1541
846
  }
1542
847
  else {
1543
- const conversationPriorDraft = priorDraft;
1544
- const hasConversation = conversationPriorDraft !== undefined && iter > 0;
848
+ const conversation = priorDraft !== undefined && iter > 0
849
+ ? [
850
+ { role: "user", content: prompt },
851
+ { role: "assistant", content: priorDraft },
852
+ ]
853
+ : undefined;
1545
854
  const current = {
1546
855
  ...(Object.hasOwn(options, "timeoutMs") ? { timeout: options.timeoutMs } : {}),
1547
856
  ...(Object.keys(agentEnv).length > 0 ? { environment: agentEnv } : {}),
1548
857
  };
1549
- const prepared = prepareInlineExecutionWithRunner({
1550
- content: hasConversation ? REFLECT_CRITIQUE_PROMPT : (prompt ?? ""),
1551
- ...(hasConversation
1552
- ? {
1553
- conversation: [
1554
- { role: "user", content: prompt ?? "" },
1555
- { role: "assistant", content: conversationPriorDraft },
1556
- ],
1557
- }
1558
- : {}),
858
+ const prepared = resolveExecution({
859
+ content: conversation ? REFLECT_CRITIQUE_PROMPT : prompt,
860
+ ...(conversation ? { conversation } : {}),
1559
861
  runner: runnerSpec,
1560
- invocationKind: "direct",
1561
862
  ...(Object.keys(current).length > 0 ? { current } : {}),
1562
863
  });
1563
- const lowered = lowerResolvedExecutionRequestWithRunner(prepared.request, prepared.runner);
1564
- onNotices(lowered.notices);
1565
- iterResult = await dispatchLoweredExecutionRequest(lowered, {
1566
- lease,
864
+ const lowered = buildExecution(prepared.request, prepared.runner);
865
+ run.notices.add(lowered.notices);
866
+ iterResult = await runExecution(lowered, {
1567
867
  ...(options.runSdk ? { runSdk: options.runSdk } : {}),
1568
- runOptions: {
1569
- ...(options.signal ? { signal: options.signal } : {}),
1570
- ...(options.runAgentOptions ?? {}),
1571
- },
868
+ runOptions: { ...(options.signal ? { signal: options.signal } : {}), ...(options.runAgentOptions ?? {}) },
1572
869
  });
1573
870
  }
1574
- const iterTelemetry = reflectLlmTelemetry(iterResult);
1575
- if (iterTelemetry)
1576
- repairAttempts += iterTelemetry.repairAttempts;
1577
- result = iterTelemetry
1578
- ? {
1579
- ...iterResult,
1580
- parsed: {
1581
- ...iterResult.parsed,
1582
- ...iterTelemetry,
1583
- repairAttempts,
1584
- },
1585
- }
871
+ const telemetry = reflectLlmTelemetry(iterResult);
872
+ if (telemetry)
873
+ repairAttempts += telemetry.repairAttempts;
874
+ result = telemetry
875
+ ? { ...iterResult, parsed: { ...iterResult.parsed, ...telemetry, repairAttempts } }
1586
876
  : iterResult;
1587
877
  if (!result.ok)
1588
- break; // surface failure after loop
1589
- // On success, extract the draft content for the next iteration.
1590
- // If the agent returns the same content as the prior draft, stop early
1591
- // (no-op refinement) to avoid wasting tokens on identical iterations.
878
+ break;
1592
879
  if (iter < maxRefineIters - 1) {
1593
- const nextDraft = reflectLlmPriorDraft(result) ?? result.stdout ?? "";
880
+ const priorFromLlm = parsedRecord(result)?.priorDraft;
881
+ const nextDraft = typeof priorFromLlm === "string" ? priorFromLlm : (result.stdout ?? "");
1594
882
  if (priorDraft !== undefined && nextDraft === priorDraft)
1595
883
  break;
1596
884
  priorDraft = nextDraft;
@@ -1599,354 +887,320 @@ async function runReflectRefineIterations(args) {
1599
887
  return { result, lastDraftPath };
1600
888
  }
1601
889
  /**
1602
- * WI-9.10: build one `akm reflect` invocation's {@link RunContext} purely
1603
- * from values `akmReflect` has already resolved by the time it calls this
1604
- * (stash, config, runnerSpec) plus the caller-supplied seams on `options` —
1605
- * no second config load, no new db handle. reflect has no `dryRun` option
1606
- * (it never writes source assets directly, only the proposal queue — see the
1607
- * module docblock) so `dryRun` is always `false` here. reflect also has no
1608
- * `sourceRun` option; the value below mirrors the same `reflect-${Date.now()}`
1609
- * convention already used inline at proposal creation time (see
1610
- * `createInput` further down this file), as a fresh, independent token —
1611
- * nothing yet reads `ctx.sourceRun`.
1612
- */
1613
- function buildReflectRunContext(args) {
1614
- const { options, stash, config, runnerSpec } = args;
1615
- return createRunContext({
1616
- stashDir: stash,
1617
- config,
1618
- eventsCtx: options.eventsCtx ?? {},
1619
- // Not yet wired into any proposal call site this stage (mirrors
1620
- // buildImproveRunContext's proposalsCtx comment in improve.ts).
1621
- proposalsCtx: options.ctx ?? {},
1622
- chat: options.chat,
1623
- getLlmRunner: () => (runnerIsLlm(runnerSpec) ? runnerSpec : null),
1624
- sourceRun: `reflect-${Date.now()}`,
1625
- dryRun: false,
1626
- signal: options.signal,
1627
- });
1628
- }
1629
- /**
1630
- * Build idempotent `reflect_invoked` / `reflect_completed` emitters. Invocation
1631
- * is delayed until canonical dispatch validates symbolic credentials, while
1632
- * deterministic pre-dispatch failures still close an invoke/complete pair.
1633
- *
1634
- * Fix #3 (observability 0.8.0): every failure path below MUST emit
1635
- * `reflect_completed` so observers can close the invoke/complete loop. The
1636
- * three success-side `reflect_completed` emit sites carry rich metadata
1637
- * (qualityRejected, sanitized, proposalId, etc.); the failure-side emits
1638
- * carry `{ok: false, reason}` plus the ref when known. Stable failure
1639
- * reasons line up with `AgentFailureReason`: "parse_error", "non_zero_exit",
1640
- * "cooldown", "timeout", "spawn_failed", "llm_*", plus the synthetic
1641
- * "ref_mismatch" / "enoent" / "draft_missing" subtypes for cases the agent
1642
- * surface conflates as "parse_error". Sub-reasons land in `subreason`.
890
+ * The proposal payload from a successful run: the agent's draft file
891
+ * (file-write contract, `DRAFT_WRITTEN confidence=<n>` on stdout) or the JSON
892
+ * payload on stdout.
1643
893
  */
1644
- function buildReflectEventEmitters(options) {
1645
- let invoked = false;
1646
- const emitInvoked = () => {
1647
- if (invoked)
1648
- return;
1649
- appendEvent({
1650
- eventType: "reflect_invoked",
1651
- // Key on item_ref when planning supplied one, otherwise the conceptId.
1652
- ...(options.ref ? { ref: options.itemRef ?? durableImproveRef(options.ref) } : {}),
1653
- metadata: {
1654
- ...(options.task ? { task: options.task } : {}),
1655
- ...(options.engine ? { engine: options.engine } : {}),
1656
- // Attribution tagging: stamp the eligibility lane so reflect_invoked can be
1657
- // sliced by lane downstream. See EligibilitySource.
1658
- ...(options.eligibilitySource ? { eligibilitySource: options.eligibilitySource } : {}),
1659
- },
1660
- }, options.eventsCtx);
1661
- invoked = true;
1662
- };
1663
- const emitFailed = (reason, subreason, ref, extra) => {
1664
- emitInvoked();
1665
- appendEvent({
1666
- eventType: "reflect_completed",
1667
- ...(ref ? { ref } : {}),
1668
- metadata: {
1669
- source: "reflect",
1670
- ok: false,
1671
- reason,
1672
- subreason,
1673
- ...(extra ?? {}),
894
+ function resolveReflectPayload(run, result, lastDraftPath, sensitiveValues) {
895
+ const { options } = run;
896
+ const draftFileExists = lastDraftPath !== undefined && fs.existsSync(lastDraftPath) && fs.statSync(lastDraftPath).size > 0;
897
+ const draftSignaled = /\bDRAFT_WRITTEN\b/.test(result.stdout ?? "");
898
+ if (draftSignaled && lastDraftPath && !draftFileExists) {
899
+ run.emitFailed("parse_error", "draft_missing", options.ref, exitCodeMeta(result));
900
+ return {
901
+ failure: reflectFailure(run, result, "parse_error", `Agent emitted DRAFT_WRITTEN but draft file is missing or empty (${lastDraftPath}). The file-write contract failed; either the agent's file tools are broken or the path was unwritable.`, true),
902
+ };
903
+ }
904
+ if (draftFileExists && lastDraftPath) {
905
+ const draftConfidence = extractDraftConfidence(result.stdout);
906
+ return {
907
+ payload: {
908
+ ref: options.ref ?? "",
909
+ content: redactSensitiveText(fs.readFileSync(lastDraftPath, "utf8"), sensitiveValues),
910
+ ...(draftConfidence !== undefined ? { confidence: draftConfidence } : {}),
1674
911
  },
1675
- }, options.eventsCtx);
1676
- };
1677
- return { emitInvoked, emitFailed };
1678
- }
1679
- function cleanupReflectDrafts(paths) {
1680
- for (const draftPath of paths) {
1681
- try {
1682
- if (fs.existsSync(draftPath))
1683
- fs.unlinkSync(draftPath);
1684
- }
1685
- catch {
1686
- // Draft cleanup is best-effort; the proposal result remains authoritative.
1687
- }
912
+ };
1688
913
  }
1689
- }
1690
- function validateReflectPayloadRef(args) {
1691
- const { payload, result, options, engineName, emitReflectFailed, executionNotices } = args;
1692
- if (!options.ref)
1693
- return undefined;
1694
914
  try {
1695
- const expectedParsed = parseRefInput(options.ref);
1696
- const actualParsed = parseRefInput(payload.ref);
1697
- if (expectedParsed.type === actualParsed.type && expectedParsed.name === actualParsed.name)
1698
- return undefined;
1699
- emitReflectFailed("parse_error", "ref_mismatch", options.ref, {
1700
- expectedRef: options.ref,
1701
- actualRef: payload.ref,
1702
- ...(result.exitCode !== null ? { exitCode: result.exitCode } : {}),
915
+ return { payload: parseAgentProposalPayload(result.stdout ?? "") };
916
+ }
917
+ catch (err) {
918
+ run.emitFailed("parse_error", "parse_error", options.ref, {
919
+ ...exitCodeMeta(result),
1703
920
  ...(reflectLlmTelemetry(result) ?? {}),
1704
921
  });
1705
922
  return {
1706
- schemaVersion: 2,
1707
- ok: false,
1708
- reason: "parse_error",
1709
- error: `Agent retargeted proposal: expected ref "${options.ref}" but got "${payload.ref}". Proposal rejected to prevent silent ref hallucination.`,
1710
- ref: options.ref,
1711
- engine: engineName,
1712
- exitCode: result.exitCode,
1713
- stdout: result.stdout,
1714
- ...(result.stderr ? { stderr: result.stderr } : {}),
1715
- ...reflectNoticeFields(executionNotices),
923
+ failure: reflectFailure(run, result, "parse_error", err instanceof Error ? err.message : String(err), true),
1716
924
  };
1717
925
  }
1718
- catch {
1719
- // Malformed refs are rejected downstream by proposal validation.
1720
- return undefined;
926
+ }
927
+ const NOISE_SUBREASONS = {
928
+ noop: "reflect_skipped_noop",
929
+ cosmetic: "reflect_skipped_cosmetic",
930
+ "low-value": "reflect_skipped_low_value",
931
+ };
932
+ /**
933
+ * Sanitize, drop a no-op/cosmetic (and optionally low-value) change, judge the
934
+ * exact content that would be persisted, then mint. Size-flagged or
935
+ * truncation-leaking content skips the judge and waits for review.
936
+ */
937
+ async function finalizeReflectProposal(args) {
938
+ const { run, assetContent, result, judge, feedback } = args;
939
+ const { options } = run;
940
+ const telemetry = reflectLlmTelemetry(result) ?? {};
941
+ const sanitized = sanitizeReflectPayload({ content: args.payload.content, ...(args.payload.frontmatter ? { frontmatter: args.payload.frontmatter } : {}) }, assetContent, args.payload.ref);
942
+ const payload = {
943
+ ...args.payload,
944
+ content: sanitized.content,
945
+ ...(sanitized.frontmatter ? { frontmatter: sanitized.frontmatter } : {}),
946
+ };
947
+ if (assetContent !== undefined) {
948
+ const changeKind = classifyReflectChange(assetContent, payload.content);
949
+ if (changeKind === "noop" ||
950
+ changeKind === "cosmetic" ||
951
+ (changeKind === "low-value" && options.lowValueFilter === true)) {
952
+ run.emitFailed("no_change", NOISE_SUBREASONS[changeKind], options.ref, { changeKind, ...telemetry });
953
+ const what = changeKind === "noop"
954
+ ? "identical to the current asset (empty diff)"
955
+ : changeKind === "low-value"
956
+ ? "a low-value prose micro-rewrite (few changed tokens, no structural changes)"
957
+ : "a cosmetic-only reformat of the current asset (whitespace/fence/YAML-folding changes)";
958
+ return reflectFailure(run, result, "no_change", `Reflect skipped: proposed content for ${payload.ref} is ${what}; no proposal created.`, false);
959
+ }
1721
960
  }
961
+ const flagged = Boolean(sanitized.sizeGuardRatio || sanitized.truncationMarkerLeaked);
962
+ const judged = judge.enabled && !flagged;
963
+ if (judged) {
964
+ const verdict = await runReflectQualityJudge(run.config, payload.content, assetContent ?? "", feedback, options.chat, {
965
+ runnerSelectionFrozen: true,
966
+ ...(judge.runner ? { llmRunner: judge.runner } : {}),
967
+ ...(Object.hasOwn(options, "timeoutMs") ? { timeoutMs: options.timeoutMs } : {}),
968
+ ...(options.signal ? { signal: options.signal } : {}),
969
+ onNotices: run.notices.add,
970
+ });
971
+ if (!verdict.pass) {
972
+ if (options.ref) {
973
+ recordLedgerAttempt({ proposalsCtx: options.ctx, eventsCtx: options.eventsCtx }, {
974
+ stashDir: run.stash,
975
+ ref: options.itemRef ?? options.ref,
976
+ source: "reflect",
977
+ outcome: "quality_rejected",
978
+ detail: verdict.reason,
979
+ });
980
+ }
981
+ appendEvent({
982
+ eventType: "reflect_completed",
983
+ ref: payload.ref,
984
+ metadata: {
985
+ source: "reflect",
986
+ qualityRejected: true,
987
+ qualityScore: verdict.score,
988
+ qualityReason: verdict.reason,
989
+ ...(verdict.criteria ? { qualityCriteria: verdict.criteria } : {}),
990
+ ...telemetry,
991
+ },
992
+ }, options.eventsCtx);
993
+ return reflectFailure(run, result, "quality_rejected", `Reflect proposal quality gate rejected: score=${verdict.score}, reason="${verdict.reason}"`, false);
994
+ }
995
+ }
996
+ // A lesson reflect wrote is marked so a later reflect on the same skill does
997
+ // not read it back as independent evidence.
998
+ const frontmatter = {
999
+ ...(payload.frontmatter ?? {}),
1000
+ ...(lenientRefType(payload.ref) === "lesson" ? { derived_from_reflect: true } : {}),
1001
+ };
1002
+ const reviewReasons = [
1003
+ ...(judge.skippedNoJudge ? ["no-judge-configured"] : []),
1004
+ ...(sanitized.sizeGuardRatio ? ["reflect-size-ratio"] : []),
1005
+ ...(sanitized.truncationMarkerLeaked ? ["reflect-truncation-leak"] : []),
1006
+ ];
1007
+ const proposal = mintProposal(run.stash, options.ctx, {
1008
+ ref: payload.ref,
1009
+ ...(options.target ? { target: options.target } : {}),
1010
+ source: "reflect",
1011
+ sourceRun: `reflect-${Date.now()}`,
1012
+ payload: { content: payload.content, ...(Object.keys(frontmatter).length > 0 ? { frontmatter } : {}) },
1013
+ ...(typeof payload.confidence === "number" ? { confidence: payload.confidence } : {}),
1014
+ ...(options.eligibilitySource ? { eligibilitySource: options.eligibilitySource } : {}),
1015
+ ...(options.itemRef ? { attemptedRefs: [options.itemRef] } : {}),
1016
+ }, reviewReasons.length > 0
1017
+ ? {
1018
+ review: {
1019
+ reason: reviewReasons.join("+"),
1020
+ gate: "reflect",
1021
+ ...(sanitized.sizeGuardRatio ? { measured: Math.round(sanitized.sizeGuardRatio.ratio * 100) } : {}),
1022
+ },
1023
+ }
1024
+ : { judged });
1025
+ appendEvent({
1026
+ eventType: "reflect_completed",
1027
+ ref: proposal.ref,
1028
+ metadata: {
1029
+ proposalId: proposal.id,
1030
+ source: "reflect",
1031
+ engine: run.engineName,
1032
+ ...(judge.skippedNoJudge ? { qualityGateSkippedNoJudge: true } : {}),
1033
+ ...(sanitized.sizeGuardRatio
1034
+ ? { sizeGuardRatio: sanitized.sizeGuardRatio.code, sizeGuardRatioValue: sanitized.sizeGuardRatio.ratio }
1035
+ : {}),
1036
+ ...(sanitized.truncationMarkerLeaked ? { truncationMarkerLeaked: true } : {}),
1037
+ ...telemetry,
1038
+ },
1039
+ }, options.eventsCtx);
1040
+ return {
1041
+ schemaVersion: 2,
1042
+ ok: true,
1043
+ proposal,
1044
+ ref: proposal.ref,
1045
+ engine: run.engineName,
1046
+ durationMs: result.durationMs,
1047
+ ...run.notices.fields(),
1048
+ };
1722
1049
  }
1723
1050
  /**
1724
- * #952 — render the composed reflect prompt for exactly one asset with no
1725
- * engine dispatch. Reuses every read-only step `akmReflect` performs before
1726
- * {@link buildReflectPrompt} (source resolution, runner resolution, feedback /
1727
- * schema-hint / related-lesson / rejected-proposal gathering) and stops right
1728
- * there: no dispatch lease is acquired, no request is sent, and — because the
1729
- * `emitReflectFailed` callback passed to {@link resolveReflectSource} here is
1730
- * a no-op — no `reflect_invoked`/`reflect_completed` event is appended either.
1731
- *
1732
- * `akm improve <ref> --show-prompt` (`improve-cli.ts`) is the CLI surface: a
1733
- * field operator uses it to see the exact prompt reflect would send, in
1734
- * seconds, without running a full improve cycle or needing a reachable
1735
- * engine.
1051
+ * `akm improve <ref> --show-prompt`: the exact prompt reflect would send for
1052
+ * one asset. Read-only: no credential, no dispatch, no event.
1736
1053
  */
1737
1054
  export async function renderReflectPromptPreview(options) {
1738
1055
  if (!options.ref) {
1739
1056
  throw new UsageError("renderReflectPromptPreview requires options.ref.", "INVALID_FLAG_VALUE");
1740
1057
  }
1741
1058
  const ref = options.ref;
1742
- const stash = resolveRunStashDir(options.stashDir);
1743
- const sourceResolved = await resolveReflectSource(options, stash, () => {
1744
- // No event emitted: this is a read-only preview, not a real invocation.
1745
- });
1746
- if ("failure" in sourceResolved) {
1747
- const { failure } = sourceResolved;
1059
+ const stash = options.stashDir ?? resolveStashDir();
1060
+ const source = await resolveReflectSource(options, stash, () => { });
1061
+ if ("failure" in source) {
1062
+ const { failure } = source;
1748
1063
  throw new UsageError((!failure.ok && failure.error) || `Reflect cannot preview ref "${ref}".`, "INVALID_FLAG_VALUE");
1749
1064
  }
1750
- const { assetContent, parsedRef } = sourceResolved;
1751
1065
  const { runnerSpec, engineName } = resolveReflectRunner(options);
1752
- const ctx = buildReflectRunContext({ options, stash, config: options.config ?? loadConfig(), runnerSpec });
1753
- const assetCtx = ctx.withFreshAssetMemo();
1754
- const sources = await gatherReflectPromptSources(options, stash, parsedRef, assetContent, assetCtx);
1755
- const canRunnerWriteFile = runnerSupportsFileWrite(runnerSpec);
1756
- // Same tmp-path synthesis a real dispatch would use (Issue A) — never
1757
- // written to, since this preview never runs the agent.
1758
- const draftFilePath = canRunnerWriteFile ? synthesizeReflectDraftPath(ref) : undefined;
1759
- const previewPromptInput = buildReflectPromptInput({
1066
+ const sources = await gatherReflectPromptSources(options, stash, source.parsedRef, source.assetContent);
1067
+ const { prompt } = buildReflectPromptText({
1760
1068
  options,
1761
- parsedRef,
1762
- assetContent,
1069
+ parsedRef: source.parsedRef,
1070
+ assetContent: source.assetContent,
1763
1071
  sources,
1764
1072
  runnerSpec,
1765
- draftFilePath,
1073
+ // The same tmp-path shape a dispatch would use; never written.
1074
+ draftFilePath: runnerSupportsFileWrite(runnerSpec) ? synthesizeReflectDraftPath(ref) : undefined,
1766
1075
  priorDraft: undefined,
1767
1076
  });
1768
- // #952 — mirror the real dispatch path's context-aware content budget (see
1769
- // computeReflectContentBudgetChars) so the preview shows the exact prompt
1770
- // reflect would send: an LLM engine with a large context window gets the
1771
- // full asset with no truncation marker, not the flat 12 000-char cap.
1772
- const contentBudgetChars = computeReflectContentBudgetChars(previewPromptInput, runnerSpec);
1773
- const { prompt } = buildReflectPrompt({
1774
- ...previewPromptInput,
1775
- ...(contentBudgetChars !== undefined ? { contentBudgetChars } : {}),
1776
- });
1777
1077
  return { ref, prompt, engine: engineName, engineKind: runnerSpec.kind };
1778
1078
  }
1779
1079
  export async function akmReflect(options = {}) {
1780
- const stash = resolveRunStashDir(options.stashDir);
1781
- // Build lazy event emitters. The invocation row is committed only after the
1782
- // canonical dispatch has validated symbolic credentials; deterministic
1783
- // pre-dispatch skips still emit it through emitReflectFailed.
1784
- const { emitInvoked: emitReflectInvoked, emitFailed: emitReflectFailed } = buildReflectEventEmitters(options);
1785
- // 2. Resolve target asset content (if a ref is supplied).
1786
- const sourceResolved = await resolveReflectSource(options, stash, emitReflectFailed);
1787
- if ("failure" in sourceResolved)
1788
- return sourceResolved.failure;
1789
- const { assetContent, parsedRef } = sourceResolved;
1790
- // 3. Resolve exactly one named engine. Standalone reflect uses --engine or
1791
- // defaults.engine; improve resolves its LLM-only strategy/process overlay.
1792
- // An incompatible explicit engine is an error and never falls through.
1080
+ const stash = options.stashDir ?? resolveStashDir();
1081
+ const { emitInvoked, emitFailed } = reflectEmitters(options);
1082
+ const source = await resolveReflectSource(options, stash, emitFailed);
1083
+ if ("failure" in source)
1084
+ return source.failure;
1085
+ const { assetContent, parsedRef } = source;
1793
1086
  const { config, activeStrategy, runnerSpec, engineName, notices: resolutionNotices } = resolveReflectRunner(options);
1794
- const executionNotices = new Map();
1795
- collectLoweringNotices(executionNotices, resolutionNotices);
1796
- const collectExecutionNotices = (notices) => collectLoweringNotices(executionNotices, notices);
1797
- let qualityJudgeSelection = resolveReflectQualityJudgeRunner(config, runnerSpec, isReflectQualityGateEnabled(activeStrategy), collectExecutionNotices);
1798
- const qualityGateSkippedNoJudge = qualityJudgeSelection.enabled && !qualityJudgeSelection.runner;
1799
- if (qualityGateSkippedNoJudge) {
1087
+ const notices = noticeSet();
1088
+ notices.add(resolutionNotices);
1089
+ const run = { options, stash, config, runnerSpec, engineName, notices, emitInvoked, emitFailed };
1090
+ // Judge selection is frozen before dispatch so a missing judge credential fails first.
1091
+ const judgeWanted = (activeStrategy?.processes?.reflect?.qualityGate?.enabled ?? false) ||
1092
+ (activeStrategy?.processes?.distill?.qualityGate?.enabled ?? true);
1093
+ let judgeRunner;
1094
+ if (judgeWanted) {
1095
+ if (runnerIsLlm(runnerSpec)) {
1096
+ judgeRunner = runnerSpec;
1097
+ }
1098
+ else {
1099
+ const resolved = resolveImproveLlmExecution({ config, processName: "reflect_proposal_quality-judge" });
1100
+ if (resolved)
1101
+ notices.add(resolved.notices);
1102
+ judgeRunner = resolved?.runner;
1103
+ }
1104
+ }
1105
+ const skippedNoJudge = judgeWanted && !judgeRunner;
1106
+ if (skippedNoJudge) {
1800
1107
  warnOnce("reflect-quality-gate-no-judge", "Reflect proposal quality gate has no LLM configured to judge proposals (set defaults.llmEngine). Skipping the gate for this run; the proposal is queued for human review instead.");
1801
- qualityJudgeSelection = Object.freeze({ enabled: false, runner: undefined });
1802
1108
  }
1803
- const qualityJudgeRunner = qualityJudgeSelection.runner;
1804
- let generationLease;
1805
- let qualityJudgeLease;
1109
+ preflightReflectDispatch(runnerSpec, notices.add);
1110
+ if (judgeRunner && judgeRunner !== runnerSpec)
1111
+ preflightReflectDispatch(judgeRunner, notices.add);
1112
+ const sources = await gatherReflectPromptSources(options, stash, parsedRef, assetContent);
1113
+ const agentEnv = options.eventSource === "improve" ? { AKM_EVENT_SOURCE: "improve" } : {};
1114
+ const sensitiveValues = collectDispatchSensitiveValues(runnerSpec, {
1115
+ ...(Object.keys(agentEnv).length > 0 ? { env: agentEnv } : {}),
1116
+ ...(options.runAgentOptions ?? {}),
1117
+ });
1118
+ const draftPaths = [];
1119
+ let result;
1120
+ let payload;
1806
1121
  try {
1807
- generationLease = acquireReflectDispatchLease(runnerSpec, collectExecutionNotices);
1808
- qualityJudgeLease =
1809
- qualityJudgeRunner === runnerSpec
1810
- ? generationLease
1811
- : qualityJudgeRunner
1812
- ? acquireReflectDispatchLease(qualityJudgeRunner, collectExecutionNotices)
1813
- : undefined;
1814
- // WI-9.10: RunContext, built only once config/runnerSpec exist so engine
1815
- // resolution's existing error-priority ordering is undisturbed (see
1816
- // buildReflectRunContext's docblock). D6: assetCtx is a fresh,
1817
- // per-invocation memo — readRelatedLessons below is its genuine
1818
- // content-read consumer.
1819
- const ctx = buildReflectRunContext({ options, stash, config, runnerSpec });
1820
- const assetCtx = ctx.withFreshAssetMemo();
1821
- // 4. Build the shared prompt inputs — feedback, hints, lessons, rejected
1822
- // proposals. These are stable across refinement iterations; only the
1823
- // `priorDraft` field changes per-iteration (R-1 / #372).
1824
- const sources = await gatherReflectPromptSources(options, stash, parsedRef, assetContent, assetCtx);
1825
- // 5. Spawn the agent — with the optional Self-Refine loop (R-1 / #372),
1826
- // extracted to {@link runReflectRefineIterations}.
1827
- const agentEnv = options.eventSource === "improve" ? { AKM_EVENT_SOURCE: "improve" } : {};
1828
- const sensitiveValues = collectDispatchSensitiveValues(runnerSpec, {
1829
- ...(Object.keys(agentEnv).length > 0 ? { env: agentEnv } : {}),
1830
- ...(options.runAgentOptions ?? {}),
1831
- });
1832
- const draftPathsToCleanup = [];
1833
- // `result` / `lastDraftPath` / `payload` are populated inside the try. Hoisted
1834
- // here so the post-try sections (R-3 ref guard, sanitizer, quality gate,
1835
- // createProposal) can use them after the drafts have been cleaned up.
1836
- let result = {};
1837
- let lastDraftPath;
1838
- let payload;
1839
- try {
1840
- const iterated = await runReflectRefineIterations({
1841
- options,
1842
- parsedRef,
1843
- assetContent,
1844
- sources,
1845
- runnerSpec,
1846
- lease: generationLease,
1847
- agentEnv,
1848
- draftPathsToCleanup,
1849
- onNotices: collectExecutionNotices,
1850
- });
1851
- emitReflectInvoked();
1852
- result = iterated.result;
1853
- lastDraftPath = iterated.lastDraftPath;
1854
- const finalResult = result;
1855
- if (!finalResult.ok) {
1856
- // B3: ENOENT / not-found gives an actionable hint.
1857
- if (isEnoentFailure(finalResult)) {
1858
- emitReflectFailed("spawn_failed", "enoent", options.ref, {
1859
- ...(finalResult.exitCode !== undefined ? { exitCode: finalResult.exitCode } : {}),
1860
- });
1861
- return {
1862
- ...failureEnvelope(finalResult, options.ref, engineName),
1863
- error: enoentHintMessage(runnerIsLlm(runnerSpec) ? engineName : runnerSpec.profile.bin),
1864
- ...reflectNoticeFields(executionNotices),
1865
- };
1866
- }
1867
- const envelope = failureEnvelope(finalResult, options.ref, engineName);
1868
- emitReflectFailed(envelope.reason, envelope.reason === "parse_error" ? "parse_error" : "agent_crash", options.ref, {
1869
- ...(envelope.exitCode !== null ? { exitCode: envelope.exitCode } : {}),
1870
- ...(reflectLlmTelemetry(finalResult) ?? {}),
1122
+ const iterated = await runReflectRefineIterations({ run, parsedRef, assetContent, sources, agentEnv, draftPaths });
1123
+ emitInvoked();
1124
+ result = iterated.result;
1125
+ if (!result.ok) {
1126
+ if (isEnoentFailure(result)) {
1127
+ emitFailed("spawn_failed", "enoent", options.ref, {
1128
+ ...(result.exitCode !== undefined ? { exitCode: result.exitCode } : {}),
1871
1129
  });
1872
- return { ...envelope, ...reflectNoticeFields(executionNotices) };
1873
- }
1874
- // Re-alias to `result` for the downstream code that references it.
1875
- result = finalResult;
1876
- const resolved = resolveReflectPayload({
1877
- result,
1878
- lastDraftPath,
1879
- sensitiveValues,
1880
- options,
1881
- engineName,
1882
- emitReflectFailed,
1883
- });
1884
- if ("failure" in resolved) {
1885
- return { ...resolved.failure, ...reflectNoticeFields(executionNotices) };
1130
+ return {
1131
+ ...baseFailureFields(result),
1132
+ schemaVersion: 2,
1133
+ ...(options.ref ? { ref: options.ref } : {}),
1134
+ engine: engineName,
1135
+ error: enoentHintMessage(runnerIsLlm(runnerSpec) ? engineName : runnerSpec.profile.bin),
1136
+ ...notices.fields(),
1137
+ };
1886
1138
  }
1887
- payload = resolved.payload;
1888
- }
1889
- catch (error) {
1890
- if (!(error instanceof ConfigError))
1891
- emitReflectInvoked();
1892
- throw error;
1893
- }
1894
- finally {
1895
- // Always remove tmp draft files — success, failure, or exception. Returns
1896
- // inside the try above trigger this block before the function exits. Code
1897
- // after this point uses the already-loaded `payload` and never touches the
1898
- // draft paths.
1899
- cleanupReflectDrafts(draftPathsToCleanup);
1900
- }
1901
- const unsafeContent = generatedContentRejection(payload.content, redactSensitiveText(payload.content, sensitiveValues));
1902
- if (unsafeContent) {
1903
- emitReflectFailed("parse_error", "parse_error", options.ref, {
1904
- ...(result.exitCode !== null ? { exitCode: result.exitCode } : {}),
1905
- });
1906
- return {
1139
+ const envelope = {
1140
+ ...baseFailureFields(result),
1907
1141
  schemaVersion: 2,
1908
- ok: false,
1909
- reason: "parse_error",
1910
- error: unsafeContent,
1911
1142
  ...(options.ref ? { ref: options.ref } : {}),
1912
1143
  engine: engineName,
1913
- exitCode: result.exitCode,
1914
- ...reflectNoticeFields(executionNotices),
1915
1144
  };
1145
+ emitFailed(envelope.reason, envelope.reason === "parse_error" ? "parse_error" : "agent_crash", options.ref, {
1146
+ ...(envelope.exitCode !== null ? { exitCode: envelope.exitCode } : {}),
1147
+ ...(reflectLlmTelemetry(result) ?? {}),
1148
+ });
1149
+ return { ...envelope, ...notices.fields() };
1916
1150
  }
1917
- const refFailure = validateReflectPayloadRef({
1918
- payload,
1919
- result,
1920
- options,
1921
- engineName,
1922
- emitReflectFailed,
1923
- executionNotices,
1924
- });
1925
- if (refFailure)
1926
- return refFailure;
1927
- const finalized = await finalizeReflectProposal({
1928
- payload,
1929
- assetContent,
1930
- result,
1931
- options,
1932
- engineName,
1933
- config,
1934
- qualityGateEnabled: qualityJudgeSelection.enabled,
1935
- qualityGateSkippedNoJudge,
1936
- qualityJudgeRunner,
1937
- qualityJudgeLease,
1938
- feedback: sources.feedback,
1939
- stash,
1940
- emitReflectFailed,
1941
- onNotices: collectExecutionNotices,
1942
- });
1943
- return { ...finalized, ...reflectNoticeFields(executionNotices) };
1151
+ const resolved = resolveReflectPayload(run, result, iterated.lastDraftPath, sensitiveValues);
1152
+ if ("failure" in resolved)
1153
+ return resolved.failure;
1154
+ payload = resolved.payload;
1155
+ }
1156
+ catch (error) {
1157
+ if (!(error instanceof ConfigError))
1158
+ emitInvoked();
1159
+ throw error;
1944
1160
  }
1945
1161
  finally {
1946
- if (qualityJudgeLease && qualityJudgeLease !== generationLease) {
1947
- disposeLoweredExecutionDispatchLease(qualityJudgeLease);
1162
+ for (const draftPath of draftPaths) {
1163
+ try {
1164
+ if (fs.existsSync(draftPath))
1165
+ fs.unlinkSync(draftPath);
1166
+ }
1167
+ catch {
1168
+ // best-effort
1169
+ }
1170
+ }
1171
+ }
1172
+ const unsafeContent = generatedContentRejection(payload.content, redactSensitiveText(payload.content, sensitiveValues));
1173
+ if (unsafeContent) {
1174
+ emitFailed("parse_error", "parse_error", options.ref, exitCodeMeta(result));
1175
+ return reflectFailure(run, result, "parse_error", unsafeContent, false);
1176
+ }
1177
+ // A retargeted proposal is refused (malformed refs are left to proposal validation).
1178
+ if (options.ref) {
1179
+ let retargeted = false;
1180
+ try {
1181
+ const expected = parseRefInput(options.ref);
1182
+ const actual = parseRefInput(payload.ref);
1183
+ retargeted = expected.type !== actual.type || expected.name !== actual.name;
1184
+ }
1185
+ catch {
1186
+ retargeted = false;
1187
+ }
1188
+ if (retargeted) {
1189
+ emitFailed("parse_error", "ref_mismatch", options.ref, {
1190
+ expectedRef: options.ref,
1191
+ actualRef: payload.ref,
1192
+ ...exitCodeMeta(result),
1193
+ ...(reflectLlmTelemetry(result) ?? {}),
1194
+ });
1195
+ return reflectFailure(run, result, "parse_error", `Agent retargeted proposal: expected ref "${options.ref}" but got "${payload.ref}". Proposal rejected to prevent silent ref hallucination.`, true);
1948
1196
  }
1949
- if (generationLease)
1950
- disposeLoweredExecutionDispatchLease(generationLease);
1951
1197
  }
1198
+ return finalizeReflectProposal({
1199
+ run,
1200
+ payload,
1201
+ assetContent,
1202
+ result,
1203
+ judge: { enabled: judgeWanted && !skippedNoJudge, skippedNoJudge, runner: judgeRunner },
1204
+ feedback: sources.feedback,
1205
+ });
1952
1206
  }