akm-cli 0.9.16-alpha.2 → 0.9.17-alpha.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (49) hide show
  1. package/CHANGELOG.md +495 -1
  2. package/dist/assets/prompts/consolidate-system.md +4 -11
  3. package/dist/assets/prompts/graph-extract-user-prompt.md +5 -5
  4. package/dist/commands/health/accept-rate.js +6 -0
  5. package/dist/commands/health/checks.js +54 -0
  6. package/dist/commands/health/improve-metrics.js +1 -5
  7. package/dist/commands/health/report-view-model.js +0 -1
  8. package/dist/commands/health.js +10 -0
  9. package/dist/commands/improve/consolidate/chunking.js +19 -35
  10. package/dist/commands/improve/consolidate/merge.js +6 -9
  11. package/dist/commands/improve/consolidate.js +104 -91
  12. package/dist/commands/improve/distill/promote-memory.js +40 -2
  13. package/dist/commands/improve/distill/quality-gate.js +186 -23
  14. package/dist/commands/improve/distill.js +42 -8
  15. package/dist/commands/improve/eligibility.js +13 -3
  16. package/dist/commands/improve/improve-cli.js +32 -9
  17. package/dist/commands/improve/improve-strategies.js +23 -1
  18. package/dist/commands/improve/improve.js +121 -84
  19. package/dist/commands/improve/loop-stages.js +241 -108
  20. package/dist/commands/improve/preparation.js +50 -17
  21. package/dist/commands/improve/reflect.js +16 -5
  22. package/dist/commands/improve/shared.js +0 -10
  23. package/dist/commands/proposal/drain.js +79 -10
  24. package/dist/commands/proposal/proposal-types.js +21 -0
  25. package/dist/commands/proposal/repository.js +108 -29
  26. package/dist/core/asset/frontmatter.js +106 -1
  27. package/dist/core/config/schema/improve-processes.js +29 -2
  28. package/dist/core/improve-result.js +9 -0
  29. package/dist/core/paths.js +7 -0
  30. package/dist/indexer/ensure-index.js +52 -7
  31. package/dist/indexer/graph/graph-extraction.js +82 -8
  32. package/dist/indexer/passes/memory-inference.js +16 -1
  33. package/dist/llm/client.js +16 -2
  34. package/dist/llm/graph-extract.js +162 -18
  35. package/dist/output/html-render.js +2 -1
  36. package/dist/output/stdout.js +24 -0
  37. package/dist/output/text.js +4 -3
  38. package/dist/scripts/akm-migrate-node.js +20 -4
  39. package/dist/scripts/akm-migrate.js +20 -4
  40. package/dist/storage/repositories/index-entries-repository.js +43 -0
  41. package/dist/storage/repositories/proposals-repository.js +4 -1
  42. package/dist/storage/state-db-integrity.js +123 -0
  43. package/dist/workflows/program/schema.js +1 -0
  44. package/docs/reference/cli.md +4 -3
  45. package/docs/reference/data-and-telemetry.md +1 -0
  46. package/package.json +1 -1
  47. package/schemas/akm-config.json +44 -0
  48. package/schemas/akm-workflow.json +1 -0
  49. package/dist/commands/improve/eval-cases.js +0 -52
package/CHANGELOG.md CHANGED
@@ -6,7 +6,501 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/).
6
6
 
7
7
  ## [Unreleased]
8
8
 
9
- ## [0.9.16-alpha.2] - 2026-09-21
9
+ ## [0.9.17-alpha.1] - 2026-09-24
10
+
11
+ ### Added
12
+
13
+ - **`akm improve --require-engines` now records its reachability probe on the
14
+ run result (R17).** `assertRequiredEnginesReachable` only ever reported a
15
+ failure (abort, exit 78); a probe that passed — including a slow or
16
+ flapping gateway that still answered in time — left no trace once the run
17
+ proceeded. It now returns one outcome per probed target (`process`,
18
+ `engine`, `endpoint`, `reachable`, `latencyMs`), threaded through a new
19
+ `AkmImproveOptions.engineProbe` and copied onto the persisted result as
20
+ `AkmImproveResult.engineProbe`. Omitted entirely when `--require-engines`
21
+ was not passed; a result persisted without it (every run before this
22
+ change) still decodes. `--require-engines --dry-run` results carry it too.
23
+ - **Reflect had no way to exclude raw wiki-ingest snapshots, which are the
24
+ longest generations in the ledger (89.5s/161.8s observed).** `wikis/articles/raw/*.md`
25
+ website snapshots index as `knowledge/wikis/articles/raw/<slug>`, and
26
+ reflect's `allowedTypes` filter is type-only, so it can't exclude a subset
27
+ of the `knowledge` type. `processes.reflect` now accepts an optional
28
+ `excludeRefPrefixes: string[]` — conceptId prefixes, matched after
29
+ stripping an optional `bundle//` from both the ref and each prefix.
30
+ `shouldSkipRef` skips a matching ref with reason `exclude-filter`, for
31
+ reflect only (distill and consolidate are memory-only and reject the key).
32
+ A trailing `/` on a prefix is ignored, so
33
+ `"knowledge/wikis/articles/raw/"` excludes the same refs as
34
+ `"knowledge/wikis/articles/raw"`.
35
+
36
+ - **`akm health` now checks state.db's own SQLite integrity.** A new hard
37
+ `state-db-integrity` check runs a read-only `PRAGMA quick_check` against
38
+ `state.db` and fails, naming the returned diagnostic lines and the repair
39
+ steps (back up, dump/restore via `sqlite3`, verify, swap in), when it
40
+ reports anything other than `ok`. The same check reports state.db's
41
+ freelist ratio (the fraction of pages `VACUUM` could reclaim) and warns
42
+ above 50%. Previously nothing in `akm health` looked past a successful
43
+ append/read round trip, which stays true on a database that is corrupt at
44
+ the SQLite level.
45
+ - **The retention purge (`akm improve`) now VACUUMs state.db when more than
46
+ half its pages are free**, immediately after the events/improve_runs/
47
+ cycle-metrics purge, recording a `state_db_vacuumed` event with pages
48
+ before/after. Opportunistic: a locked/busy database is skipped, not
49
+ raised, so it never fails the purge pass it follows.
50
+
51
+ ### Changed
52
+
53
+ - **The orphan-state GC pass no longer probes index.db once per pending
54
+ row.** `runOrphanStateGcPass` used to call `getEntryByRef` (up to two
55
+ statements each, via its bare-ref fallback) for every pending
56
+ `asset_salience` / `asset_outcome` row — 2,101 pending rows cost 83–100s
57
+ per run. It now builds one snapshot of every live `item_ref` in index.db up
58
+ front and matches every pending row against it in memory: O(1) index.db
59
+ queries per run instead of one probe per row, with the same live/orphan
60
+ resolution (including the bundle-qualified-exact and bare-conceptId-suffix
61
+ fallback) as before.
62
+ - **Memory inference no longer forces a full reindex for the file(s) it
63
+ writes.** The post-inference maintenance step used to call the full
64
+ `reindexFn` (42–220s per run, typically for one written derived fact)
65
+ whenever memory inference split a parent. `runMemoryInferencePass` now
66
+ reports the exact paths it wrote or rewrote (`writtenPaths`, sourced from
67
+ the run's write-provenance journal), and the maintenance pass indexes just
68
+ those files with `indexWrittenAssets` instead — closing and reopening the
69
+ shared index.db handle around the call with the same discipline the full
70
+ reindex used (#584). The separate post-consolidation full reindex is
71
+ removed outright rather than re-gated: it used to fire whenever
72
+ `consolidation.processed > 0` (memories the LLM judged), but
73
+ merge/delete/contradict ops are advisory and never auto-applied, and the
74
+ one op that does execute — promote — writes a proposal to state.db, not to
75
+ the stash. Consolidation therefore cannot change a file the index reads,
76
+ so the reindex had no precondition it could ever satisfy.
77
+ - **The improve loop's reflect dispatch now checks the proposal
78
+ fingerprint/rejection-backoff guard *before* calling reflect, not just
79
+ after.** `fingerprint_match` and `rejection_backoff` were evaluated only
80
+ inside `createProposal`, which runs after reflect's full generation and
81
+ quality-judge call — so a ref already guaranteed to be skipped still paid
82
+ the LLM cost (measured: 2–16% of reflect LLM seconds spent on refs the
83
+ guard then discarded). The guard's fingerprint is an input fingerprint
84
+ (target ref, source, before-hash, model id), computable before dispatch, so
85
+ `checkProposalGuard` (`src/commands/proposal/repository.ts`) exposes the
86
+ identical check `createProposal` runs post-generation — the two share one
87
+ implementation and can never disagree. `runLoopReflectPass`
88
+ (`src/commands/improve/loop-stages.ts`) now calls it first; a hit skips
89
+ `reflectFn` entirely and lands in the existing `reflect-cooldown` bucket
90
+ with the same `reflect_invoked` event the signal-delta cursor
91
+ (`buildLatestProposalTsMap`) reads, so cursor advancement and run-result
92
+ classification are unchanged. `createProposal`'s post-generation check
93
+ remains the authoritative gate.
94
+ - **Consolidate's plan schema and prompt are promote-only.** The apply loop
95
+ only ever executed `promote` — `merge`/`delete`/`contradict` were advisory
96
+ by design and never applied — but the schema still asked for all four ops
97
+ plus a free-text `warnings` array, and completion tokens rose from 7–8k to
98
+ 21–30k per run after the 35B-A3B model switch with no change in
99
+ promotions. `CONSOLIDATE_PLAN_JSON_SCHEMA` and `consolidate-system.md` now
100
+ request only `promote` (with `reason` capped at 200 chars), and `isValidOp`
101
+ rejects any other op shape — e.g. from a model that ignores the schema —
102
+ with the existing "skipping invalid operation" warning instead of treating
103
+ it as an actionable plan entry. `ConsolidateResult.merged` / `deleted` /
104
+ `contradicted` and the `planned` op breakdown are unchanged in shape and
105
+ stay zero.
106
+ - **`improve-maintenance-passes.test.ts` moved under `tests/integration/`.**
107
+ The suite opens a real `state.db` via `openStateDatabase`, which AGENTS.md's
108
+ ORG-03..06 rule places under `tests/integration/`, not `tests/`; no content
109
+ change. Also corrected
110
+ `docs/architecture/specs/improve-collapse-churn-detector-design.md` §2.5,
111
+ which described the post-loop collapse-detector gate as `consolidationRan
112
+ OR recombination.processed > 0` — no `recombination` value is plumbed into
113
+ `runImprovePostLoopStage` and no recombine pass exists in the codebase, so
114
+ the spec now matches the shipped `consolidationRan`-only gate and notes
115
+ that the recombine-triggered pass is not implemented.
116
+ - **Graph-extraction relations are now compact `[from, type, to]` triples
117
+ instead of `{"from","to","type"}` objects, and the batch graph-extraction
118
+ call now sends a `responseSchema`.** The object-keyed form cost 10+ tokens
119
+ per relation for no signal, and completion tokens cost far more than
120
+ prompt tokens; a compact triple form measured −51% / −18% completion
121
+ tokens on two chunks. `graph-extract.ts`'s single-asset and batch prompts
122
+ and JSON schemas now ask for `["from", "type", "to"]` (`type` may be `""`);
123
+ `parseGraphExtraction` accepts both the triple form and the legacy object
124
+ form (a relation-level `confidence` is still read from a legacy object,
125
+ though the schema no longer offers it — the prompt never asked for one).
126
+ Separately, production runs graph extraction batched
127
+ (`processes.graphExtraction.batchSize`), and `extractGraphFromBodies` sent
128
+ no `responseSchema` at all, so the R12b output-bounding schema only ever
129
+ reached the single-asset path. The batch call now sends the same
130
+ `maxItems`-bounded schema (scoped to the batch's asset count) through the
131
+ same `supportsJsonSchema`-gated `responseSchema` field the single-asset
132
+ call uses. `GRAPH_EXTRACT_PROMPT_VERSION` bumps `v2` → `v3`, so every file
133
+ re-extracts once on the next graph pass — entity semantics, caps, chunking
134
+ and batch sizing are unchanged.
135
+
136
+ ### Removed
137
+
138
+ - **The write-only distill/proposal eval-cases path.** `writeEvalCase`
139
+ (`src/commands/improve/eval-cases.ts`) wrote a Markdown file per rejection
140
+ under `$STATE/improve/eval-cases/<stash>/` that nothing ever read back, and
141
+ `countEvalCases` reported a cumulative on-disk file count as if it were a
142
+ per-run number (surfaced as `evalCasesWritten` on the improve result and in
143
+ `akm health`'s improve metrics). A rejected proposal row (see above) now
144
+ carries the same information through a path something actually reads.
145
+ Deleted `eval-cases.ts` and its two `loop-stages.ts` call sites, the
146
+ `evalCasesWritten` field from `AkmImproveResult` and every health-metrics
147
+ reader/aggregator, and the `improve_completed` event's `evalCasesWritten`
148
+ field. `decodeImproveResult` still accepts (and ignores) `evalCasesWritten`
149
+ on an envelope an older release wrote, and existing eval-case files on disk
150
+ are untouched — `getEvalCasesDir` (`core/paths.ts`) stays, since
151
+ `scripts/akm-migrate/migrate/writer-relocation.ts` still uses it to
152
+ relocate them from the legacy `$STASH/.akm/eval-cases/` path.
153
+
154
+ ### Fixed
155
+
156
+ - **The lesson quality judge's ACTIONABILITY criterion carried no signal, and
157
+ the judge's request/parser let a differently-spelled or extra key change
158
+ the verdict (R16).** Splinter measured ACTIONABILITY at AUC 0.46 against
159
+ accept/reject outcomes — no better than chance — and averaging it into the
160
+ score pulled every verdict toward its 3.0 mode, i.e. the review band.
161
+ `buildJudgePrompt` no longer asks for it;
162
+ `LESSON_JUDGE_CRITERIA_KEYS` is now `novelty`/`nonRedundancy` only.
163
+ Separately, `runQualityJudge`'s request sent no `responseSchema` while the
164
+ prompt text spelled criteria as NON-REDUNDANCY / FEEDBACK ALIGNMENT, so a
165
+ model that echoed a differently-cased or -spelled key turned the verdict
166
+ into a parse failure routed to review; and `parseJudgeResponse` averaged
167
+ over every key present in `scores`, so an unexpected extra key changed the
168
+ score. `runQualityJudge` now sends a strict `responseSchema` — built from
169
+ the judge's own expected criteria keys, `additionalProperties: false` at
170
+ both levels — through the same `supportsJsonSchema`-gated
171
+ `request.responseSchema` path `src/llm/graph-extract.ts` uses, a no-op for
172
+ providers that don't opt in; and `parseJudgeResponse` now reads, validates,
173
+ and averages only the expected keys, silently ignoring any other key
174
+ instead of averaging or validating it. A missing expected key is still a
175
+ parse failure, unchanged.
176
+ - **The reflect quality-gate's "no judge configured" warning named a config
177
+ key nothing reads.** It told users to set
178
+ `improve.strategies.<name>.processes.reflect.qualityGate.engine`, but
179
+ `qualityGate` is `{ enabled }` passthrough — `resolveReflectQualityJudgeRunner`
180
+ always uses the generation runner when it is an LLM, or falls back to
181
+ `defaults.llmEngine` via `resolveImproveLlmExecution` with no profile/process
182
+ layer, so that key was never read. The warning now names only
183
+ `defaults.llmEngine`.
184
+ - **The distill/reflect LLM-as-judge quality gate inherited the generation
185
+ runner's temperature, and its averaged score hid which criterion actually
186
+ failed.** `runQualityJudge`'s request only pinned `enableThinking: false`,
187
+ so the judge ran at whatever temperature generation used — measured at 0.3,
188
+ the verdict flipped on 10/16 identical inputs, vs. 0/16 at temperature 0.
189
+ The request now also pins `temperature: 0`, for both the distill and
190
+ reflect judges that share this function, independent of the runner's
191
+ configured temperature. Separately, both judge prompts asked for one
192
+ averaged float, so a criterion carrying no signal was invisible in
193
+ production. They now ask for per-criterion integer scores
194
+ (`buildJudgePrompt`: novelty/actionability/nonRedundancy;
195
+ `buildReflectJudgePrompt`: feedbackAlignment/preservation/quality), averaged
196
+ in code to the same `score` the unchanged 3.5/2.5 thresholds gate on. The
197
+ parser accepts this new `{"scores": {...}, "reason"}` shape and still
198
+ accepts the old `{"score": <float>, "reason"}` shape a model may return;
199
+ each criterion (or the bare score) must be a finite number in 1..5 or the
200
+ response routes to review exactly as a parse failure does today. The
201
+ per-criterion scores, when present, are now carried through
202
+ `QualityJudgeResult.criteria` into the `distill_invoked` event metadata and
203
+ rejection-envelope frontmatter `writeQualityRejection` writes, and into
204
+ reflect's `reflect_completed` rejection event as `qualityCriteria`.
205
+ - **The judge parser accepted a partial `scores` object and auto-passed it.**
206
+ `parseJudgeResponse` validated only that whatever criterion keys arrived
207
+ held finite 1-5 values, then averaged over those keys alone — so a
208
+ truncated judge response like `{"scores": {"novelty": 5}, "reason": "…"}`
209
+ parsed to `score: 5.0` and `pass: true`, promoting content the judge never
210
+ finished evaluating on its other criteria. `runQualityJudge` now passes the
211
+ criterion key set its prompt asked for (`buildJudgePrompt`:
212
+ novelty/actionability/nonRedundancy; `buildReflectJudgePrompt`:
213
+ feedbackAlignment/preservation/quality) down to `parseJudgeResponse`, which
214
+ returns a parse failure — routed to review, exactly as a malformed response
215
+ is today — when any expected key is missing from `scores`.
216
+ - **The reflect pre-generation proposal-guard skip (R9) emitted `reflect_invoked` with no paired `reflect_completed`.** `runLoopReflectPass`'s guard-skip branch in `loop-stages.ts` appended a synthetic `reflect_invoked` event to advance the signal-delta cursor, but never called `reflectFn`, so `reflect.ts`'s own `reflect_completed` emission never ran either — a new, permanent source of unpaired `reflect_invoked` rows for every fingerprint/backoff hit, violating the invoke/complete pairing invariant `buildReflectEventEmitters` documents. The branch now also appends a matching `reflect_completed` (`ok:false`, `reason:"cooldown"`, `subreason:"pre_generation_guard"`), mirroring `emitFailed`'s shape.
217
+ - **R9's pre-generation proposal guard covered reflect only — distill paid for a full generation + judge call before the same fingerprint/backoff guard could reject it.** `runLoopDistillPass` had no equivalent of `runLoopReflectPass`'s pre-check, even though `createProposal`'s post-generation guard (and every rejected row R10 now mints under `source: "distill"`) applies to distill just as much. `runLoopDistillPass` now calls `checkProposalGuard` against the derived lesson/knowledge ref (distill's real `createProposal` call never targets the input ref) before dispatching `distillFn`; a hit routes to the pass's existing `distill-skipped` bucket and emits `distill_invoked` with a `skipped` outcome so `buildLatestProposalTsMap`'s signal cursor still advances.
218
+ - **The distill pre-generation proposal guard could suppress a legitimate dispatch by checking a ref distill would never target.** For a memory input, distill's real `createProposal` call targets one of two refs decided at dispatch time inside `planMemoryKnowledgePromotion` — the derived knowledge ref when the deterministic promotion heuristic fires, the derived lesson ref otherwise — but `runLoopDistillPass`'s pre-check checked both candidate refs and skipped on the FIRST guard hit, so a stale fingerprint/backoff hit on the ref distill would NOT have targeted silently suppressed dispatch until that ref's fingerprint happened to change. The pre-check now resolves the SAME target `planMemoryKnowledgePromotion` would via `wouldPromoteMemoryToKnowledge` (`distill/promote-memory.ts`) — a thin wrapper that delegates to `planMemoryKnowledgePromotion` itself so the classification can never drift from the real dispatch decision, with no LLM call — and checks only that ref; content and the classification's `durableInputRef` are read via `planned.ref` alone, matching `akmDistill`'s real dispatch, while `planned.itemRef ?? planned.ref` feeds only the feedback-events query.
219
+ - **The `akm improve` triage pre-pass drain's judgment LLM calls were unattributed in the usage report.** `runTriagePrePass`'s `drainProposalsFn` call dispatched judgment calls with no `withLlmStage` wrapper, unlike the standalone `akm proposal drain` CLI path, so they landed in `byProcessEngineModel` as unattributed (5 calls, 24s per run) instead of under a `triage` stage. The pre-pass drain is now wrapped in `withLlmStage("triage", …, { engine, process: "triage.judgment" })`, mirroring the CLI path.
220
+ - **The batch graph-extraction provider-storm guard only recognized one error
221
+ code.** After a failed batch call, `extractGraphFromBodies` skipped the
222
+ per-asset fallback retry only for `LlmCallError`s coded `provider_error` —
223
+ but a dead endpoint more often raises `network_error` (a dropped
224
+ connection) or `provider_html_error` (a provider serving an HTML error
225
+ page), both of which still paid the full per-asset fallback storm the
226
+ guard exists to prevent. The predicate is now `isTransportFailure`
227
+ (`src/llm/client.ts`), shared with `chatCompletion`'s retry classifier so
228
+ the two cannot drift apart, and covers `provider_error`, `network_error`,
229
+ and `provider_html_error`.
230
+ - **`akm health`'s `state-db-integrity` check no longer crashes when the freelist/page-count read fails.** `getStateDbFreelistInfo` had a `finally` but no `catch` around its read-only open and pragma reads, unlike its sibling `runStateDbQuickCheck` — a throw there (e.g. an unopenable state.db) escaped `akm health` as an unclassified exit 70 on exactly the damaged database the check exists to report. It now returns a zeroed `StateDbFreelistInfo` with an `error` field, and the check renders that as a failed check instead of throwing.
231
+ - **The post-purge VACUUM's `state_db_vacuumed` event now honors the caller's `EventsContext`.** `vacuumStateDbIfReclaimable` appended its event with a direct `insertEvent` call, bypassing `EventsContext.readOnly` and the injectable clock its sibling purge events (`events_purged`, `improve_runs_purged`, `improve_cycle_metrics_purged`) use in the same `runRetentionPurgePass` callback. It now appends the event via `appendEvent` with the caller's `EventsContext` plumbed through.
232
+ - **Consolidate's per-chunk prompt excerpt truncated the raw file (frontmatter
233
+ + body) instead of the body.** `buildChunkPrompt` sliced `body.slice(0,
234
+ bodyTruncation)` off the unstripped file; a memory whose frontmatter alone
235
+ exceeded the excerpt length was judged on metadata only and never showed
236
+ its own body text. The excerpt now truncates `stripFrontmatterBody(body)`;
237
+ hot/queued detection is unchanged and still reads the raw body.
238
+ - **Consolidate's chunk prompt carried an unused ~14k-char standards block and
239
+ a header the model sometimes echoed back as a bogus `ref`.** Every chunk
240
+ prompt resolved and injected a "Standards to follow" section
241
+ (`resolveStandardsContext("memories/_consolidated", ...)`), but the chunk
242
+ output is a promote-only op list that never reads it. Separately, the
243
+ chunk header (`Chunk N of M, memories <first>–<last>:`) named the chunk's
244
+ boundary memories with an en dash between two `memories/<name>` refs; on
245
+ 2026-09-24 the judge model returned promote ops whose `ref` was exactly
246
+ that `memories/<first>–memories/<last>` range, naming a memory that does
247
+ not exist and losing the promotion. `buildChunkPrompt` no longer takes a
248
+ `standardsContext` and the header is now
249
+ `Chunk N of M (<count> memories):` — no refs in it.
250
+ - **Consolidate re-judged memories that were already promoted verbatim into
251
+ `knowledge/`.** That duplication was previously discovered only after the
252
+ LLM (`shouldSkipPromotionBodyDuplicate`), so a pool where the large
253
+ majority of memories were already-promoted duplicates still paid the full
254
+ chunk/LLM cost on all of them before being skipped.
255
+ `inspectConsolidationPool` now drops those memories before any chunking or
256
+ LLM work, sharing one `loadExistingKnowledgeBodyHashes` call and the same
257
+ `cacheHash` domain with the post-LLM check so the two cannot disagree. The
258
+ dropped count is reported as `prefilteredAlreadyPromoted` on the
259
+ consolidate result and in a warning line. The pre-filter also now runs
260
+ *before* the `consolidate.limit` cap (previously after), so a run with a
261
+ limit set selects its oldest-modified window from the pre-filtered pool
262
+ instead of re-selecting and re-dropping the same permanently-undeletable
263
+ duplicates every run while fresh memories past the cap went unreached; the
264
+ preview/eligibility path (`preparation.ts`) computes and passes the same
265
+ hash set so the reported candidate pool agrees with what the run will act
266
+ on. A live (non-preview) `akm improve` run reuses that same hash set for
267
+ the actual `akmConsolidate` call instead of recomputing it, so a run still
268
+ walks `knowledge/` only once.
269
+ - **`improve`'s start-of-run index rescan ran after triage dirtied the stash,
270
+ not before it.** Proposal triage promotes accepted proposals straight into
271
+ the flat `knowledge/` root, and the blocking `ensureIndex` call that is
272
+ supposed to give the run a current index ran only afterward (inside
273
+ `collectEligibleRefs`'s setup), so every triage promotion guaranteed the
274
+ very full rescan it should have preceded — up to ~27 minutes, holding the
275
+ index lock against co-scheduled writers. `ensureIndex` now runs before the
276
+ triage pre-pass, and triage's own writes are indexed incrementally
277
+ (`indexWrittenAssets`) so `collectEligibleRefs` still sees them without a
278
+ second full walk. Because `indexWrittenAssets` upserts a file's
279
+ `content_hash` without bumping `builtAt`, index staleness detection
280
+ (`ensure-index.ts`) is now per-file: a file newer than the last build is
281
+ only treated as stale when its current content actually differs from what
282
+ is indexed, so incrementally-reindexed content stops re-triggering the
283
+ same full rescan on every subsequent run. The implicit reindex's timing
284
+ breakdown (walk/llm/embed/finalize), previously discarded, is now logged
285
+ at verbose level and surfaced on the improve result as `ensureIndexDurationMs`.
286
+ - **Distill quality rejections vanished instead of persisting, so backoff and
287
+ Reflexion never saw them and the same ref was re-selected and re-rejected
288
+ on every run** (two refs were rejected 11× and 10×). `writeQualityRejection`
289
+ wrote only a `$STATE`-side file and an event, never a `proposals` row, so
290
+ `rejection_backoff`/`fingerprint_match` (proposal/repository.ts) and the
291
+ Reflexion "previously rejected" context had nothing to find; the distill
292
+ signal-delta cursor (`buildLatestProposalTsMap`) also only advanced for
293
+ `queued`/`skipped`/`validation_failed` outcomes, so a rejected ref stayed
294
+ eligible forever. `writeQualityRejection` now mints a real proposal through
295
+ the same `createProposal`/`archiveProposal` path every other proposal
296
+ source uses: a `quality_rejected` outcome is minted pending then archived
297
+ to `rejected` carrying the judge's reason; a `review_needed` outcome stays
298
+ `pending` in the normal queue, where triage — a human, or the drain's
299
+ judgment tier when one is configured — decides, the same path every other
300
+ pending distill proposal (including quality-gate passes) already takes.
301
+ The cursor now also advances on both outcomes (still excluding
302
+ `llm_failed`, where no real attempt produced anything). A retry for the
303
+ same target, source, and model is skipped by `fingerprint_match` (the
304
+ input fingerprint recorded at mint, retained `archiveRetentionDays`,
305
+ default 90 days); the 30-day `rejection_backoff` window only applies once
306
+ the target's before-hash or the model differs. Because these machine
307
+ rejections are now real `rejected` rows under `source: "distill"`, `akm
308
+ health`'s distill accept rate (`computeAcceptRateBySource`,
309
+ src/commands/health/accept-rate.ts) drops relative to earlier releases and
310
+ no longer measures reviewer acceptance alone. Nothing gates on that
311
+ metric.
312
+ - **`writeQualityRejection` could throw instead of returning a rejection
313
+ result.** Minting the proposal row above runs the mint-time canonical
314
+ validator (`createProposal` → `rejectProposal`,
315
+ `src/commands/proposal/repository.ts`), which throws `UsageError` for
316
+ structurally-invalid content — e.g. a `lessons/` ref whose body lacks
317
+ `description`/`when_to_use`. `writeQualityRejection` is the terminal,
318
+ non-throwing rejection path and none of its callers handled a throw. The
319
+ proposal row is bookkeeping for backoff/Reflexion, never the authoritative
320
+ record of the rejection, so a validator throw now degrades to "no row
321
+ minted" — the envelope file and `distill_invoked` event are still written,
322
+ matching the existing fingerprint/backoff skip behavior.
323
+ - **A `review_needed` quality-gate rejection could be auto-promoted by the
324
+ triage drain's judgment tier with no human ever seeing it.**
325
+ `writeQualityRejection` minted a `review_needed` outcome as an ordinary
326
+ pending proposal under `source: "distill"` (knowledge promotions from
327
+ `promote-memory.ts` take the same path); the `personal-stash` drain policy
328
+ defers `distill` proposals to the judgment tier, which can auto-accept
329
+ under `applyMode: promote` + `experimental.improveAutonomy` — so content
330
+ the quality judge explicitly refused to auto-queue (the 2.5–3.5
331
+ review-needed band) could be promoted without a human in the loop.
332
+ `writeQualityRejection` now stamps a `review_needed` mint with a
333
+ `{ outcome: "deferred", reason: "quality-review", gate: "quality-gate" }`
334
+ gate decision (best-effort: a stamp failure warns and continues, like the
335
+ existing mint/archive tolerance), and `classifyPendingProposals`
336
+ (`proposal/drain.ts`) skips any pending row carrying it — leaving it
337
+ pending and untouched, before the drain's own policy-deferred re-stamp
338
+ loop would otherwise overwrite the stamp.
339
+ - **Consolidate's post-LLM promote-dedup hash double-stripped frontmatter.**
340
+ `shouldSkipPromotionBodyDuplicate`'s `bodyHash` was computed as
341
+ `cacheHash(parseFrontmatter(memoryContent).content.trim())` — the body was
342
+ already frontmatter-stripped before being handed to `cacheHash`, which
343
+ strips it again internally — diverging from the single-strip
344
+ `cacheHash(raw)` domain `loadExistingKnowledgeBodyHashes` and the pre-filter
345
+ use for a source memory body that begins with its own `---` block. The
346
+ check now hashes `cacheHash(memoryContent)` directly, so the two sides of
347
+ the dedup comparison agree.
348
+ - **Consolidate's per-chunk prompt still warned against proposing `delete`
349
+ for `(captureMode: hot)` memories.** The consolidate op schema and system
350
+ prompt dropped `delete` (along with `merge`/`contradict`), leaving
351
+ `buildChunkPrompt`'s top-of-prompt hot-ref block as the only remaining
352
+ mention of `delete` anywhere in the prompt — a retired op name that
353
+ `isValidOp` now rejects if the model echoes it back, wasting tokens on
354
+ "skipping invalid operation" warnings. The block and the `hotRefs`
355
+ collection that fed it are removed; the inline `(captureMode: hot)`
356
+ annotation on each memory line is unchanged.
357
+ - **Graph extraction sent no `json_schema` and no per-asset chunk cap, so a
358
+ long file could pay for dozens of LLM calls whose output was then sliced
359
+ down to the same 32-entity/32-relation limit anyway** (one file spent 21 of
360
+ 27 calls and 12.9k completion tokens this way). The single-asset extraction
361
+ call (`extractGraphFromBody`) now sends a `responseSchema` (entities/
362
+ relations capped at 32 each, `additionalProperties: false` otherwise), via
363
+ the same `supportsJsonSchema`-gated request path memory-infer.ts uses — no
364
+ `maxTokens` is sent; cost is bounded by the schema's `maxItems` caps alone,
365
+ per AGENTS.md's "LLM Defaults" (a hardcoded cap risked silent truncation
366
+ with zero headroom for JSON punctuation or reasoning tokens). A body
367
+ chunked beyond the new
368
+ `processes.graphExtraction.maxChunksPerAsset` (default 8) now stops after
369
+ the first N chunks instead of processing every one; the skipped chunks are
370
+ reported as `truncatedChunks` in the run's graph-extraction telemetry so
371
+ the coverage loss is visible rather than silently absorbed. The `improve`
372
+ loop's dispatch (`loop-stages.ts`) now also forwards a configured
373
+ `maxChunksPerAsset` to the extraction call, mirroring the existing
374
+ `topN`/`batchSize` wiring — without this the config key had no effect in a
375
+ real `akm improve` run and the default of 8 always applied.
376
+ - **The graph-extraction `responseSchema` forbade the `confidence` field the
377
+ parser itself reads.** `additionalProperties: false` on both the root
378
+ object and each relation item made `confidence` impossible on a
379
+ `supportsJsonSchema` provider, even though `parseGraphExtraction` uses
380
+ `rel.confidence` to drop relations below `MIN_RELATION_CONFIDENCE` and
381
+ `item.confidence` to feed the merged extraction confidence — silently
382
+ turning the confidence filter into dead code on exactly the providers the
383
+ schema targets. `confidence: {"type": "number"}` is now allowed at both
384
+ levels; `additionalProperties: false` still forbids anything else.
385
+ - **A pending proposal went stale the moment akm's own bookkeeping touched
386
+ its target, and promote refused it forever (R20).** `resolveProposalTargetInfo`
387
+ captured the target's raw `beforeHash` at mint; the SAME nightly run's
388
+ `writeSalienceToFrontmatter` (distill) and memory inference's
389
+ `inferenceProcessed` stamp then rewrote the target's frontmatter before
390
+ promote ran, so `promoteProposalWithLease`'s guard (`repository.ts` ~L2406)
391
+ and `drain.ts`'s dry-run mirror (`assertProposalTargetFresh`) refused every
392
+ affected proposal with "target changed after proposal was created" — the
393
+ same 11+ reflect proposals, every day, on splinter. `resolveProposalTargetInfo`
394
+ now also captures `beforeHashNormalized` (`core/asset/frontmatter.ts`'s new
395
+ `computeNormalizedContentHash`, over the target with
396
+ `BOOKKEEPING_FRONTMATTER_KEYS` — `salience`/`salienceInputs`/`inferenceProcessed`
397
+ — stripped and the remaining frontmatter canonically re-serialized); the
398
+ promote guard and its dry-run mirror both prefer it over the raw
399
+ `beforeHash` when present, so a bookkeeping-only rewrite no longer stales a
400
+ proposal out while a real content change still refuses. Promotion also now
401
+ carries the live target's bookkeeping keys forward
402
+ (`carryForwardBookkeepingFrontmatter`) when the proposal's own frontmatter
403
+ doesn't set them, so accepting never drops `inferenceProcessed` and forces
404
+ memory inference to reprocess the memory. A legacy proposal minted before
405
+ this field existed keeps its exact original raw-hash check.
406
+ `computeNormalizedContentHash` also normalizes the body boundary the same
407
+ way `assembleAssetFromString` does (leading newlines stripped, exactly one
408
+ trailing newline) before hashing, and treats an empty frontmatter block as
409
+ `{}` instead of falling back to the raw hash — both `writeSalienceToFrontmatter`
410
+ and the memory-inference `assembleAsset` rewrite shift where the body starts,
411
+ which without this normalization still staled a proposal out unless the
412
+ target's frontmatter was already in that exact on-disk shape.
413
+ - **A stale-target promote failure was retried, and refused, identically
414
+ every drain run forever (R20).** The drain already categorized a
415
+ "target changed/was created after proposal" failure as `stale-target`
416
+ (`categorizeDrainFailure`), but left the row pending either way — so the
417
+ same proposals failed the same way on every subsequent `akm proposal
418
+ drain` / triage pass. Both promote-failure sites (`drainProposals`'s
419
+ deterministic loop and `runJudgmentTier`) now auto-reject a stale-target
420
+ failure once, stamping `gateDecision: { outcome: "auto-rejected", reason:
421
+ "stale-target" }` instead of leaving it to retry. This is not a merit
422
+ rejection, so `checkFingerprintAndBackoff`'s rejection-backoff window
423
+ (`repository.ts`) now excludes stale-target rows — the ref stays
424
+ re-proposable against its current content — and the Reflexion
425
+ "previously rejected" context (`reflect.ts`'s `readRejectedProposals`,
426
+ `distill.ts`'s `buildDistillMessages`) and the accept-rate health metric
427
+ (`health/accept-rate.ts`) now exclude stale-target rejections too, so a
428
+ procedural refusal doesn't misrepresent content quality. `--dry-run` now
429
+ predicts the same outcome: a stale-target promote failure it detects is
430
+ reported under `rejected`, matching what a real run does, instead of under
431
+ `failed`.
432
+
433
+ - **Reflect quality-gate rejections were mislabelled `parse_error` and fed
434
+ back into later prompts as learned "avoid" patterns.** When the reflect
435
+ quality judge rejected an otherwise well-parsed proposal, the result
436
+ carried `reason: "parse_error"` — a real parse failure and a judge
437
+ rejection were indistinguishable. The improve loop injects non-excluded
438
+ reflect failures into the next reflect prompt's "Avoid These Patterns"
439
+ block, so a single gate rejection could poison every subsequent candidate
440
+ in the run. Judge rejections now carry a distinct `quality_rejected`
441
+ reason, stay in the `reflect-failed` metrics bucket, and are excluded from
442
+ that avoid-patterns injection like the existing deterministic skips.
443
+ - **Legacy rejected proposals no longer throw before reflect/distill prompt
444
+ dispatch.** `readRejectedProposals` (reflect.ts) and the equivalent mapper
445
+ in distill.ts built their "previously rejected" context via
446
+ `proposalContent(p)`, which throws when a proposal's `changes[0]?.after` is
447
+ undefined — the shape `storedToChanges` deliberately returns for rows
448
+ archived before the `changes` field existed (the large majority of
449
+ real-world rejected-proposal history). The throw happened before the
450
+ signal cursor advanced, so a ref with any such legacy rejection errored on
451
+ every run instead of ever completing. Both call sites now read the preview
452
+ from `payload.content`, which is populated for every row regardless of
453
+ its `changes` shape.
454
+ - **Failed graph extractions are no longer cached as permanent hits.** A
455
+ provider outage upserted thousands of `{"entities":[],"status":"failed"}`
456
+ results into `llm_enrichment_cache` and the persisted graph, and both
457
+ cache-hit paths (the DB lookup and reuse from the previous graph) treated
458
+ them as valid hits forever after — the affected files never retried.
459
+ `status: "failed"` results are now treated as a miss and are never written
460
+ to the cache; existing rows are left on disk and are overwritten naturally
461
+ on the next successful extraction. `src/llm/graph-extract.ts` also no
462
+ longer falls back to a per-asset retry for every body in a batch after a
463
+ `provider_error` — the provider has already demonstrated it is failing, so
464
+ each asset in that batch is recorded as failed directly. Graph extraction
465
+ now aborts the rest of the run (returning the partial results already
466
+ extracted) once the failure rate crosses 50% over at least 4 attempted
467
+ extraction dispatches, mirroring consolidate's existing failure-rate guard.
468
+ The abort counts one attempt per `extractGraphFromBodies` dispatch, not per
469
+ file inside its batch — per-file counting let a single batched
470
+ `provider_error` trip the guard after one HTTP failure whenever
471
+ `graphExtractionBatchSize` was at its default of 4.
472
+ - **Memory consolidation's cooldown could never engage.** `consolidate_completed`
473
+ was only emitted when a run planned zero merge/delete/contradict operations —
474
+ advisory ops the model plans daily and that are never auto-applied — so the
475
+ event had, in practice, never fired and the pool-delta gate stayed
476
+ permanently in its bootstrap "run every time" state. The event now fires
477
+ whenever the LLM pass itself completes, recording the unapplied advisory op
478
+ count (`advisoryOpsUnapplied`) instead of withholding the event. Separately,
479
+ the memory-volume override (`memoryVolumeConsolidationThreshold`, forcing a
480
+ run when the eligible pool exceeds the threshold) is now bootstrap-only: once
481
+ a `consolidate_completed` event exists for the source, the pool-delta gate
482
+ governs on its own, even when the pool is large. `akm improve --plan`'s
483
+ `consolidation.gates.delta.reason` no longer reports "memory pool has work"
484
+ for both a real pool delta and the bootstrap case (no `consolidate_completed`
485
+ event yet, so no delta was evaluated) — bootstrap now reports its own reason.
486
+
487
+ ## [0.9.16] - 2026-09-22
488
+
489
+ ### Fixed
490
+
491
+ - **Result documents larger than 64 KiB are no longer truncated on a piped
492
+ stdout.** `akm show`/`search`/`config get … | python3 -c …` (or `| head`, or
493
+ any other pipe) returned exactly 65,536 bytes — the Linux pipe-buffer size —
494
+ producing unparseable JSON, while the same command with `--output <file>`
495
+ wrote the complete document. The two stdout writers (`deliverRendered` for
496
+ json/yaml/text/md/html, `outputJsonl` for jsonl) used `console.log`, and on
497
+ Bun `console.log` issues a single `write(2)` against a non-blocking fd 1 and
498
+ silently discards whatever the kernel did not accept; a pipe accepts at most
499
+ one buffer's worth. Both now go through `writeStdout`
500
+ (`src/output/stdout.ts`), which uses `process.stdout.write` — that handles
501
+ the short write correctly, and the queued remainder keeps the process alive
502
+ until it drains. Byte-for-byte output is unchanged on every format; only the
503
+ transport moved.
10
504
 
11
505
  ### Added
12
506
 
@@ -1,21 +1,14 @@
1
1
  You are the akm consolidate assistant analyzing memory assets.
2
2
 
3
3
  Rules:
4
- 1. MERGE: Two or more memories are substantially duplicated or closely related → propose merging. Return the primary ref to keep and secondary refs to delete. Do NOT include mergedContent — the merge will be executed in a separate step.
5
- 2. DELETE: Memory is clearly outdated, contradicted, or redundant → propose deletion. NEVER propose delete for memories annotated `(captureMode: hot)` — they are user-explicit and only the user can retire them. The downstream guard will refuse these regardless, so proposing them just wastes tokens.
6
- 3. PROMOTE: Memory expresses a stable, reusable fact suitable as a `knowledge/` asset → propose promotion. Do NOT delete the source memory. NEVER propose promote / merge / contradict for memories annotated `(already queued)` — they have a pending proposal whose body matches; a duplicate will be deterministically dropped, so proposing them just wastes tokens.
7
- 4. CONTRADICT: Two memories assert logically exclusive facts such that following BOTH simultaneously is impossible — not merely related or overlapping. You MUST cite the exact sentence from Memory A and the exact sentence from Memory B that are in direct conflict. If you cannot cite specific opposing sentences, use KEEP instead. Sharing a topic, tool, domain, or workflow stage is NOT sufficient. Only direct factual opposites qualify: opposing recommended commands, opposing boolean flags, opposing version numbers, or mutually exclusive instructions. Use confidence ≥ 0.92 only; omit the op entirely if below that threshold.
8
- 5. KEEP: Memory is unique and current → omit from output.
4
+ 1. PROMOTE: Memory expresses a stable, reusable fact suitable as a `knowledge/` asset → propose promotion. Do NOT delete the source memory. NEVER propose promote for memories annotated `(already queued)` — they have a pending proposal whose body matches; a duplicate will be deterministically dropped, so proposing them just wastes tokens.
5
+ 2. KEEP: Memory is unique and current → omit from output.
9
6
 
10
7
  Return ONLY JSON (no prose, no code fences):
11
8
  {
12
9
  "operations": [
13
- { "op": "merge", "primary": "memories/<name>", "secondaries": ["memories/<name>", ...], "mergeStrategy": "synthesize", "confidence": 0.95 },
14
- { "op": "delete", "ref": "memories/<name>", "reason": "<brief reason>", "confidence": 0.90 },
15
- { "op": "promote", "ref": "memories/<name>", "knowledgeRef": "knowledge/<suggested-slug>", "reason": "<brief reason>", "description": "<one sentence describing the new knowledge asset>", "confidence": 0.92 },
16
- { "op": "contradict", "ref": "memories/<name>", "contradictedByRef": "memories/<name>", "reason": "<brief reason>", "confidence": 0.88 }
17
- ],
18
- "warnings": ["<optional concerns>"]
10
+ { "op": "promote", "ref": "memories/<name>", "knowledgeRef": "knowledge/<suggested-slug>", "reason": "<brief reason>", "description": "<one sentence describing the new knowledge asset>", "confidence": 0.92 }
11
+ ]
19
12
  }
20
13
 
21
14
  For every operation, emit a `confidence` field in [0, 1] expressing your certainty that the operation is correct and safe. Use 0.95+ only when evidence is unambiguous. Omit the field rather than guessing if you are uncertain.
@@ -1,10 +1,10 @@
1
1
  Extract entities and relations from the asset body below.
2
2
 
3
3
  Rules:
4
- - Output ONLY a JSON object: {"entities": ["Entity One", ...], "relations": [{"from": "A", "to": "B", "type": "uses"}, ...]}.
4
+ - Output ONLY a JSON object: {"entities": ["Entity One", ...], "relations": [["A", "uses", "B"], ...]}.
5
5
  - Entities are short, canonical noun phrases (project names, services, tools, people, technical concepts). Do NOT emit file or directory paths (anything containing "/" or "\") — they are dropped downstream.
6
- - Relations connect two entities that both appear in the entities array.
7
- - "type" is a short verb phrase (e.g. "uses", "depends on", "owns", "documents"). Optional; omit when unsure.
6
+ - Each relation is a 3-element array: [from, type, to]. Relations connect two entities that both appear in the entities array.
7
+ - "type" is a short verb phrase (e.g. "uses", "depends on", "owns", "documents"). Use "" when unsure.
8
8
  - Drop pleasantries, meta-commentary, and timestamps.
9
9
  - Limit to at most {{MAX_ENTITIES}} entities and {{MAX_RELATIONS}} relations per asset.
10
10
  - Return {"entities": [], "relations": []} if the body has no extractable graph content.
@@ -19,7 +19,7 @@ for rate limiting. The terraform-provisioner deploys everything to the prod clus
19
19
  Owner: @alice.
20
20
 
21
21
  Output:
22
- {"entities":["auth-service","PostgreSQL","redis-cache","terraform-provisioner","prod cluster","@alice"],"relations":[{"from":"auth-service","to":"PostgreSQL","type":"uses"},{"from":"auth-service","to":"redis-cache","type":"depends on"},{"from":"terraform-provisioner","to":"prod cluster","type":"deploys"},{"from":"terraform-provisioner","to":"auth-service","type":"deploys"},{"from":"@alice","to":"auth-service","type":"owns"}]}
22
+ {"entities":["auth-service","PostgreSQL","redis-cache","terraform-provisioner","prod cluster","@alice"],"relations":[["auth-service","uses","PostgreSQL"],["auth-service","depends on","redis-cache"],["terraform-provisioner","deploys","prod cluster"],["terraform-provisioner","deploys","auth-service"],["@alice","owns","auth-service"]]}
23
23
 
24
24
  Input:
25
25
  ## Meeting: API Redesign
@@ -27,7 +27,7 @@ Discussed moving from REST to GraphQL. The frontend team will use Apollo Client.
27
27
  Backend needs to implement resolvers. Timeline: Q2.
28
28
 
29
29
  Output:
30
- {"entities":["REST","GraphQL","Apollo Client","frontend team","backend","resolvers","Q2"],"relations":[{"from":"frontend team","to":"Apollo Client","type":"uses"},{"from":"backend","to":"resolvers","type":"implements"},{"from":"frontend team","to":"GraphQL","type":"migrates to"}]}
30
+ {"entities":["REST","GraphQL","Apollo Client","frontend team","backend","resolvers","Q2"],"relations":[["frontend team","uses","Apollo Client"],["backend","implements","resolvers"],["frontend team","migrates to","GraphQL"]]}
31
31
 
32
32
  ===============
33
33
 
@@ -15,6 +15,7 @@
15
15
  * same read extended to the accepted/rejected archive.
16
16
  */
17
17
  import { resolveStashDir } from "../../core/common.js";
18
+ import { isStaleTargetRejection } from "../proposal/proposal-types.js";
18
19
  import { listProposals } from "../proposal/repository.js";
19
20
  /**
20
21
  * Compute accept-rate-per-source metrics from the proposal store. Defaults to
@@ -28,6 +29,11 @@ export function computeAcceptRateBySource(stashDir) {
28
29
  for (const status of statuses) {
29
30
  const proposals = listProposals(stash, { status, includeArchive });
30
31
  for (const p of proposals) {
32
+ // A stale-target auto-reject (STALE, R20) is procedural, not a
33
+ // judgement on the content — counting it would understate the
34
+ // source's real accept rate for content the drain will re-propose.
35
+ if (status === "rejected" && isStaleTargetRejection(p))
36
+ continue;
31
37
  const src = p.source || "(unknown)";
32
38
  const entry = bySource.get(src) ?? { accepted: 0, rejected: 0, pending: 0 };
33
39
  if (status === "accepted")