akm-cli 0.9.16-alpha.2 → 0.9.17-alpha.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +495 -1
- package/dist/assets/prompts/consolidate-system.md +4 -11
- package/dist/assets/prompts/graph-extract-user-prompt.md +5 -5
- package/dist/commands/health/accept-rate.js +6 -0
- package/dist/commands/health/checks.js +54 -0
- package/dist/commands/health/improve-metrics.js +1 -5
- package/dist/commands/health/report-view-model.js +0 -1
- package/dist/commands/health.js +10 -0
- package/dist/commands/improve/consolidate/chunking.js +19 -35
- package/dist/commands/improve/consolidate/merge.js +6 -9
- package/dist/commands/improve/consolidate.js +104 -91
- package/dist/commands/improve/distill/promote-memory.js +40 -2
- package/dist/commands/improve/distill/quality-gate.js +186 -23
- package/dist/commands/improve/distill.js +42 -8
- package/dist/commands/improve/eligibility.js +13 -3
- package/dist/commands/improve/improve-cli.js +32 -9
- package/dist/commands/improve/improve-strategies.js +23 -1
- package/dist/commands/improve/improve.js +121 -84
- package/dist/commands/improve/loop-stages.js +241 -108
- package/dist/commands/improve/preparation.js +50 -17
- package/dist/commands/improve/reflect.js +16 -5
- package/dist/commands/improve/shared.js +0 -10
- package/dist/commands/proposal/drain.js +79 -10
- package/dist/commands/proposal/proposal-types.js +21 -0
- package/dist/commands/proposal/repository.js +108 -29
- package/dist/core/asset/frontmatter.js +106 -1
- package/dist/core/config/schema/improve-processes.js +29 -2
- package/dist/core/improve-result.js +9 -0
- package/dist/core/paths.js +7 -0
- package/dist/indexer/ensure-index.js +52 -7
- package/dist/indexer/graph/graph-extraction.js +82 -8
- package/dist/indexer/passes/memory-inference.js +16 -1
- package/dist/llm/client.js +16 -2
- package/dist/llm/graph-extract.js +162 -18
- package/dist/output/html-render.js +2 -1
- package/dist/output/stdout.js +24 -0
- package/dist/output/text.js +4 -3
- package/dist/scripts/akm-migrate-node.js +20 -4
- package/dist/scripts/akm-migrate.js +20 -4
- package/dist/storage/repositories/index-entries-repository.js +43 -0
- package/dist/storage/repositories/proposals-repository.js +4 -1
- package/dist/storage/state-db-integrity.js +123 -0
- package/dist/workflows/program/schema.js +1 -0
- package/docs/reference/cli.md +4 -3
- package/docs/reference/data-and-telemetry.md +1 -0
- package/package.json +1 -1
- package/schemas/akm-config.json +44 -0
- package/schemas/akm-workflow.json +1 -0
- package/dist/commands/improve/eval-cases.js +0 -52
package/CHANGELOG.md
CHANGED
|
@@ -6,7 +6,501 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/).
|
|
|
6
6
|
|
|
7
7
|
## [Unreleased]
|
|
8
8
|
|
|
9
|
-
## [0.9.
|
|
9
|
+
## [0.9.17-alpha.1] - 2026-09-24
|
|
10
|
+
|
|
11
|
+
### Added
|
|
12
|
+
|
|
13
|
+
- **`akm improve --require-engines` now records its reachability probe on the
|
|
14
|
+
run result (R17).** `assertRequiredEnginesReachable` only ever reported a
|
|
15
|
+
failure (abort, exit 78); a probe that passed — including a slow or
|
|
16
|
+
flapping gateway that still answered in time — left no trace once the run
|
|
17
|
+
proceeded. It now returns one outcome per probed target (`process`,
|
|
18
|
+
`engine`, `endpoint`, `reachable`, `latencyMs`), threaded through a new
|
|
19
|
+
`AkmImproveOptions.engineProbe` and copied onto the persisted result as
|
|
20
|
+
`AkmImproveResult.engineProbe`. Omitted entirely when `--require-engines`
|
|
21
|
+
was not passed; a result persisted without it (every run before this
|
|
22
|
+
change) still decodes. `--require-engines --dry-run` results carry it too.
|
|
23
|
+
- **Reflect had no way to exclude raw wiki-ingest snapshots, which are the
|
|
24
|
+
longest generations in the ledger (89.5s/161.8s observed).** `wikis/articles/raw/*.md`
|
|
25
|
+
website snapshots index as `knowledge/wikis/articles/raw/<slug>`, and
|
|
26
|
+
reflect's `allowedTypes` filter is type-only, so it can't exclude a subset
|
|
27
|
+
of the `knowledge` type. `processes.reflect` now accepts an optional
|
|
28
|
+
`excludeRefPrefixes: string[]` — conceptId prefixes, matched after
|
|
29
|
+
stripping an optional `bundle//` from both the ref and each prefix.
|
|
30
|
+
`shouldSkipRef` skips a matching ref with reason `exclude-filter`, for
|
|
31
|
+
reflect only (distill and consolidate are memory-only and reject the key).
|
|
32
|
+
A trailing `/` on a prefix is ignored, so
|
|
33
|
+
`"knowledge/wikis/articles/raw/"` excludes the same refs as
|
|
34
|
+
`"knowledge/wikis/articles/raw"`.
|
|
35
|
+
|
|
36
|
+
- **`akm health` now checks state.db's own SQLite integrity.** A new hard
|
|
37
|
+
`state-db-integrity` check runs a read-only `PRAGMA quick_check` against
|
|
38
|
+
`state.db` and fails, naming the returned diagnostic lines and the repair
|
|
39
|
+
steps (back up, dump/restore via `sqlite3`, verify, swap in), when it
|
|
40
|
+
reports anything other than `ok`. The same check reports state.db's
|
|
41
|
+
freelist ratio (the fraction of pages `VACUUM` could reclaim) and warns
|
|
42
|
+
above 50%. Previously nothing in `akm health` looked past a successful
|
|
43
|
+
append/read round trip, which stays true on a database that is corrupt at
|
|
44
|
+
the SQLite level.
|
|
45
|
+
- **The retention purge (`akm improve`) now VACUUMs state.db when more than
|
|
46
|
+
half its pages are free**, immediately after the events/improve_runs/
|
|
47
|
+
cycle-metrics purge, recording a `state_db_vacuumed` event with pages
|
|
48
|
+
before/after. Opportunistic: a locked/busy database is skipped, not
|
|
49
|
+
raised, so it never fails the purge pass it follows.
|
|
50
|
+
|
|
51
|
+
### Changed
|
|
52
|
+
|
|
53
|
+
- **The orphan-state GC pass no longer probes index.db once per pending
|
|
54
|
+
row.** `runOrphanStateGcPass` used to call `getEntryByRef` (up to two
|
|
55
|
+
statements each, via its bare-ref fallback) for every pending
|
|
56
|
+
`asset_salience` / `asset_outcome` row — 2,101 pending rows cost 83–100s
|
|
57
|
+
per run. It now builds one snapshot of every live `item_ref` in index.db up
|
|
58
|
+
front and matches every pending row against it in memory: O(1) index.db
|
|
59
|
+
queries per run instead of one probe per row, with the same live/orphan
|
|
60
|
+
resolution (including the bundle-qualified-exact and bare-conceptId-suffix
|
|
61
|
+
fallback) as before.
|
|
62
|
+
- **Memory inference no longer forces a full reindex for the file(s) it
|
|
63
|
+
writes.** The post-inference maintenance step used to call the full
|
|
64
|
+
`reindexFn` (42–220s per run, typically for one written derived fact)
|
|
65
|
+
whenever memory inference split a parent. `runMemoryInferencePass` now
|
|
66
|
+
reports the exact paths it wrote or rewrote (`writtenPaths`, sourced from
|
|
67
|
+
the run's write-provenance journal), and the maintenance pass indexes just
|
|
68
|
+
those files with `indexWrittenAssets` instead — closing and reopening the
|
|
69
|
+
shared index.db handle around the call with the same discipline the full
|
|
70
|
+
reindex used (#584). The separate post-consolidation full reindex is
|
|
71
|
+
removed outright rather than re-gated: it used to fire whenever
|
|
72
|
+
`consolidation.processed > 0` (memories the LLM judged), but
|
|
73
|
+
merge/delete/contradict ops are advisory and never auto-applied, and the
|
|
74
|
+
one op that does execute — promote — writes a proposal to state.db, not to
|
|
75
|
+
the stash. Consolidation therefore cannot change a file the index reads,
|
|
76
|
+
so the reindex had no precondition it could ever satisfy.
|
|
77
|
+
- **The improve loop's reflect dispatch now checks the proposal
|
|
78
|
+
fingerprint/rejection-backoff guard *before* calling reflect, not just
|
|
79
|
+
after.** `fingerprint_match` and `rejection_backoff` were evaluated only
|
|
80
|
+
inside `createProposal`, which runs after reflect's full generation and
|
|
81
|
+
quality-judge call — so a ref already guaranteed to be skipped still paid
|
|
82
|
+
the LLM cost (measured: 2–16% of reflect LLM seconds spent on refs the
|
|
83
|
+
guard then discarded). The guard's fingerprint is an input fingerprint
|
|
84
|
+
(target ref, source, before-hash, model id), computable before dispatch, so
|
|
85
|
+
`checkProposalGuard` (`src/commands/proposal/repository.ts`) exposes the
|
|
86
|
+
identical check `createProposal` runs post-generation — the two share one
|
|
87
|
+
implementation and can never disagree. `runLoopReflectPass`
|
|
88
|
+
(`src/commands/improve/loop-stages.ts`) now calls it first; a hit skips
|
|
89
|
+
`reflectFn` entirely and lands in the existing `reflect-cooldown` bucket
|
|
90
|
+
with the same `reflect_invoked` event the signal-delta cursor
|
|
91
|
+
(`buildLatestProposalTsMap`) reads, so cursor advancement and run-result
|
|
92
|
+
classification are unchanged. `createProposal`'s post-generation check
|
|
93
|
+
remains the authoritative gate.
|
|
94
|
+
- **Consolidate's plan schema and prompt are promote-only.** The apply loop
|
|
95
|
+
only ever executed `promote` — `merge`/`delete`/`contradict` were advisory
|
|
96
|
+
by design and never applied — but the schema still asked for all four ops
|
|
97
|
+
plus a free-text `warnings` array, and completion tokens rose from 7–8k to
|
|
98
|
+
21–30k per run after the 35B-A3B model switch with no change in
|
|
99
|
+
promotions. `CONSOLIDATE_PLAN_JSON_SCHEMA` and `consolidate-system.md` now
|
|
100
|
+
request only `promote` (with `reason` capped at 200 chars), and `isValidOp`
|
|
101
|
+
rejects any other op shape — e.g. from a model that ignores the schema —
|
|
102
|
+
with the existing "skipping invalid operation" warning instead of treating
|
|
103
|
+
it as an actionable plan entry. `ConsolidateResult.merged` / `deleted` /
|
|
104
|
+
`contradicted` and the `planned` op breakdown are unchanged in shape and
|
|
105
|
+
stay zero.
|
|
106
|
+
- **`improve-maintenance-passes.test.ts` moved under `tests/integration/`.**
|
|
107
|
+
The suite opens a real `state.db` via `openStateDatabase`, which AGENTS.md's
|
|
108
|
+
ORG-03..06 rule places under `tests/integration/`, not `tests/`; no content
|
|
109
|
+
change. Also corrected
|
|
110
|
+
`docs/architecture/specs/improve-collapse-churn-detector-design.md` §2.5,
|
|
111
|
+
which described the post-loop collapse-detector gate as `consolidationRan
|
|
112
|
+
OR recombination.processed > 0` — no `recombination` value is plumbed into
|
|
113
|
+
`runImprovePostLoopStage` and no recombine pass exists in the codebase, so
|
|
114
|
+
the spec now matches the shipped `consolidationRan`-only gate and notes
|
|
115
|
+
that the recombine-triggered pass is not implemented.
|
|
116
|
+
- **Graph-extraction relations are now compact `[from, type, to]` triples
|
|
117
|
+
instead of `{"from","to","type"}` objects, and the batch graph-extraction
|
|
118
|
+
call now sends a `responseSchema`.** The object-keyed form cost 10+ tokens
|
|
119
|
+
per relation for no signal, and completion tokens cost far more than
|
|
120
|
+
prompt tokens; a compact triple form measured −51% / −18% completion
|
|
121
|
+
tokens on two chunks. `graph-extract.ts`'s single-asset and batch prompts
|
|
122
|
+
and JSON schemas now ask for `["from", "type", "to"]` (`type` may be `""`);
|
|
123
|
+
`parseGraphExtraction` accepts both the triple form and the legacy object
|
|
124
|
+
form (a relation-level `confidence` is still read from a legacy object,
|
|
125
|
+
though the schema no longer offers it — the prompt never asked for one).
|
|
126
|
+
Separately, production runs graph extraction batched
|
|
127
|
+
(`processes.graphExtraction.batchSize`), and `extractGraphFromBodies` sent
|
|
128
|
+
no `responseSchema` at all, so the R12b output-bounding schema only ever
|
|
129
|
+
reached the single-asset path. The batch call now sends the same
|
|
130
|
+
`maxItems`-bounded schema (scoped to the batch's asset count) through the
|
|
131
|
+
same `supportsJsonSchema`-gated `responseSchema` field the single-asset
|
|
132
|
+
call uses. `GRAPH_EXTRACT_PROMPT_VERSION` bumps `v2` → `v3`, so every file
|
|
133
|
+
re-extracts once on the next graph pass — entity semantics, caps, chunking
|
|
134
|
+
and batch sizing are unchanged.
|
|
135
|
+
|
|
136
|
+
### Removed
|
|
137
|
+
|
|
138
|
+
- **The write-only distill/proposal eval-cases path.** `writeEvalCase`
|
|
139
|
+
(`src/commands/improve/eval-cases.ts`) wrote a Markdown file per rejection
|
|
140
|
+
under `$STATE/improve/eval-cases/<stash>/` that nothing ever read back, and
|
|
141
|
+
`countEvalCases` reported a cumulative on-disk file count as if it were a
|
|
142
|
+
per-run number (surfaced as `evalCasesWritten` on the improve result and in
|
|
143
|
+
`akm health`'s improve metrics). A rejected proposal row (see above) now
|
|
144
|
+
carries the same information through a path something actually reads.
|
|
145
|
+
Deleted `eval-cases.ts` and its two `loop-stages.ts` call sites, the
|
|
146
|
+
`evalCasesWritten` field from `AkmImproveResult` and every health-metrics
|
|
147
|
+
reader/aggregator, and the `improve_completed` event's `evalCasesWritten`
|
|
148
|
+
field. `decodeImproveResult` still accepts (and ignores) `evalCasesWritten`
|
|
149
|
+
on an envelope an older release wrote, and existing eval-case files on disk
|
|
150
|
+
are untouched — `getEvalCasesDir` (`core/paths.ts`) stays, since
|
|
151
|
+
`scripts/akm-migrate/migrate/writer-relocation.ts` still uses it to
|
|
152
|
+
relocate them from the legacy `$STASH/.akm/eval-cases/` path.
|
|
153
|
+
|
|
154
|
+
### Fixed
|
|
155
|
+
|
|
156
|
+
- **The lesson quality judge's ACTIONABILITY criterion carried no signal, and
|
|
157
|
+
the judge's request/parser let a differently-spelled or extra key change
|
|
158
|
+
the verdict (R16).** Splinter measured ACTIONABILITY at AUC 0.46 against
|
|
159
|
+
accept/reject outcomes — no better than chance — and averaging it into the
|
|
160
|
+
score pulled every verdict toward its 3.0 mode, i.e. the review band.
|
|
161
|
+
`buildJudgePrompt` no longer asks for it;
|
|
162
|
+
`LESSON_JUDGE_CRITERIA_KEYS` is now `novelty`/`nonRedundancy` only.
|
|
163
|
+
Separately, `runQualityJudge`'s request sent no `responseSchema` while the
|
|
164
|
+
prompt text spelled criteria as NON-REDUNDANCY / FEEDBACK ALIGNMENT, so a
|
|
165
|
+
model that echoed a differently-cased or -spelled key turned the verdict
|
|
166
|
+
into a parse failure routed to review; and `parseJudgeResponse` averaged
|
|
167
|
+
over every key present in `scores`, so an unexpected extra key changed the
|
|
168
|
+
score. `runQualityJudge` now sends a strict `responseSchema` — built from
|
|
169
|
+
the judge's own expected criteria keys, `additionalProperties: false` at
|
|
170
|
+
both levels — through the same `supportsJsonSchema`-gated
|
|
171
|
+
`request.responseSchema` path `src/llm/graph-extract.ts` uses, a no-op for
|
|
172
|
+
providers that don't opt in; and `parseJudgeResponse` now reads, validates,
|
|
173
|
+
and averages only the expected keys, silently ignoring any other key
|
|
174
|
+
instead of averaging or validating it. A missing expected key is still a
|
|
175
|
+
parse failure, unchanged.
|
|
176
|
+
- **The reflect quality-gate's "no judge configured" warning named a config
|
|
177
|
+
key nothing reads.** It told users to set
|
|
178
|
+
`improve.strategies.<name>.processes.reflect.qualityGate.engine`, but
|
|
179
|
+
`qualityGate` is `{ enabled }` passthrough — `resolveReflectQualityJudgeRunner`
|
|
180
|
+
always uses the generation runner when it is an LLM, or falls back to
|
|
181
|
+
`defaults.llmEngine` via `resolveImproveLlmExecution` with no profile/process
|
|
182
|
+
layer, so that key was never read. The warning now names only
|
|
183
|
+
`defaults.llmEngine`.
|
|
184
|
+
- **The distill/reflect LLM-as-judge quality gate inherited the generation
|
|
185
|
+
runner's temperature, and its averaged score hid which criterion actually
|
|
186
|
+
failed.** `runQualityJudge`'s request only pinned `enableThinking: false`,
|
|
187
|
+
so the judge ran at whatever temperature generation used — measured at 0.3,
|
|
188
|
+
the verdict flipped on 10/16 identical inputs, vs. 0/16 at temperature 0.
|
|
189
|
+
The request now also pins `temperature: 0`, for both the distill and
|
|
190
|
+
reflect judges that share this function, independent of the runner's
|
|
191
|
+
configured temperature. Separately, both judge prompts asked for one
|
|
192
|
+
averaged float, so a criterion carrying no signal was invisible in
|
|
193
|
+
production. They now ask for per-criterion integer scores
|
|
194
|
+
(`buildJudgePrompt`: novelty/actionability/nonRedundancy;
|
|
195
|
+
`buildReflectJudgePrompt`: feedbackAlignment/preservation/quality), averaged
|
|
196
|
+
in code to the same `score` the unchanged 3.5/2.5 thresholds gate on. The
|
|
197
|
+
parser accepts this new `{"scores": {...}, "reason"}` shape and still
|
|
198
|
+
accepts the old `{"score": <float>, "reason"}` shape a model may return;
|
|
199
|
+
each criterion (or the bare score) must be a finite number in 1..5 or the
|
|
200
|
+
response routes to review exactly as a parse failure does today. The
|
|
201
|
+
per-criterion scores, when present, are now carried through
|
|
202
|
+
`QualityJudgeResult.criteria` into the `distill_invoked` event metadata and
|
|
203
|
+
rejection-envelope frontmatter `writeQualityRejection` writes, and into
|
|
204
|
+
reflect's `reflect_completed` rejection event as `qualityCriteria`.
|
|
205
|
+
- **The judge parser accepted a partial `scores` object and auto-passed it.**
|
|
206
|
+
`parseJudgeResponse` validated only that whatever criterion keys arrived
|
|
207
|
+
held finite 1-5 values, then averaged over those keys alone — so a
|
|
208
|
+
truncated judge response like `{"scores": {"novelty": 5}, "reason": "…"}`
|
|
209
|
+
parsed to `score: 5.0` and `pass: true`, promoting content the judge never
|
|
210
|
+
finished evaluating on its other criteria. `runQualityJudge` now passes the
|
|
211
|
+
criterion key set its prompt asked for (`buildJudgePrompt`:
|
|
212
|
+
novelty/actionability/nonRedundancy; `buildReflectJudgePrompt`:
|
|
213
|
+
feedbackAlignment/preservation/quality) down to `parseJudgeResponse`, which
|
|
214
|
+
returns a parse failure — routed to review, exactly as a malformed response
|
|
215
|
+
is today — when any expected key is missing from `scores`.
|
|
216
|
+
- **The reflect pre-generation proposal-guard skip (R9) emitted `reflect_invoked` with no paired `reflect_completed`.** `runLoopReflectPass`'s guard-skip branch in `loop-stages.ts` appended a synthetic `reflect_invoked` event to advance the signal-delta cursor, but never called `reflectFn`, so `reflect.ts`'s own `reflect_completed` emission never ran either — a new, permanent source of unpaired `reflect_invoked` rows for every fingerprint/backoff hit, violating the invoke/complete pairing invariant `buildReflectEventEmitters` documents. The branch now also appends a matching `reflect_completed` (`ok:false`, `reason:"cooldown"`, `subreason:"pre_generation_guard"`), mirroring `emitFailed`'s shape.
|
|
217
|
+
- **R9's pre-generation proposal guard covered reflect only — distill paid for a full generation + judge call before the same fingerprint/backoff guard could reject it.** `runLoopDistillPass` had no equivalent of `runLoopReflectPass`'s pre-check, even though `createProposal`'s post-generation guard (and every rejected row R10 now mints under `source: "distill"`) applies to distill just as much. `runLoopDistillPass` now calls `checkProposalGuard` against the derived lesson/knowledge ref (distill's real `createProposal` call never targets the input ref) before dispatching `distillFn`; a hit routes to the pass's existing `distill-skipped` bucket and emits `distill_invoked` with a `skipped` outcome so `buildLatestProposalTsMap`'s signal cursor still advances.
|
|
218
|
+
- **The distill pre-generation proposal guard could suppress a legitimate dispatch by checking a ref distill would never target.** For a memory input, distill's real `createProposal` call targets one of two refs decided at dispatch time inside `planMemoryKnowledgePromotion` — the derived knowledge ref when the deterministic promotion heuristic fires, the derived lesson ref otherwise — but `runLoopDistillPass`'s pre-check checked both candidate refs and skipped on the FIRST guard hit, so a stale fingerprint/backoff hit on the ref distill would NOT have targeted silently suppressed dispatch until that ref's fingerprint happened to change. The pre-check now resolves the SAME target `planMemoryKnowledgePromotion` would via `wouldPromoteMemoryToKnowledge` (`distill/promote-memory.ts`) — a thin wrapper that delegates to `planMemoryKnowledgePromotion` itself so the classification can never drift from the real dispatch decision, with no LLM call — and checks only that ref; content and the classification's `durableInputRef` are read via `planned.ref` alone, matching `akmDistill`'s real dispatch, while `planned.itemRef ?? planned.ref` feeds only the feedback-events query.
|
|
219
|
+
- **The `akm improve` triage pre-pass drain's judgment LLM calls were unattributed in the usage report.** `runTriagePrePass`'s `drainProposalsFn` call dispatched judgment calls with no `withLlmStage` wrapper, unlike the standalone `akm proposal drain` CLI path, so they landed in `byProcessEngineModel` as unattributed (5 calls, 24s per run) instead of under a `triage` stage. The pre-pass drain is now wrapped in `withLlmStage("triage", …, { engine, process: "triage.judgment" })`, mirroring the CLI path.
|
|
220
|
+
- **The batch graph-extraction provider-storm guard only recognized one error
|
|
221
|
+
code.** After a failed batch call, `extractGraphFromBodies` skipped the
|
|
222
|
+
per-asset fallback retry only for `LlmCallError`s coded `provider_error` —
|
|
223
|
+
but a dead endpoint more often raises `network_error` (a dropped
|
|
224
|
+
connection) or `provider_html_error` (a provider serving an HTML error
|
|
225
|
+
page), both of which still paid the full per-asset fallback storm the
|
|
226
|
+
guard exists to prevent. The predicate is now `isTransportFailure`
|
|
227
|
+
(`src/llm/client.ts`), shared with `chatCompletion`'s retry classifier so
|
|
228
|
+
the two cannot drift apart, and covers `provider_error`, `network_error`,
|
|
229
|
+
and `provider_html_error`.
|
|
230
|
+
- **`akm health`'s `state-db-integrity` check no longer crashes when the freelist/page-count read fails.** `getStateDbFreelistInfo` had a `finally` but no `catch` around its read-only open and pragma reads, unlike its sibling `runStateDbQuickCheck` — a throw there (e.g. an unopenable state.db) escaped `akm health` as an unclassified exit 70 on exactly the damaged database the check exists to report. It now returns a zeroed `StateDbFreelistInfo` with an `error` field, and the check renders that as a failed check instead of throwing.
|
|
231
|
+
- **The post-purge VACUUM's `state_db_vacuumed` event now honors the caller's `EventsContext`.** `vacuumStateDbIfReclaimable` appended its event with a direct `insertEvent` call, bypassing `EventsContext.readOnly` and the injectable clock its sibling purge events (`events_purged`, `improve_runs_purged`, `improve_cycle_metrics_purged`) use in the same `runRetentionPurgePass` callback. It now appends the event via `appendEvent` with the caller's `EventsContext` plumbed through.
|
|
232
|
+
- **Consolidate's per-chunk prompt excerpt truncated the raw file (frontmatter
|
|
233
|
+
+ body) instead of the body.** `buildChunkPrompt` sliced `body.slice(0,
|
|
234
|
+
bodyTruncation)` off the unstripped file; a memory whose frontmatter alone
|
|
235
|
+
exceeded the excerpt length was judged on metadata only and never showed
|
|
236
|
+
its own body text. The excerpt now truncates `stripFrontmatterBody(body)`;
|
|
237
|
+
hot/queued detection is unchanged and still reads the raw body.
|
|
238
|
+
- **Consolidate's chunk prompt carried an unused ~14k-char standards block and
|
|
239
|
+
a header the model sometimes echoed back as a bogus `ref`.** Every chunk
|
|
240
|
+
prompt resolved and injected a "Standards to follow" section
|
|
241
|
+
(`resolveStandardsContext("memories/_consolidated", ...)`), but the chunk
|
|
242
|
+
output is a promote-only op list that never reads it. Separately, the
|
|
243
|
+
chunk header (`Chunk N of M, memories <first>–<last>:`) named the chunk's
|
|
244
|
+
boundary memories with an en dash between two `memories/<name>` refs; on
|
|
245
|
+
2026-09-24 the judge model returned promote ops whose `ref` was exactly
|
|
246
|
+
that `memories/<first>–memories/<last>` range, naming a memory that does
|
|
247
|
+
not exist and losing the promotion. `buildChunkPrompt` no longer takes a
|
|
248
|
+
`standardsContext` and the header is now
|
|
249
|
+
`Chunk N of M (<count> memories):` — no refs in it.
|
|
250
|
+
- **Consolidate re-judged memories that were already promoted verbatim into
|
|
251
|
+
`knowledge/`.** That duplication was previously discovered only after the
|
|
252
|
+
LLM (`shouldSkipPromotionBodyDuplicate`), so a pool where the large
|
|
253
|
+
majority of memories were already-promoted duplicates still paid the full
|
|
254
|
+
chunk/LLM cost on all of them before being skipped.
|
|
255
|
+
`inspectConsolidationPool` now drops those memories before any chunking or
|
|
256
|
+
LLM work, sharing one `loadExistingKnowledgeBodyHashes` call and the same
|
|
257
|
+
`cacheHash` domain with the post-LLM check so the two cannot disagree. The
|
|
258
|
+
dropped count is reported as `prefilteredAlreadyPromoted` on the
|
|
259
|
+
consolidate result and in a warning line. The pre-filter also now runs
|
|
260
|
+
*before* the `consolidate.limit` cap (previously after), so a run with a
|
|
261
|
+
limit set selects its oldest-modified window from the pre-filtered pool
|
|
262
|
+
instead of re-selecting and re-dropping the same permanently-undeletable
|
|
263
|
+
duplicates every run while fresh memories past the cap went unreached; the
|
|
264
|
+
preview/eligibility path (`preparation.ts`) computes and passes the same
|
|
265
|
+
hash set so the reported candidate pool agrees with what the run will act
|
|
266
|
+
on. A live (non-preview) `akm improve` run reuses that same hash set for
|
|
267
|
+
the actual `akmConsolidate` call instead of recomputing it, so a run still
|
|
268
|
+
walks `knowledge/` only once.
|
|
269
|
+
- **`improve`'s start-of-run index rescan ran after triage dirtied the stash,
|
|
270
|
+
not before it.** Proposal triage promotes accepted proposals straight into
|
|
271
|
+
the flat `knowledge/` root, and the blocking `ensureIndex` call that is
|
|
272
|
+
supposed to give the run a current index ran only afterward (inside
|
|
273
|
+
`collectEligibleRefs`'s setup), so every triage promotion guaranteed the
|
|
274
|
+
very full rescan it should have preceded — up to ~27 minutes, holding the
|
|
275
|
+
index lock against co-scheduled writers. `ensureIndex` now runs before the
|
|
276
|
+
triage pre-pass, and triage's own writes are indexed incrementally
|
|
277
|
+
(`indexWrittenAssets`) so `collectEligibleRefs` still sees them without a
|
|
278
|
+
second full walk. Because `indexWrittenAssets` upserts a file's
|
|
279
|
+
`content_hash` without bumping `builtAt`, index staleness detection
|
|
280
|
+
(`ensure-index.ts`) is now per-file: a file newer than the last build is
|
|
281
|
+
only treated as stale when its current content actually differs from what
|
|
282
|
+
is indexed, so incrementally-reindexed content stops re-triggering the
|
|
283
|
+
same full rescan on every subsequent run. The implicit reindex's timing
|
|
284
|
+
breakdown (walk/llm/embed/finalize), previously discarded, is now logged
|
|
285
|
+
at verbose level and surfaced on the improve result as `ensureIndexDurationMs`.
|
|
286
|
+
- **Distill quality rejections vanished instead of persisting, so backoff and
|
|
287
|
+
Reflexion never saw them and the same ref was re-selected and re-rejected
|
|
288
|
+
on every run** (two refs were rejected 11× and 10×). `writeQualityRejection`
|
|
289
|
+
wrote only a `$STATE`-side file and an event, never a `proposals` row, so
|
|
290
|
+
`rejection_backoff`/`fingerprint_match` (proposal/repository.ts) and the
|
|
291
|
+
Reflexion "previously rejected" context had nothing to find; the distill
|
|
292
|
+
signal-delta cursor (`buildLatestProposalTsMap`) also only advanced for
|
|
293
|
+
`queued`/`skipped`/`validation_failed` outcomes, so a rejected ref stayed
|
|
294
|
+
eligible forever. `writeQualityRejection` now mints a real proposal through
|
|
295
|
+
the same `createProposal`/`archiveProposal` path every other proposal
|
|
296
|
+
source uses: a `quality_rejected` outcome is minted pending then archived
|
|
297
|
+
to `rejected` carrying the judge's reason; a `review_needed` outcome stays
|
|
298
|
+
`pending` in the normal queue, where triage — a human, or the drain's
|
|
299
|
+
judgment tier when one is configured — decides, the same path every other
|
|
300
|
+
pending distill proposal (including quality-gate passes) already takes.
|
|
301
|
+
The cursor now also advances on both outcomes (still excluding
|
|
302
|
+
`llm_failed`, where no real attempt produced anything). A retry for the
|
|
303
|
+
same target, source, and model is skipped by `fingerprint_match` (the
|
|
304
|
+
input fingerprint recorded at mint, retained `archiveRetentionDays`,
|
|
305
|
+
default 90 days); the 30-day `rejection_backoff` window only applies once
|
|
306
|
+
the target's before-hash or the model differs. Because these machine
|
|
307
|
+
rejections are now real `rejected` rows under `source: "distill"`, `akm
|
|
308
|
+
health`'s distill accept rate (`computeAcceptRateBySource`,
|
|
309
|
+
src/commands/health/accept-rate.ts) drops relative to earlier releases and
|
|
310
|
+
no longer measures reviewer acceptance alone. Nothing gates on that
|
|
311
|
+
metric.
|
|
312
|
+
- **`writeQualityRejection` could throw instead of returning a rejection
|
|
313
|
+
result.** Minting the proposal row above runs the mint-time canonical
|
|
314
|
+
validator (`createProposal` → `rejectProposal`,
|
|
315
|
+
`src/commands/proposal/repository.ts`), which throws `UsageError` for
|
|
316
|
+
structurally-invalid content — e.g. a `lessons/` ref whose body lacks
|
|
317
|
+
`description`/`when_to_use`. `writeQualityRejection` is the terminal,
|
|
318
|
+
non-throwing rejection path and none of its callers handled a throw. The
|
|
319
|
+
proposal row is bookkeeping for backoff/Reflexion, never the authoritative
|
|
320
|
+
record of the rejection, so a validator throw now degrades to "no row
|
|
321
|
+
minted" — the envelope file and `distill_invoked` event are still written,
|
|
322
|
+
matching the existing fingerprint/backoff skip behavior.
|
|
323
|
+
- **A `review_needed` quality-gate rejection could be auto-promoted by the
|
|
324
|
+
triage drain's judgment tier with no human ever seeing it.**
|
|
325
|
+
`writeQualityRejection` minted a `review_needed` outcome as an ordinary
|
|
326
|
+
pending proposal under `source: "distill"` (knowledge promotions from
|
|
327
|
+
`promote-memory.ts` take the same path); the `personal-stash` drain policy
|
|
328
|
+
defers `distill` proposals to the judgment tier, which can auto-accept
|
|
329
|
+
under `applyMode: promote` + `experimental.improveAutonomy` — so content
|
|
330
|
+
the quality judge explicitly refused to auto-queue (the 2.5–3.5
|
|
331
|
+
review-needed band) could be promoted without a human in the loop.
|
|
332
|
+
`writeQualityRejection` now stamps a `review_needed` mint with a
|
|
333
|
+
`{ outcome: "deferred", reason: "quality-review", gate: "quality-gate" }`
|
|
334
|
+
gate decision (best-effort: a stamp failure warns and continues, like the
|
|
335
|
+
existing mint/archive tolerance), and `classifyPendingProposals`
|
|
336
|
+
(`proposal/drain.ts`) skips any pending row carrying it — leaving it
|
|
337
|
+
pending and untouched, before the drain's own policy-deferred re-stamp
|
|
338
|
+
loop would otherwise overwrite the stamp.
|
|
339
|
+
- **Consolidate's post-LLM promote-dedup hash double-stripped frontmatter.**
|
|
340
|
+
`shouldSkipPromotionBodyDuplicate`'s `bodyHash` was computed as
|
|
341
|
+
`cacheHash(parseFrontmatter(memoryContent).content.trim())` — the body was
|
|
342
|
+
already frontmatter-stripped before being handed to `cacheHash`, which
|
|
343
|
+
strips it again internally — diverging from the single-strip
|
|
344
|
+
`cacheHash(raw)` domain `loadExistingKnowledgeBodyHashes` and the pre-filter
|
|
345
|
+
use for a source memory body that begins with its own `---` block. The
|
|
346
|
+
check now hashes `cacheHash(memoryContent)` directly, so the two sides of
|
|
347
|
+
the dedup comparison agree.
|
|
348
|
+
- **Consolidate's per-chunk prompt still warned against proposing `delete`
|
|
349
|
+
for `(captureMode: hot)` memories.** The consolidate op schema and system
|
|
350
|
+
prompt dropped `delete` (along with `merge`/`contradict`), leaving
|
|
351
|
+
`buildChunkPrompt`'s top-of-prompt hot-ref block as the only remaining
|
|
352
|
+
mention of `delete` anywhere in the prompt — a retired op name that
|
|
353
|
+
`isValidOp` now rejects if the model echoes it back, wasting tokens on
|
|
354
|
+
"skipping invalid operation" warnings. The block and the `hotRefs`
|
|
355
|
+
collection that fed it are removed; the inline `(captureMode: hot)`
|
|
356
|
+
annotation on each memory line is unchanged.
|
|
357
|
+
- **Graph extraction sent no `json_schema` and no per-asset chunk cap, so a
|
|
358
|
+
long file could pay for dozens of LLM calls whose output was then sliced
|
|
359
|
+
down to the same 32-entity/32-relation limit anyway** (one file spent 21 of
|
|
360
|
+
27 calls and 12.9k completion tokens this way). The single-asset extraction
|
|
361
|
+
call (`extractGraphFromBody`) now sends a `responseSchema` (entities/
|
|
362
|
+
relations capped at 32 each, `additionalProperties: false` otherwise), via
|
|
363
|
+
the same `supportsJsonSchema`-gated request path memory-infer.ts uses — no
|
|
364
|
+
`maxTokens` is sent; cost is bounded by the schema's `maxItems` caps alone,
|
|
365
|
+
per AGENTS.md's "LLM Defaults" (a hardcoded cap risked silent truncation
|
|
366
|
+
with zero headroom for JSON punctuation or reasoning tokens). A body
|
|
367
|
+
chunked beyond the new
|
|
368
|
+
`processes.graphExtraction.maxChunksPerAsset` (default 8) now stops after
|
|
369
|
+
the first N chunks instead of processing every one; the skipped chunks are
|
|
370
|
+
reported as `truncatedChunks` in the run's graph-extraction telemetry so
|
|
371
|
+
the coverage loss is visible rather than silently absorbed. The `improve`
|
|
372
|
+
loop's dispatch (`loop-stages.ts`) now also forwards a configured
|
|
373
|
+
`maxChunksPerAsset` to the extraction call, mirroring the existing
|
|
374
|
+
`topN`/`batchSize` wiring — without this the config key had no effect in a
|
|
375
|
+
real `akm improve` run and the default of 8 always applied.
|
|
376
|
+
- **The graph-extraction `responseSchema` forbade the `confidence` field the
|
|
377
|
+
parser itself reads.** `additionalProperties: false` on both the root
|
|
378
|
+
object and each relation item made `confidence` impossible on a
|
|
379
|
+
`supportsJsonSchema` provider, even though `parseGraphExtraction` uses
|
|
380
|
+
`rel.confidence` to drop relations below `MIN_RELATION_CONFIDENCE` and
|
|
381
|
+
`item.confidence` to feed the merged extraction confidence — silently
|
|
382
|
+
turning the confidence filter into dead code on exactly the providers the
|
|
383
|
+
schema targets. `confidence: {"type": "number"}` is now allowed at both
|
|
384
|
+
levels; `additionalProperties: false` still forbids anything else.
|
|
385
|
+
- **A pending proposal went stale the moment akm's own bookkeeping touched
|
|
386
|
+
its target, and promote refused it forever (R20).** `resolveProposalTargetInfo`
|
|
387
|
+
captured the target's raw `beforeHash` at mint; the SAME nightly run's
|
|
388
|
+
`writeSalienceToFrontmatter` (distill) and memory inference's
|
|
389
|
+
`inferenceProcessed` stamp then rewrote the target's frontmatter before
|
|
390
|
+
promote ran, so `promoteProposalWithLease`'s guard (`repository.ts` ~L2406)
|
|
391
|
+
and `drain.ts`'s dry-run mirror (`assertProposalTargetFresh`) refused every
|
|
392
|
+
affected proposal with "target changed after proposal was created" — the
|
|
393
|
+
same 11+ reflect proposals, every day, on splinter. `resolveProposalTargetInfo`
|
|
394
|
+
now also captures `beforeHashNormalized` (`core/asset/frontmatter.ts`'s new
|
|
395
|
+
`computeNormalizedContentHash`, over the target with
|
|
396
|
+
`BOOKKEEPING_FRONTMATTER_KEYS` — `salience`/`salienceInputs`/`inferenceProcessed`
|
|
397
|
+
— stripped and the remaining frontmatter canonically re-serialized); the
|
|
398
|
+
promote guard and its dry-run mirror both prefer it over the raw
|
|
399
|
+
`beforeHash` when present, so a bookkeeping-only rewrite no longer stales a
|
|
400
|
+
proposal out while a real content change still refuses. Promotion also now
|
|
401
|
+
carries the live target's bookkeeping keys forward
|
|
402
|
+
(`carryForwardBookkeepingFrontmatter`) when the proposal's own frontmatter
|
|
403
|
+
doesn't set them, so accepting never drops `inferenceProcessed` and forces
|
|
404
|
+
memory inference to reprocess the memory. A legacy proposal minted before
|
|
405
|
+
this field existed keeps its exact original raw-hash check.
|
|
406
|
+
`computeNormalizedContentHash` also normalizes the body boundary the same
|
|
407
|
+
way `assembleAssetFromString` does (leading newlines stripped, exactly one
|
|
408
|
+
trailing newline) before hashing, and treats an empty frontmatter block as
|
|
409
|
+
`{}` instead of falling back to the raw hash — both `writeSalienceToFrontmatter`
|
|
410
|
+
and the memory-inference `assembleAsset` rewrite shift where the body starts,
|
|
411
|
+
which without this normalization still staled a proposal out unless the
|
|
412
|
+
target's frontmatter was already in that exact on-disk shape.
|
|
413
|
+
- **A stale-target promote failure was retried, and refused, identically
|
|
414
|
+
every drain run forever (R20).** The drain already categorized a
|
|
415
|
+
"target changed/was created after proposal" failure as `stale-target`
|
|
416
|
+
(`categorizeDrainFailure`), but left the row pending either way — so the
|
|
417
|
+
same proposals failed the same way on every subsequent `akm proposal
|
|
418
|
+
drain` / triage pass. Both promote-failure sites (`drainProposals`'s
|
|
419
|
+
deterministic loop and `runJudgmentTier`) now auto-reject a stale-target
|
|
420
|
+
failure once, stamping `gateDecision: { outcome: "auto-rejected", reason:
|
|
421
|
+
"stale-target" }` instead of leaving it to retry. This is not a merit
|
|
422
|
+
rejection, so `checkFingerprintAndBackoff`'s rejection-backoff window
|
|
423
|
+
(`repository.ts`) now excludes stale-target rows — the ref stays
|
|
424
|
+
re-proposable against its current content — and the Reflexion
|
|
425
|
+
"previously rejected" context (`reflect.ts`'s `readRejectedProposals`,
|
|
426
|
+
`distill.ts`'s `buildDistillMessages`) and the accept-rate health metric
|
|
427
|
+
(`health/accept-rate.ts`) now exclude stale-target rejections too, so a
|
|
428
|
+
procedural refusal doesn't misrepresent content quality. `--dry-run` now
|
|
429
|
+
predicts the same outcome: a stale-target promote failure it detects is
|
|
430
|
+
reported under `rejected`, matching what a real run does, instead of under
|
|
431
|
+
`failed`.
|
|
432
|
+
|
|
433
|
+
- **Reflect quality-gate rejections were mislabelled `parse_error` and fed
|
|
434
|
+
back into later prompts as learned "avoid" patterns.** When the reflect
|
|
435
|
+
quality judge rejected an otherwise well-parsed proposal, the result
|
|
436
|
+
carried `reason: "parse_error"` — a real parse failure and a judge
|
|
437
|
+
rejection were indistinguishable. The improve loop injects non-excluded
|
|
438
|
+
reflect failures into the next reflect prompt's "Avoid These Patterns"
|
|
439
|
+
block, so a single gate rejection could poison every subsequent candidate
|
|
440
|
+
in the run. Judge rejections now carry a distinct `quality_rejected`
|
|
441
|
+
reason, stay in the `reflect-failed` metrics bucket, and are excluded from
|
|
442
|
+
that avoid-patterns injection like the existing deterministic skips.
|
|
443
|
+
- **Legacy rejected proposals no longer throw before reflect/distill prompt
|
|
444
|
+
dispatch.** `readRejectedProposals` (reflect.ts) and the equivalent mapper
|
|
445
|
+
in distill.ts built their "previously rejected" context via
|
|
446
|
+
`proposalContent(p)`, which throws when a proposal's `changes[0]?.after` is
|
|
447
|
+
undefined — the shape `storedToChanges` deliberately returns for rows
|
|
448
|
+
archived before the `changes` field existed (the large majority of
|
|
449
|
+
real-world rejected-proposal history). The throw happened before the
|
|
450
|
+
signal cursor advanced, so a ref with any such legacy rejection errored on
|
|
451
|
+
every run instead of ever completing. Both call sites now read the preview
|
|
452
|
+
from `payload.content`, which is populated for every row regardless of
|
|
453
|
+
its `changes` shape.
|
|
454
|
+
- **Failed graph extractions are no longer cached as permanent hits.** A
|
|
455
|
+
provider outage upserted thousands of `{"entities":[],"status":"failed"}`
|
|
456
|
+
results into `llm_enrichment_cache` and the persisted graph, and both
|
|
457
|
+
cache-hit paths (the DB lookup and reuse from the previous graph) treated
|
|
458
|
+
them as valid hits forever after — the affected files never retried.
|
|
459
|
+
`status: "failed"` results are now treated as a miss and are never written
|
|
460
|
+
to the cache; existing rows are left on disk and are overwritten naturally
|
|
461
|
+
on the next successful extraction. `src/llm/graph-extract.ts` also no
|
|
462
|
+
longer falls back to a per-asset retry for every body in a batch after a
|
|
463
|
+
`provider_error` — the provider has already demonstrated it is failing, so
|
|
464
|
+
each asset in that batch is recorded as failed directly. Graph extraction
|
|
465
|
+
now aborts the rest of the run (returning the partial results already
|
|
466
|
+
extracted) once the failure rate crosses 50% over at least 4 attempted
|
|
467
|
+
extraction dispatches, mirroring consolidate's existing failure-rate guard.
|
|
468
|
+
The abort counts one attempt per `extractGraphFromBodies` dispatch, not per
|
|
469
|
+
file inside its batch — per-file counting let a single batched
|
|
470
|
+
`provider_error` trip the guard after one HTTP failure whenever
|
|
471
|
+
`graphExtractionBatchSize` was at its default of 4.
|
|
472
|
+
- **Memory consolidation's cooldown could never engage.** `consolidate_completed`
|
|
473
|
+
was only emitted when a run planned zero merge/delete/contradict operations —
|
|
474
|
+
advisory ops the model plans daily and that are never auto-applied — so the
|
|
475
|
+
event had, in practice, never fired and the pool-delta gate stayed
|
|
476
|
+
permanently in its bootstrap "run every time" state. The event now fires
|
|
477
|
+
whenever the LLM pass itself completes, recording the unapplied advisory op
|
|
478
|
+
count (`advisoryOpsUnapplied`) instead of withholding the event. Separately,
|
|
479
|
+
the memory-volume override (`memoryVolumeConsolidationThreshold`, forcing a
|
|
480
|
+
run when the eligible pool exceeds the threshold) is now bootstrap-only: once
|
|
481
|
+
a `consolidate_completed` event exists for the source, the pool-delta gate
|
|
482
|
+
governs on its own, even when the pool is large. `akm improve --plan`'s
|
|
483
|
+
`consolidation.gates.delta.reason` no longer reports "memory pool has work"
|
|
484
|
+
for both a real pool delta and the bootstrap case (no `consolidate_completed`
|
|
485
|
+
event yet, so no delta was evaluated) — bootstrap now reports its own reason.
|
|
486
|
+
|
|
487
|
+
## [0.9.16] - 2026-09-22
|
|
488
|
+
|
|
489
|
+
### Fixed
|
|
490
|
+
|
|
491
|
+
- **Result documents larger than 64 KiB are no longer truncated on a piped
|
|
492
|
+
stdout.** `akm show`/`search`/`config get … | python3 -c …` (or `| head`, or
|
|
493
|
+
any other pipe) returned exactly 65,536 bytes — the Linux pipe-buffer size —
|
|
494
|
+
producing unparseable JSON, while the same command with `--output <file>`
|
|
495
|
+
wrote the complete document. The two stdout writers (`deliverRendered` for
|
|
496
|
+
json/yaml/text/md/html, `outputJsonl` for jsonl) used `console.log`, and on
|
|
497
|
+
Bun `console.log` issues a single `write(2)` against a non-blocking fd 1 and
|
|
498
|
+
silently discards whatever the kernel did not accept; a pipe accepts at most
|
|
499
|
+
one buffer's worth. Both now go through `writeStdout`
|
|
500
|
+
(`src/output/stdout.ts`), which uses `process.stdout.write` — that handles
|
|
501
|
+
the short write correctly, and the queued remainder keeps the process alive
|
|
502
|
+
until it drains. Byte-for-byte output is unchanged on every format; only the
|
|
503
|
+
transport moved.
|
|
10
504
|
|
|
11
505
|
### Added
|
|
12
506
|
|
|
@@ -1,21 +1,14 @@
|
|
|
1
1
|
You are the akm consolidate assistant analyzing memory assets.
|
|
2
2
|
|
|
3
3
|
Rules:
|
|
4
|
-
1.
|
|
5
|
-
2.
|
|
6
|
-
3. PROMOTE: Memory expresses a stable, reusable fact suitable as a `knowledge/` asset → propose promotion. Do NOT delete the source memory. NEVER propose promote / merge / contradict for memories annotated `(already queued)` — they have a pending proposal whose body matches; a duplicate will be deterministically dropped, so proposing them just wastes tokens.
|
|
7
|
-
4. CONTRADICT: Two memories assert logically exclusive facts such that following BOTH simultaneously is impossible — not merely related or overlapping. You MUST cite the exact sentence from Memory A and the exact sentence from Memory B that are in direct conflict. If you cannot cite specific opposing sentences, use KEEP instead. Sharing a topic, tool, domain, or workflow stage is NOT sufficient. Only direct factual opposites qualify: opposing recommended commands, opposing boolean flags, opposing version numbers, or mutually exclusive instructions. Use confidence ≥ 0.92 only; omit the op entirely if below that threshold.
|
|
8
|
-
5. KEEP: Memory is unique and current → omit from output.
|
|
4
|
+
1. PROMOTE: Memory expresses a stable, reusable fact suitable as a `knowledge/` asset → propose promotion. Do NOT delete the source memory. NEVER propose promote for memories annotated `(already queued)` — they have a pending proposal whose body matches; a duplicate will be deterministically dropped, so proposing them just wastes tokens.
|
|
5
|
+
2. KEEP: Memory is unique and current → omit from output.
|
|
9
6
|
|
|
10
7
|
Return ONLY JSON (no prose, no code fences):
|
|
11
8
|
{
|
|
12
9
|
"operations": [
|
|
13
|
-
{ "op": "
|
|
14
|
-
|
|
15
|
-
{ "op": "promote", "ref": "memories/<name>", "knowledgeRef": "knowledge/<suggested-slug>", "reason": "<brief reason>", "description": "<one sentence describing the new knowledge asset>", "confidence": 0.92 },
|
|
16
|
-
{ "op": "contradict", "ref": "memories/<name>", "contradictedByRef": "memories/<name>", "reason": "<brief reason>", "confidence": 0.88 }
|
|
17
|
-
],
|
|
18
|
-
"warnings": ["<optional concerns>"]
|
|
10
|
+
{ "op": "promote", "ref": "memories/<name>", "knowledgeRef": "knowledge/<suggested-slug>", "reason": "<brief reason>", "description": "<one sentence describing the new knowledge asset>", "confidence": 0.92 }
|
|
11
|
+
]
|
|
19
12
|
}
|
|
20
13
|
|
|
21
14
|
For every operation, emit a `confidence` field in [0, 1] expressing your certainty that the operation is correct and safe. Use 0.95+ only when evidence is unambiguous. Omit the field rather than guessing if you are uncertain.
|
|
@@ -1,10 +1,10 @@
|
|
|
1
1
|
Extract entities and relations from the asset body below.
|
|
2
2
|
|
|
3
3
|
Rules:
|
|
4
|
-
- Output ONLY a JSON object: {"entities": ["Entity One", ...], "relations": [
|
|
4
|
+
- Output ONLY a JSON object: {"entities": ["Entity One", ...], "relations": [["A", "uses", "B"], ...]}.
|
|
5
5
|
- Entities are short, canonical noun phrases (project names, services, tools, people, technical concepts). Do NOT emit file or directory paths (anything containing "/" or "\") — they are dropped downstream.
|
|
6
|
-
- Relations connect two entities that both appear in the entities array.
|
|
7
|
-
- "type" is a short verb phrase (e.g. "uses", "depends on", "owns", "documents").
|
|
6
|
+
- Each relation is a 3-element array: [from, type, to]. Relations connect two entities that both appear in the entities array.
|
|
7
|
+
- "type" is a short verb phrase (e.g. "uses", "depends on", "owns", "documents"). Use "" when unsure.
|
|
8
8
|
- Drop pleasantries, meta-commentary, and timestamps.
|
|
9
9
|
- Limit to at most {{MAX_ENTITIES}} entities and {{MAX_RELATIONS}} relations per asset.
|
|
10
10
|
- Return {"entities": [], "relations": []} if the body has no extractable graph content.
|
|
@@ -19,7 +19,7 @@ for rate limiting. The terraform-provisioner deploys everything to the prod clus
|
|
|
19
19
|
Owner: @alice.
|
|
20
20
|
|
|
21
21
|
Output:
|
|
22
|
-
{"entities":["auth-service","PostgreSQL","redis-cache","terraform-provisioner","prod cluster","@alice"],"relations":[
|
|
22
|
+
{"entities":["auth-service","PostgreSQL","redis-cache","terraform-provisioner","prod cluster","@alice"],"relations":[["auth-service","uses","PostgreSQL"],["auth-service","depends on","redis-cache"],["terraform-provisioner","deploys","prod cluster"],["terraform-provisioner","deploys","auth-service"],["@alice","owns","auth-service"]]}
|
|
23
23
|
|
|
24
24
|
Input:
|
|
25
25
|
## Meeting: API Redesign
|
|
@@ -27,7 +27,7 @@ Discussed moving from REST to GraphQL. The frontend team will use Apollo Client.
|
|
|
27
27
|
Backend needs to implement resolvers. Timeline: Q2.
|
|
28
28
|
|
|
29
29
|
Output:
|
|
30
|
-
{"entities":["REST","GraphQL","Apollo Client","frontend team","backend","resolvers","Q2"],"relations":[
|
|
30
|
+
{"entities":["REST","GraphQL","Apollo Client","frontend team","backend","resolvers","Q2"],"relations":[["frontend team","uses","Apollo Client"],["backend","implements","resolvers"],["frontend team","migrates to","GraphQL"]]}
|
|
31
31
|
|
|
32
32
|
===============
|
|
33
33
|
|
|
@@ -15,6 +15,7 @@
|
|
|
15
15
|
* same read extended to the accepted/rejected archive.
|
|
16
16
|
*/
|
|
17
17
|
import { resolveStashDir } from "../../core/common.js";
|
|
18
|
+
import { isStaleTargetRejection } from "../proposal/proposal-types.js";
|
|
18
19
|
import { listProposals } from "../proposal/repository.js";
|
|
19
20
|
/**
|
|
20
21
|
* Compute accept-rate-per-source metrics from the proposal store. Defaults to
|
|
@@ -28,6 +29,11 @@ export function computeAcceptRateBySource(stashDir) {
|
|
|
28
29
|
for (const status of statuses) {
|
|
29
30
|
const proposals = listProposals(stash, { status, includeArchive });
|
|
30
31
|
for (const p of proposals) {
|
|
32
|
+
// A stale-target auto-reject (STALE, R20) is procedural, not a
|
|
33
|
+
// judgement on the content — counting it would understate the
|
|
34
|
+
// source's real accept rate for content the drain will re-propose.
|
|
35
|
+
if (status === "rejected" && isStaleTargetRejection(p))
|
|
36
|
+
continue;
|
|
31
37
|
const src = p.source || "(unknown)";
|
|
32
38
|
const entry = bySource.get(src) ?? { accepted: 0, rejected: 0, pending: 0 };
|
|
33
39
|
if (status === "accepted")
|