akm-cli 0.9.27 → 0.9.28-alpha.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -6,6 +6,28 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/).
6
6
 
7
7
  ## [Unreleased]
8
8
 
9
+ ## [0.9.28-alpha.1] - 2026-10-08
10
+
11
+ ### Fixed
12
+
13
+ - **Distill skips a memory marked `beliefState: deprecated` or `superseded`.** Such a note is no longer true and gave
14
+ no lesson (26 of about 900 memories carry the state; distill ran on 17 of them). The improve loop records a
15
+ `distill-skipped` action and an `improve_skipped` event (`distill_deprecated_or_superseded`), and the attempt goes
16
+ in the ledger as `unchanged`; an explicit `--scope` ref still runs. `contradicted` is still distilled.
17
+ - The distill judge scores a lesson 1–2 on non-redundancy when a listed asset
18
+ states most of what it says, not only when it states the same rule, and is
19
+ told to compare only with the listed assets, never with the source memory.
20
+ Before, a lesson that restated a skill the library holds scored 3 ("largely
21
+ redundant") and went to a reviewer. Judge-only replay of the lessons the
22
+ writer produced (local qwen3.8-27b, 3 repeats): of 4 public lessons for
23
+ memories that deserve none, 12 verdicts passed 3 before and 1 after, and of
24
+ 12 such own lessons 33 of 36 passed before and 27 after, while the 31
25
+ lesson-worthy own lessons lost no verdict (2 of 93 rejected before, 0 after).
26
+ - **The pair judge's reason no longer swaps A and B.** In a replay of 120 recorded pairs, 18 of 90 retirement
27
+ reasons said the opposite of what the judge's claim lists decided (for example "B contains all claims from A" for a
28
+ pair where B was retired). The lists were right, so no retirement changed, but the reason a reviewer reads was
29
+ wrong. The prompt now asks the reason to name the asset that can be deleted; two replays gave 0 and 1 of about 85.
30
+
9
31
  ## [0.9.27] - 2026-10-07
10
32
 
11
33
  The stable release of the 0.9.27 line: 0.9.27-alpha.1, alpha.2 and alpha.3.
@@ -16,4 +16,4 @@ Then classify the relation as exactly one of:
16
16
 
17
17
  Set "redundant" to the asset that could be deleted with no loss: "A" for "duplicate", the asset with the empty list for "subsumed", "A" for "supersedes"; else null. Set "stale" to "A" when the relation is "supersedes", else null.
18
18
 
19
- Answer ONLY with JSON: {"onlyInA": ["..."], "onlyInB": ["..."], "relation": "...", "redundant": "A"|"B"|null, "stale": "A"|null, "confidence": 0.0-1.0, "reason": "<at most 25 words>"}
19
+ Answer ONLY with JSON: {"onlyInA": ["..."], "onlyInB": ["..."], "relation": "...", "redundant": "A"|"B"|null, "stale": "A"|null, "confidence": 0.0-1.0, "reason": "<at most 25 words: name the asset that can be deleted and what the other asset still holds>"}
@@ -28,7 +28,7 @@ import { checkDeadUrls } from "../url-checker.js";
28
28
  import { isDistillCandidateRef } from "./eligibility.js";
29
29
  import { shouldSkipRef } from "./improve-strategies.js";
30
30
  import { recordLedgerAttempt, stateKey, stripBundle } from "./ledger.js";
31
- import { hasOnlyBarePositiveFeedback, isFlaggedSinceLastEdit, pushRecentError } from "./preparation.js";
31
+ import { hasOnlyBarePositiveFeedback, isDeprecatedOrSuperseded, isFlaggedSinceLastEdit, pushRecentError, } from "./preparation.js";
32
32
  import { recordNoOp, resetConsecutiveNoOps } from "./salience.js";
33
33
  import { attributeStage, errMessage } from "./stage.js";
34
34
  export function prepareImproveLoopEnv(args) {
@@ -198,6 +198,7 @@ async function runLoopReflectPass(planned, env, tally) {
198
198
  }
199
199
  const FLAGGED_WRONG_REASON = "flagged wrong since its last edit";
200
200
  const BARE_POSITIVE_REASON = "only positive feedback, without a reason";
201
+ const DEPRECATED_REASON = "marked deprecated or superseded";
201
202
  async function runLoopDistillPass(planned, refType, isDistillOnly, env, tally) {
202
203
  const { options, primaryStashDir, improveProfile, resolvedPlan } = env;
203
204
  const distillSkip = shouldSkipRef(planned.ref, "distill", improveProfile);
@@ -232,6 +233,12 @@ async function runLoopDistillPass(planned, refType, isDistillOnly, env, tally) {
232
233
  recordLoopAttempt(planned, env, "distill", "unchanged", BARE_POSITIVE_REASON);
233
234
  return recordSkip(tally, planned.ref, BARE_POSITIVE_REASON, { env, reason: "distill_positive_without_reason" });
234
235
  }
236
+ // A memory marked deprecated or superseded is no longer true, so it gives no lesson. The ledger holds it; an
237
+ // explicit `--scope` ref still runs.
238
+ if (!explicitRefScope && isDeprecatedOrSuperseded(planned)) {
239
+ recordLoopAttempt(planned, env, "distill", "unchanged", DEPRECATED_REASON);
240
+ return recordSkip(tally, planned.ref, DEPRECATED_REASON, { env, reason: "distill_deprecated_or_superseded" });
241
+ }
235
242
  const result = await attributeStage(resolvedPlan, "distill", () => env.distillFn({
236
243
  ref: planned.ref,
237
244
  ...(planned.itemRef ? { itemRef: planned.itemRef } : {}),
@@ -575,6 +575,22 @@ export function isFlaggedSinceLastEdit(candidate, eventsCtx) {
575
575
  return typeof meta.contentHash === "string" ? meta.contentHash === bodyHash : Date.parse(e.ts) > editedAtMs;
576
576
  });
577
577
  }
578
+ /**
579
+ * Whether `candidate`'s file declares `beliefState: deprecated` or `superseded`: a note its owner marked no longer
580
+ * true is no source for a lesson (the 2026-10-08 distill-tune run: 26 of ~900 memories carry the state, none gave
581
+ * a lesson). `contradicted` is left out: it means a conflict is open, not that the note is retired.
582
+ */
583
+ export function isDeprecatedOrSuperseded(candidate) {
584
+ if (!candidate.filePath)
585
+ return false;
586
+ try {
587
+ const state = parseFrontmatter(fs.readFileSync(candidate.filePath, "utf8")).data.beliefState;
588
+ return state === "deprecated" || state === "superseded";
589
+ }
590
+ catch {
591
+ return false;
592
+ }
593
+ }
578
594
  /**
579
595
  * Whether the only feedback `candidate` has inside the signal window is positive and says nothing: no reason, no
580
596
  * note. A bare `--positive` only records that a note helped, which gives the writer nothing to distil, so it restates
@@ -223,7 +223,7 @@ export function buildJudgePrompt(lessonContent, sourceContent, related, feedback
223
223
  "",
224
224
  "Score this lesson on each criterion from 1 (poor) to 5 (excellent):",
225
225
  "1. REUSABLE: Does the lesson state a rule an agent can use on another occasion, with the reason it holds? Score 1-2 when it only records what was done, shipped, decided, found or is pending, on a date or for one build, machine or project, or how a system is set up now, however it is phrased. Score 4-5 for a rule with its reason.",
226
- "2. NON-REDUNDANCY: Is the lesson new next to the existing assets shown below? Score 1-2 only when one of them already states the same rule. Assets on other subjects change nothing: score 4-5 when none is shown or none is on the same subject.",
226
+ '2. NON-REDUNDANCY: Compare the lesson only with the assets listed under "Existing assets nearest the new lesson" (each starts with "Existing asset ref:"); never with the source memory. When that list is absent, score 5. Score 1-2 when a listed asset already states the same rule, or states most of what the lesson says in broader words. Score 4-5 when the lesson gives a rule none of the listed assets gives, or when they are on other subjects.',
227
227
  "3. GROUNDING: Is every statement in the lesson stated by the source or its feedback, in any words? Check each cause, step, number, rule and limit in the lesson against them. Score 4-5 when each is stated. Score 3 when one stretches what the source says. Score 1-2 when any is in neither, when the lesson drops a limit the source states (one place checked, not confirmed, a guess) and says more than it, or when it is about another subject than the source.",
228
228
  "",
229
229
  "Source memory:",
@@ -188,7 +188,7 @@ the set of types the code actually emits at HEAD (verified against every
188
188
  | `improve_invoked` | Start of an `akm improve` run | `ref` (scope); `strategy`, `scope`, `dryRun`, `eligibleCount` |
189
189
  | `improve_completed` | `akm improve` run finished | run stats |
190
190
  | `improve_failed` | `akm improve` run errored | error |
191
- | `improve_skipped` | `akm improve` left a ref, a lane, or a group of refs out | `reason` (`no_new_signal`, `not_retrieved`, `distill_no_new_signal`, `budget_exhausted`, `budget_exhausted_batch`, `asset_missing_on_disk`, `strategy_filtered_all_passes`, `autonomy_gated`, `engine_unavailable`, `pool_below_min_size`, `consolidation_no_memory_updates`, `below_min_new_sessions`, `derived_memory_reflect_skipped`, `memory_distill_requires_feedback`, `distill_flagged_wrong`, `distill_positive_without_reason`); `count`, `remaining`, `strategy`, `lane` or `configKey` where they apply |
191
+ | `improve_skipped` | `akm improve` left a ref, a lane, or a group of refs out | `reason` (`no_new_signal`, `not_retrieved`, `distill_no_new_signal`, `budget_exhausted`, `budget_exhausted_batch`, `asset_missing_on_disk`, `strategy_filtered_all_passes`, `autonomy_gated`, `engine_unavailable`, `pool_below_min_size`, `consolidation_no_memory_updates`, `below_min_new_sessions`, `derived_memory_reflect_skipped`, `memory_distill_requires_feedback`, `distill_flagged_wrong`, `distill_positive_without_reason`, `distill_deprecated_or_superseded`); `count`, `remaining`, `strategy`, `lane` or `configKey` where they apply |
192
192
  | `improve_lock_recovered` | Stale improve lock cleared at startup | |
193
193
  | `improve_review_needed` | `akm feedback` pushed a high-utility asset's utility below the review threshold — a review-needed escalation is recorded (not a proposal, so it can't accidentally overwrite the asset) | `ref`, `previousUtility`, `nextUtility` |
194
194
  | `reflect_invoked` | Start of reflect phase in `akm improve` | `ref`, engine |
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "akm-cli",
3
- "version": "0.9.27",
3
+ "version": "0.9.28-alpha.1",
4
4
  "type": "module",
5
5
  "description": "akm (Agent Knowledge Manager) — a portable, local-first capability library for AI agents. Discover, load, share, and improve reusable skills, scripts, workflows, and knowledge across any shell-capable coding agent, including Claude Code, OpenCode, and Cursor.",
6
6
  "keywords": [