akm-cli 0.9.16-alpha.2 → 0.9.17-alpha.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (49) hide show
  1. package/CHANGELOG.md +495 -1
  2. package/dist/assets/prompts/consolidate-system.md +4 -11
  3. package/dist/assets/prompts/graph-extract-user-prompt.md +5 -5
  4. package/dist/commands/health/accept-rate.js +6 -0
  5. package/dist/commands/health/checks.js +54 -0
  6. package/dist/commands/health/improve-metrics.js +1 -5
  7. package/dist/commands/health/report-view-model.js +0 -1
  8. package/dist/commands/health.js +10 -0
  9. package/dist/commands/improve/consolidate/chunking.js +19 -35
  10. package/dist/commands/improve/consolidate/merge.js +6 -9
  11. package/dist/commands/improve/consolidate.js +104 -91
  12. package/dist/commands/improve/distill/promote-memory.js +40 -2
  13. package/dist/commands/improve/distill/quality-gate.js +186 -23
  14. package/dist/commands/improve/distill.js +42 -8
  15. package/dist/commands/improve/eligibility.js +13 -3
  16. package/dist/commands/improve/improve-cli.js +32 -9
  17. package/dist/commands/improve/improve-strategies.js +23 -1
  18. package/dist/commands/improve/improve.js +121 -84
  19. package/dist/commands/improve/loop-stages.js +241 -108
  20. package/dist/commands/improve/preparation.js +50 -17
  21. package/dist/commands/improve/reflect.js +16 -5
  22. package/dist/commands/improve/shared.js +0 -10
  23. package/dist/commands/proposal/drain.js +79 -10
  24. package/dist/commands/proposal/proposal-types.js +21 -0
  25. package/dist/commands/proposal/repository.js +108 -29
  26. package/dist/core/asset/frontmatter.js +106 -1
  27. package/dist/core/config/schema/improve-processes.js +29 -2
  28. package/dist/core/improve-result.js +9 -0
  29. package/dist/core/paths.js +7 -0
  30. package/dist/indexer/ensure-index.js +52 -7
  31. package/dist/indexer/graph/graph-extraction.js +82 -8
  32. package/dist/indexer/passes/memory-inference.js +16 -1
  33. package/dist/llm/client.js +16 -2
  34. package/dist/llm/graph-extract.js +162 -18
  35. package/dist/output/html-render.js +2 -1
  36. package/dist/output/stdout.js +24 -0
  37. package/dist/output/text.js +4 -3
  38. package/dist/scripts/akm-migrate-node.js +20 -4
  39. package/dist/scripts/akm-migrate.js +20 -4
  40. package/dist/storage/repositories/index-entries-repository.js +43 -0
  41. package/dist/storage/repositories/proposals-repository.js +4 -1
  42. package/dist/storage/state-db-integrity.js +123 -0
  43. package/dist/workflows/program/schema.js +1 -0
  44. package/docs/reference/cli.md +4 -3
  45. package/docs/reference/data-and-telemetry.md +1 -0
  46. package/package.json +1 -1
  47. package/schemas/akm-config.json +44 -0
  48. package/schemas/akm-workflow.json +1 -0
  49. package/dist/commands/improve/eval-cases.js +0 -52
@@ -13,6 +13,7 @@ import { isLlmCredentialAvailable, resolveEngine, } from "../../integrations/age
13
13
  import { executionEngineDefinitionsFromConfig } from "../../integrations/agent/execution-definitions.js";
14
14
  import { loadModelMap, mergeModelMapLayers, parseModelMapLayer, readInstalledModelMapText, resolveModelMapAlias, userModelMapPath, } from "../../integrations/agent/model-map.js";
15
15
  import { probeEndpointOnce } from "../../llm/client.js";
16
+ import { STATE_DB_FREELIST_WARN_RATIO, } from "../../storage/state-db-integrity.js";
16
17
  import { listKeys } from "../env/env.js";
17
18
  import { resolveImprovePlan } from "../improve/improve-strategies.js";
18
19
  import { ENGINE_LAST_USED_LOOKBACK_DAYS } from "./engine-usage.js";
@@ -868,6 +869,59 @@ export const HEALTH_CHECKS = [
868
869
  evidence: { path: ctx.stateDbPath, durationMs: ctx.probe.durationMs },
869
870
  }),
870
871
  },
872
+ {
873
+ // R0: nothing looked at state.db's own SQLite-level integrity before
874
+ // this — the round-trip probe above only proves one row can be appended
875
+ // and read back, which stays true on a database that fails
876
+ // `PRAGMA quick_check` elsewhere (corrupt indexes, out-of-order rowids).
877
+ // Also reports the freelist ratio (fraction of pages VACUUM could
878
+ // reclaim) so a bloated-but-uncorrupted file is visible as a warning
879
+ // rather than silence.
880
+ name: "state-db-integrity",
881
+ channel: "hard",
882
+ run: (ctx) => {
883
+ const { ok, lines, error } = ctx.stateDbIntegrity;
884
+ const { ratio: freelistRatio, error: freelistError } = ctx.stateDbFreelist;
885
+ if (!ok) {
886
+ const detail = error ?? lines.join("; ");
887
+ return {
888
+ name: "state-db-integrity",
889
+ kind: "deterministic",
890
+ status: "fail",
891
+ confidence: "high",
892
+ message: `state.db failed PRAGMA quick_check: ${detail}. Repair: back up state.db, then run ` +
893
+ `sqlite3 state.db ".dump" | sqlite3 state.new.db, verify state.new.db passes quick_check, and swap it in.`,
894
+ evidence: { path: ctx.stateDbPath, lines, freelistRatio },
895
+ };
896
+ }
897
+ if (freelistError) {
898
+ return {
899
+ name: "state-db-integrity",
900
+ kind: "deterministic",
901
+ status: "fail",
902
+ confidence: "high",
903
+ message: `state.db passed PRAGMA quick_check, but reading its freelist/page-count failed: ${freelistError}.`,
904
+ evidence: { path: ctx.stateDbPath, lines, freelistError },
905
+ };
906
+ }
907
+ const freelistWarn = freelistRatio > STATE_DB_FREELIST_WARN_RATIO;
908
+ return {
909
+ name: "state-db-integrity",
910
+ kind: "deterministic",
911
+ status: freelistWarn ? "warn" : "pass",
912
+ confidence: "high",
913
+ message: freelistWarn
914
+ ? `state.db passed PRAGMA quick_check, but ${(freelistRatio * 100).toFixed(1)}% of its pages are free (reclaimable by VACUUM).`
915
+ : "state.db passed PRAGMA quick_check.",
916
+ evidence: {
917
+ path: ctx.stateDbPath,
918
+ freelistCount: ctx.stateDbFreelist.freelistCount,
919
+ pageCount: ctx.stateDbFreelist.pageCount,
920
+ freelistRatio,
921
+ },
922
+ };
923
+ },
924
+ },
871
925
  {
872
926
  name: "state-db-migrations",
873
927
  channel: "hard",
@@ -92,7 +92,6 @@ function createUnknownImproveMetrics() {
92
92
  autoAccept: { promoted: 0, validationFailed: 0 },
93
93
  reflectsWithErrorContext: 0,
94
94
  coverageGapCount: 0,
95
- evalCasesWritten: 0,
96
95
  deadUrlCount: 0,
97
96
  deadUrlsChecked: 0,
98
97
  deadUrlsTotal: 0,
@@ -344,14 +343,13 @@ function applyDistillSkippedAggregate(metrics, result) {
344
343
  }
345
344
  }
346
345
  }
347
- /** autoAccept + the small envelope-level counters (reflectsWithErrorContext, coverageGaps, evalCasesWritten, deadUrls). */
346
+ /** autoAccept + the small envelope-level counters (reflectsWithErrorContext, coverageGaps, deadUrls). */
348
347
  function applyMiscCounters(metrics, result) {
349
348
  metrics.autoAccept.promoted += toFiniteNumber(result.gateAutoAcceptedCount);
350
349
  metrics.autoAccept.validationFailed += toFiniteNumber(result.gateAutoAcceptFailedCount);
351
350
  metrics.reflectsWithErrorContext += toFiniteNumber(result.reflectsWithErrorContext);
352
351
  if (Array.isArray(result.coverageGaps))
353
352
  metrics.coverageGapCount += result.coverageGaps.length;
354
- metrics.evalCasesWritten += toFiniteNumber(result.evalCasesWritten);
355
353
  if (Array.isArray(result.deadUrls))
356
354
  metrics.deadUrlCount += result.deadUrls.length;
357
355
  const deadUrlCoverage = result.deadUrlCoverage;
@@ -593,7 +591,6 @@ function mergeImproveMetrics(dst, src) {
593
591
  dst.autoAccept.validationFailed += src.autoAccept.validationFailed;
594
592
  dst.reflectsWithErrorContext += src.reflectsWithErrorContext;
595
593
  dst.coverageGapCount += src.coverageGapCount;
596
- dst.evalCasesWritten += src.evalCasesWritten;
597
594
  dst.deadUrlCount += src.deadUrlCount;
598
595
  dst.deadUrlsChecked += src.deadUrlsChecked;
599
596
  dst.deadUrlsTotal += src.deadUrlsTotal;
@@ -801,7 +798,6 @@ export function projectImproveRunSummary(row, wallTimeMs, taskId) {
801
798
  memoryInference: perRow.memoryInference,
802
799
  graphExtraction: perRow.graphExtraction,
803
800
  reflectsWithErrorContext: perRow.reflectsWithErrorContext,
804
- evalCasesWritten: perRow.evalCasesWritten,
805
801
  orphansPurged,
806
802
  lintFixed,
807
803
  lintFlagged,
@@ -113,7 +113,6 @@ function reshapeRun(r) {
113
113
  lintFixed: r.lintFixed,
114
114
  reflectsWithErrorContext: r.reflectsWithErrorContext,
115
115
  orphansPurged: r.orphansPurged,
116
- evalCasesWritten: r.evalCasesWritten,
117
116
  };
118
117
  }
119
118
  function compareRuns(a, b) {
@@ -18,6 +18,7 @@ import { countImproveRunsSince } from "../storage/repositories/improve-runs-repo
18
18
  import { closeDatabase, openReadonlyExistingDatabase } from "../storage/repositories/index-connection.js";
19
19
  import { getAllEntries } from "../storage/repositories/index-entries-repository.js";
20
20
  import { queryTaskHistory } from "../storage/repositories/task-history-repository.js";
21
+ import { getStateDbFreelistInfo, runStateDbQuickCheck } from "../storage/state-db-integrity.js";
21
22
  import { pkgVersion } from "../version.js";
22
23
  import { collectImproveAdvisories } from "./health/advisories.js";
23
24
  import { HEALTH_CHECKS, probeActiveImproveStrategy, runHealthEngineProbes, runPendingStateMigrationsCheck, SESSION_EXTRACTION_LEDGER_WINDOW_DAYS, } from "./health/checks.js";
@@ -124,6 +125,11 @@ function gatherTaskHistoryPhase(db, logsDb, since, stateDbPath, now) {
124
125
  const requiredTables = ["events", "proposals", "schema_migrations", "task_history"];
125
126
  const missingTables = requiredTables.filter((name) => !tableNames.includes(name));
126
127
  const probe = probeStateDbRoundTrip(stateDbPath);
128
+ // R0: read-only, independent of the round-trip probe above — quick_check
129
+ // catches corruption a successful append/read cannot (out-of-order rowids,
130
+ // bad index entry counts), and the freelist reading is purely informational.
131
+ const stateDbIntegrity = runStateDbQuickCheck(stateDbPath);
132
+ const stateDbFreelist = getStateDbFreelistInfo(stateDbPath);
127
133
  const taskRows = queryTaskHistory(db, { since });
128
134
  const { withLogs: taskRowsWithLogs, backed: existingLogRows } = partitionLogBackedRows(taskRows, logsDb);
129
135
  const failedTaskRows = taskRows.filter((row) => row.status === "failed");
@@ -145,6 +151,8 @@ function gatherTaskHistoryPhase(db, logsDb, since, stateDbPath, now) {
145
151
  tableNames,
146
152
  missingTables,
147
153
  probe,
154
+ stateDbIntegrity,
155
+ stateDbFreelist,
148
156
  taskRowCount: taskRows.length,
149
157
  taskRowsWithLogsCount: taskRowsWithLogs.length,
150
158
  existingLogRowsCount: existingLogRows.length,
@@ -580,6 +588,8 @@ export async function akmHealth(options = {}) {
580
588
  tableNames,
581
589
  missingTables,
582
590
  probe,
591
+ stateDbIntegrity: taskHistory.stateDbIntegrity,
592
+ stateDbFreelist: taskHistory.stateDbFreelist,
583
593
  taskRowCount: taskHistory.taskRowCount,
584
594
  taskFailRate: taskHistory.taskFailRate,
585
595
  taskRowsWithLogsCount: taskHistory.taskRowsWithLogsCount,
@@ -6,7 +6,7 @@
6
6
  // no orchestrator coupling.
7
7
  import fs from "node:fs";
8
8
  import { parseFrontmatter } from "../../../core/asset/frontmatter.js";
9
- import { cacheHash } from "../content-hash.js";
9
+ import { cacheHash, stripFrontmatterBody } from "../content-hash.js";
10
10
  /**
11
11
  * Conservative chars-per-token estimate used when computing prompt budgets.
12
12
  * English text averages roughly 4 chars/token for most LLM tokenizers. We use
@@ -66,25 +66,19 @@ export function computeSafeChunkSize(contextLength, bodyTruncation, maxChunkSize
66
66
  /**
67
67
  * Build the per-chunk user prompt fed to the consolidate LLM.
68
68
  *
69
- * Each memory is annotated with two flags that drive the system-prompt
70
- * rules at lines 181-186:
71
- * - `(captureMode: hot)` — user-explicit memory; system prompt rule 2
72
- * forbids proposing delete. ~60 wasted LLM verdicts/4h on this user's
73
- * stack before this annotation.
69
+ * Each memory is annotated with two flags:
70
+ * - `(captureMode: hot)` — user-explicit memory, surfaced to the model as
71
+ * provenance context.
74
72
  * - `(already queued)` — the memory's body hash matches a pending
75
- * consolidate proposal; system prompt rule 3 forbids proposing
76
- * promote/merge/contradict. ~107/4h before this annotation.
73
+ * consolidate proposal; the system prompt's PROMOTE rule forbids
74
+ * proposing promote for these. ~107/4h before this annotation.
77
75
  *
78
76
  * Both annotations are visible to the LLM. `pendingProposalBodyHashes`
79
77
  * is precomputed once per run by `loadPendingConsolidateProposalHashes`
80
78
  * so the cost stays O(memories) inside the chunk loop.
81
79
  */
82
- export function buildChunkPrompt(sourceName, memories, chunkIndex, totalChunks, bodyTruncation, pendingProposalBodyHashes = new Set(), standardsContext = "") {
83
- const start = memories[0] ? `memories/${memories[0].name}` : "";
84
- const lastMemory = memories[memories.length - 1];
85
- const end = lastMemory ? `memories/${lastMemory.name}` : "";
80
+ export function buildChunkPrompt(sourceName, memories, chunkIndex, totalChunks, bodyTruncation, pendingProposalBodyHashes = new Set()) {
86
81
  const annotationsByIndex = [];
87
- const hotRefs = [];
88
82
  for (const m of memories) {
89
83
  let body = "";
90
84
  try {
@@ -93,41 +87,31 @@ export function buildChunkPrompt(sourceName, memories, chunkIndex, totalChunks,
93
87
  catch {
94
88
  body = "(unreadable)";
95
89
  }
90
+ // Hot/queued detection stays on the raw body — frontmatter is exactly
91
+ // what parseFrontmatter and the pending-proposal hash domain need.
96
92
  const parsed = parseFrontmatter(body);
97
93
  const isHot = parsed.data.captureMode === "hot";
98
94
  // Use cacheHash (case-preserving stripped body) to match the domain used
99
95
  // by loadPendingConsolidateProposalHashes and the body-embedding cache.
100
96
  const bodyHash = cacheHash(body);
101
97
  const isAlreadyQueued = pendingProposalBodyHashes.has(bodyHash);
102
- annotationsByIndex.push({ isHot, isAlreadyQueued, body });
103
- if (isHot)
104
- hotRefs.push(`memories/${m.name}`);
98
+ // The excerpt shown to the LLM is the body ONLY — frontmatter can run
99
+ // longer than bodyTruncation (~21% of memories), which used to leave the
100
+ // model judging metadata instead of content. stripFrontmatterBody falls
101
+ // back to raw.trim() on unparseable/absent frontmatter, so this is a
102
+ // no-op for the "(unreadable)" placeholder.
103
+ const excerpt = stripFrontmatterBody(body);
104
+ annotationsByIndex.push({ isHot, isAlreadyQueued, excerpt });
105
105
  }
106
106
  const lines = [
107
107
  `Source: ${sourceName}`,
108
- `Chunk ${chunkIndex + 1} of ${totalChunks}, memories ${start}–${end}:`,
108
+ `Chunk ${chunkIndex + 1} of ${totalChunks} (${memories.length} memories):`,
109
109
  "",
110
110
  ];
111
- if (standardsContext.trim()) {
112
- lines.push("Standards to follow (the rulebook for this target):");
113
- lines.push(standardsContext.trim());
114
- lines.push("");
115
- }
116
- // Top-of-prompt protection block for hot refs. Neutral phrasing — avoid
117
- // op-words like "promote", "merge", "contradict" so the model doesn't
118
- // accidentally treat the warning as a hint to use that op elsewhere
119
- // (variant B leaked the word "contradict" into the control sample
120
- // during the diagnostic).
121
- if (hotRefs.length > 0) {
122
- lines.push("⛔ DO NOT propose any `delete` operation for these refs — they are user-explicit (captureMode: hot) and the downstream guard refuses them regardless. Proposing delete for any of these only wastes tokens.");
123
- for (const ref of hotRefs)
124
- lines.push(` - ${ref}`);
125
- lines.push("");
126
- }
127
111
  for (let i = 0; i < memories.length; i++) {
128
112
  const m = memories[i];
129
113
  // `annotationsByIndex` has exactly one entry per memory (built in the loop above).
130
- const { isHot, isAlreadyQueued, body } = annotationsByIndex[i];
114
+ const { isHot, isAlreadyQueued, excerpt } = annotationsByIndex[i];
131
115
  const annotations = [];
132
116
  if (isHot)
133
117
  annotations.push("captureMode: hot");
@@ -138,7 +122,7 @@ export function buildChunkPrompt(sourceName, memories, chunkIndex, totalChunks,
138
122
  lines.push(`Description: ${m.description || "(none)"}`);
139
123
  lines.push(`Tags: ${m.tags.length > 0 ? m.tags.join(", ") : "(none)"}`);
140
124
  lines.push("---");
141
- lines.push(body.slice(0, bodyTruncation));
125
+ lines.push(excerpt.slice(0, bodyTruncation));
142
126
  lines.push("");
143
127
  }
144
128
  return lines.join("\n");
@@ -1,22 +1,19 @@
1
1
  // This Source Code Form is subject to the terms of the Mozilla Public
2
2
  // License, v. 2.0. If a copy of the MPL was not distributed with this
3
3
  // file, You can obtain one at https://mozilla.org/MPL/2.0/.
4
+ // R12a: merge/delete/contradict are no longer requested (see
5
+ // CONSOLIDATE_PLAN_JSON_SCHEMA / consolidate-system.md) — only promote is
6
+ // ever executed. An op of one of those retired shapes — e.g. from a model
7
+ // that ignores the schema — is rejected here so it degrades to the generic
8
+ // "skipping invalid operation" warning in the chunk-judge loop rather than
9
+ // being treated as an actionable (if advisory) plan entry.
4
10
  export function isValidOp(op) {
5
11
  if (typeof op !== "object" || op === null)
6
12
  return false;
7
13
  const o = op;
8
- if (o.op === "merge") {
9
- return typeof o.primary === "string" && Array.isArray(o.secondaries);
10
- }
11
- if (o.op === "delete") {
12
- return typeof o.ref === "string";
13
- }
14
14
  if (o.op === "promote") {
15
15
  return typeof o.ref === "string" && typeof o.knowledgeRef === "string";
16
16
  }
17
- if (o.op === "contradict") {
18
- return typeof o.ref === "string" && typeof o.contradictedByRef === "string";
19
- }
20
17
  return false;
21
18
  }
22
19
  export function mergePlans(chunks, knownRefs) {
@@ -11,7 +11,6 @@ import { parseFrontmatter } from "../../core/asset/frontmatter.js";
11
11
  import { conceptIdFromTypeName, displayRef, parseRefInput } from "../../core/asset/resolve-ref.js";
12
12
  import { getImproveProcessConfig, loadConfig } from "../../core/config/config.js";
13
13
  import { parseEmbeddedJsonResponse } from "../../core/parse.js";
14
- import { resolveStandardsContext } from "../../core/standards/resolve-standards-context.js";
15
14
  import { openStateDatabase } from "../../core/state-db.js";
16
15
  import { parseSinceToIsoLenient } from "../../core/time.js";
17
16
  import { warn, warnVerbose } from "../../core/warn.js";
@@ -48,15 +47,18 @@ const CONSOLIDATE_SYSTEM_PROMPT = consolidateSystemPrompt;
48
47
  /**
49
48
  * JSON Schema for structured consolidate plans (PR 1 of the asset-writers
50
49
  * decision — see knowledge/projects/akm/asset-writers-investigation/00-synthesis).
51
- * Mirrors the {ops[], warnings?[]} shape currently described in
52
- * CONSOLIDATE_SYSTEM_PROMPT. Providers with `supportsJsonSchema: true` enforce
53
- * the shape upstream so the chunk-level "invalid plan from AI — skipping"
54
- * branch in `runConsolidate` becomes unreachable on schema-honouring providers.
50
+ * Mirrors the {ops[]} shape currently described in CONSOLIDATE_SYSTEM_PROMPT.
51
+ * Providers with `supportsJsonSchema: true` enforce the shape upstream so the
52
+ * chunk-level "invalid plan from AI — skipping" branch in `runConsolidate`
53
+ * becomes unreachable on schema-honouring providers.
55
54
  *
56
- * The four operation variants (merge / delete / promote / contradict) are
57
- * modeled as a oneOf so a structured-output provider can still tell them apart
58
- * by the required `op` discriminator. `parseEmbeddedJsonResponse` keeps
59
- * working as a fallback parser for providers that ignore the schema.
55
+ * Promote-only (R12a): `merge`/`delete`/`contradict` were advisory-only — the
56
+ * apply loop only ever executed `promote` — and cost 21-30k completion tokens
57
+ * per run for output nothing acted on. `warnings` is dropped for the same
58
+ * reason. `parseEmbeddedJsonResponse` keeps working as a fallback parser for
59
+ * providers that ignore the schema; `isValidOp` rejects any op shape other
60
+ * than `promote` so a non-compliant response degrades to a skipped-op warning
61
+ * rather than being treated as an actionable plan.
60
62
  */
61
63
  export const CONSOLIDATE_PLAN_JSON_SCHEMA = {
62
64
  type: "object",
@@ -65,70 +67,21 @@ export const CONSOLIDATE_PLAN_JSON_SCHEMA = {
65
67
  properties: {
66
68
  operations: {
67
69
  type: "array",
68
- description: "Ordered list of consolidate operations the planner proposes.",
70
+ description: "Ordered list of promote operations the planner proposes.",
69
71
  items: {
70
- oneOf: [
71
- {
72
- type: "object",
73
- required: ["op", "primary", "secondaries", "mergeStrategy"],
74
- additionalProperties: false,
75
- properties: {
76
- op: { type: "string", enum: ["merge"] },
77
- primary: { type: "string", minLength: 1 },
78
- secondaries: {
79
- type: "array",
80
- minItems: 1,
81
- maxItems: 1,
82
- items: { type: "string", minLength: 1 },
83
- },
84
- mergeStrategy: { type: "string", minLength: 1 },
85
- confidence: { type: "number", minimum: 0, maximum: 1 },
86
- },
87
- },
88
- {
89
- type: "object",
90
- required: ["op", "ref", "reason"],
91
- additionalProperties: false,
92
- properties: {
93
- op: { type: "string", enum: ["delete"] },
94
- ref: { type: "string", minLength: 1 },
95
- reason: { type: "string", minLength: 1 },
96
- confidence: { type: "number", minimum: 0, maximum: 1 },
97
- },
98
- },
99
- {
100
- type: "object",
101
- required: ["op", "ref", "knowledgeRef", "reason"],
102
- additionalProperties: false,
103
- properties: {
104
- op: { type: "string", enum: ["promote"] },
105
- ref: { type: "string", minLength: 1 },
106
- knowledgeRef: { type: "string", minLength: 1 },
107
- reason: { type: "string", minLength: 1 },
108
- description: { type: "string" },
109
- confidence: { type: "number", minimum: 0, maximum: 1 },
110
- },
111
- },
112
- {
113
- type: "object",
114
- required: ["op", "ref", "contradictedByRef", "reason"],
115
- additionalProperties: false,
116
- properties: {
117
- op: { type: "string", enum: ["contradict"] },
118
- ref: { type: "string", minLength: 1 },
119
- contradictedByRef: { type: "string", minLength: 1 },
120
- reason: { type: "string", minLength: 1 },
121
- confidence: { type: "number", minimum: 0, maximum: 1 },
122
- },
123
- },
124
- ],
72
+ type: "object",
73
+ required: ["op", "ref", "knowledgeRef", "reason"],
74
+ additionalProperties: false,
75
+ properties: {
76
+ op: { type: "string", enum: ["promote"] },
77
+ ref: { type: "string", minLength: 1 },
78
+ knowledgeRef: { type: "string", minLength: 1 },
79
+ reason: { type: "string", minLength: 1, maxLength: 200 },
80
+ description: { type: "string" },
81
+ confidence: { type: "number", minimum: 0, maximum: 1 },
82
+ },
125
83
  },
126
84
  },
127
- warnings: {
128
- type: "array",
129
- description: "Optional list of human-readable concerns the planner wants to surface.",
130
- items: { type: "string" },
131
- },
132
85
  },
133
86
  };
134
87
  async function clusterMemoriesBySimilarity(memories, config, stateDb, signal) {
@@ -574,7 +527,7 @@ function resolveConsolidationSourceOwner(opts, stashDir) {
574
527
  * Read and narrow the exact pool the live pass consumes, without embedding,
575
528
  * LLM, proposal, event, or asset writes. Used by both preview and execution.
576
529
  */
577
- export function inspectConsolidationPool(opts, stashDir, warnings, access) {
530
+ export function inspectConsolidationPool(opts, stashDir, warnings, existingKnowledgeBodyHashes = new Set(), access) {
578
531
  const readOnly = access?.readOnly === true;
579
532
  const sourceOwner = resolveConsolidationSourceOwner(opts, stashDir);
580
533
  let memories = loadMemoriesForSource(sourceOwner, warnings, readOnly);
@@ -591,6 +544,13 @@ export function inspectConsolidationPool(opts, stashDir, warnings, access) {
591
544
  if (opts.limit === undefined && memories.length > 150) {
592
545
  warnings.push(`Consolidation: pool has ${memories.length} memories and no limit is set. Consider adding a limit to your consolidate config to prevent timeouts on slow LLM endpoints.`);
593
546
  }
547
+ // Drop already-promoted memories before the limit cap selects its window,
548
+ // so the cap picks from memories the run can actually act on (R2-1: with
549
+ // the pre-filter applied afterward, the cap's oldest-modified-first window
550
+ // was drawn from the unfiltered pool, re-selecting and re-dropping the same
551
+ // permanently-undeletable duplicates on every run).
552
+ const { memories: prefiltered, prefilteredAlreadyPromoted } = prefilterAlreadyPromotedMemories(memories, existingKnowledgeBodyHashes);
553
+ memories = prefiltered;
594
554
  if (opts.limit !== undefined && memories.length > opts.limit) {
595
555
  const mtimeOf = (memory) => {
596
556
  try {
@@ -605,18 +565,60 @@ export function inspectConsolidationPool(opts, stashDir, warnings, access) {
605
565
  warnings.push(`Consolidation: pool capped at ${opts.limit} of ${memories.length} memories (limit option, oldest-modified first).`);
606
566
  memories = memories.slice(0, opts.limit);
607
567
  }
608
- return { poolSize, candidatePoolSize: memories.length, dedupPoolSize, memories };
568
+ return { poolSize, candidatePoolSize: memories.length, dedupPoolSize, memories, prefilteredAlreadyPromoted };
569
+ }
570
+ /**
571
+ * Drop memories whose body already exists verbatim in `knowledge/` before any
572
+ * chunking or LLM work. Hashes the raw file with `cacheHash` (case-preserving
573
+ * stripped body), the same way `loadExistingKnowledgeBodyHashes` built
574
+ * `existingKnowledgeBodyHashes` — so the two sides of the comparison are
575
+ * computed identically. (H1: the post-LLM check, `shouldSkipPromotionBody-
576
+ * Duplicate`, used to hash `cacheHash(parseFrontmatter(memoryContent).content
577
+ * .trim())` — double-stripping frontmatter and diverging from this pre-filter
578
+ * for a memory body starting with its own `---` block. It now hashes
579
+ * `cacheHash(memoryContent)` directly, matching this single-strip domain.)
580
+ * An unreadable memory is kept (fail-safe: let the later passes surface the
581
+ * read error).
582
+ */
583
+ function prefilterAlreadyPromotedMemories(memories, existingKnowledgeBodyHashes) {
584
+ if (existingKnowledgeBodyHashes.size === 0)
585
+ return { memories, prefilteredAlreadyPromoted: 0 };
586
+ const kept = [];
587
+ let prefilteredAlreadyPromoted = 0;
588
+ for (const memory of memories) {
589
+ let raw;
590
+ try {
591
+ raw = fs.readFileSync(memory.filePath, "utf8");
592
+ }
593
+ catch {
594
+ kept.push(memory);
595
+ continue;
596
+ }
597
+ if (existingKnowledgeBodyHashes.has(cacheHash(raw))) {
598
+ prefilteredAlreadyPromoted++;
599
+ }
600
+ else {
601
+ kept.push(memory);
602
+ }
603
+ }
604
+ return { memories: kept, prefilteredAlreadyPromoted };
609
605
  }
610
606
  /**
611
607
  * Pass 1 — narrow the memory pool before any LLM work: drop stale DB entries,
612
- * apply incremental-since narrowing, and cap to `opts.limit` (oldest-modified
613
- * first). Returns an early envelope when the pool empties at any stage;
614
- * otherwise returns the narrowed pool and the state the plan/apply passes
615
- * consume. Behavior-identical to the former inlined narrowing block.
608
+ * apply incremental-since narrowing, pre-filter memories already promoted
609
+ * verbatim into `knowledge/` (R5 (b) — discovered previously only after the
610
+ * LLM chunk call, paying for the judgement on ~84% of the pool just to skip
611
+ * it), and cap to `opts.limit` (oldest-modified first, drawn from the
612
+ * pre-filtered pool — R2-1). Returns an early envelope when the pool empties
613
+ * at any stage; otherwise returns the narrowed pool and the state the
614
+ * plan/apply passes consume.
616
615
  */
617
- async function narrowConsolidationPool(opts, stashDir, startMs, warnings) {
618
- const snapshot = inspectConsolidationPool(opts, stashDir, warnings);
619
- const memories = snapshot.memories;
616
+ async function narrowConsolidationPool(opts, stashDir, startMs, warnings, existingKnowledgeBodyHashes) {
617
+ const snapshot = inspectConsolidationPool(opts, stashDir, warnings, existingKnowledgeBodyHashes);
618
+ const { memories, prefilteredAlreadyPromoted } = snapshot;
619
+ if (prefilteredAlreadyPromoted > 0) {
620
+ warnings.push(`Consolidation: pre-filtered ${prefilteredAlreadyPromoted} memor${prefilteredAlreadyPromoted === 1 ? "y" : "ies"} whose body already exists verbatim in knowledge/ before chunking.`);
621
+ }
620
622
  // (The former WS-3b Step 0a homeostatic demotion pass was removed — R4:
621
623
  // it was default-off and self-undoing (the next salience recompute
622
624
  // unconditionally overwrote the demoted values). Continuous decay now lives
@@ -629,10 +631,11 @@ async function narrowConsolidationPool(opts, stashDir, startMs, warnings) {
629
631
  target: opts.target ?? stashDir,
630
632
  warnings,
631
633
  durationMs: Date.now() - startMs,
634
+ prefilteredAlreadyPromoted,
632
635
  }),
633
636
  };
634
637
  }
635
- return { done: false, memories, dedupPoolSize: snapshot.dedupPoolSize };
638
+ return { done: false, memories, dedupPoolSize: snapshot.dedupPoolSize, prefilteredAlreadyPromoted };
636
639
  }
637
640
  /**
638
641
  * Pass 2 — turn the narrowed pool into an executable plan. Sizes chunks to the
@@ -681,7 +684,7 @@ function recordChunkJudgedNoAction(chunk, ops, accounting) {
681
684
  * are byte-identical, and every counter-increment point is unmoved.
682
685
  */
683
686
  async function judgeConsolidationChunks(args) {
684
- const { chunks, opts, config, llmRunner, lease, sourceName, bodyTruncation, pendingProposalBodyHashes, standardsContext, warnings, accounting, } = args;
687
+ const { chunks, opts, config, llmRunner, lease, sourceName, bodyTruncation, pendingProposalBodyHashes, warnings, accounting, } = args;
685
688
  const chunkOpsArrays = [];
686
689
  // judgedNoAction tracks memories the LLM saw inside a chunk but proposed
687
690
  // no op for. Computed per chunk as `chunk.length − unique(targetRefs in ops)`.
@@ -748,7 +751,7 @@ async function judgeConsolidationChunks(args) {
748
751
  continue;
749
752
  }
750
753
  warn(`[consolidate] chunk ${chunkIdx + 1}/${chunks.length} (${chunk.length} memories) …`);
751
- const userPrompt = buildChunkPrompt(sourceName, chunk, chunkIdx, chunks.length, bodyTruncation, pendingProposalBodyHashes, standardsContext);
754
+ const userPrompt = buildChunkPrompt(sourceName, chunk, chunkIdx, chunks.length, bodyTruncation, pendingProposalBodyHashes);
752
755
  // Single chunk LLM call, wrapped in the feature gate. Deduplicated across
753
756
  // the first attempt and the retry below (the two blocks were byte-identical
754
757
  // apart from their fallback error string). responseSchema lift (PR 1,
@@ -971,10 +974,6 @@ async function planConsolidation(opts, config, stashDir, _startMs, memories, war
971
974
  const pendingProposalBodyHashes = loadPendingConsolidateProposalHashes(stashDir);
972
975
  warn(`[consolidate] ${budgetedMemories.length} memories / ${chunks.length} chunk(s) / chunk_size=${chunkSize}` +
973
976
  ` / pending-proposal hashes: ${pendingProposalBodyHashes.size}`);
974
- // Consolidate output merges memories (non-wiki) → stash authoring standards.
975
- // Resolved ONCE per run and passed to each chunk prompt (facts not re-read
976
- // per chunk).
977
- const standardsContext = resolveStandardsContext("memories/_consolidated", stashDir);
978
977
  const chunkOpsArrays = await judgeConsolidationChunks({
979
978
  chunks,
980
979
  opts,
@@ -984,7 +983,6 @@ async function planConsolidation(opts, config, stashDir, _startMs, memories, war
984
983
  sourceName,
985
984
  bodyTruncation,
986
985
  pendingProposalBodyHashes,
987
- standardsContext,
988
986
  warnings,
989
987
  accounting,
990
988
  });
@@ -1008,11 +1006,18 @@ async function planConsolidation(opts, config, stashDir, _startMs, memories, war
1008
1006
  }
1009
1007
  }
1010
1008
  async function akmConsolidateInner(opts, config, stashDir, startMs, warnings, sharedStateDb) {
1009
+ // Loaded once and shared with the pre-filter (narrowConsolidationPool) and
1010
+ // the post-LLM promote-dedup check (shouldSkipPromotionBodyDuplicate) below
1011
+ // — knowledge/ can hold thousands of files, so walking it twice per run
1012
+ // would double that cost for no benefit. When the caller already walked
1013
+ // knowledge/ for the pool preview (runConsolidationPass, R2-1/R3-1), reuse
1014
+ // that set instead of walking it again here.
1015
+ const existingKnowledgeBodyHashes = opts.existingKnowledgeBodyHashes ?? loadExistingKnowledgeBodyHashes(stashDir);
1011
1016
  // -- Pass 1: narrow the memory pool (may early-return an envelope) ----------
1012
- const narrowed = await narrowConsolidationPool(opts, stashDir, startMs, warnings);
1017
+ const narrowed = await narrowConsolidationPool(opts, stashDir, startMs, warnings, existingKnowledgeBodyHashes);
1013
1018
  if (narrowed.done)
1014
1019
  return narrowed.result;
1015
- const { memories, dedupPoolSize } = narrowed;
1020
+ const { memories, dedupPoolSize, prefilteredAlreadyPromoted } = narrowed;
1016
1021
  // -- Pass 2: build the LLM plan (populates the shared accounting counters) ---
1017
1022
  const accounting = createConsolidateAccounting();
1018
1023
  const { allOps, totalChunks, llmPoolSize, deferredMemories, embedTelemetry, sourceName } = await planConsolidation(opts, config, stashDir, startMs, memories, warnings, sharedStateDb, accounting);
@@ -1035,6 +1040,7 @@ async function akmConsolidateInner(opts, config, stashDir, startMs, warnings, sh
1035
1040
  planned: allOps,
1036
1041
  warnings,
1037
1042
  durationMs: Date.now() - startMs,
1043
+ prefilteredAlreadyPromoted,
1038
1044
  });
1039
1045
  }
1040
1046
  warn(`[consolidate] plan: ${allOps.length} operation(s)`);
@@ -1052,7 +1058,7 @@ async function akmConsolidateInner(opts, config, stashDir, startMs, warnings, sh
1052
1058
  memoryByRef,
1053
1059
  promoted,
1054
1060
  promotedSourceRefs: new Set(),
1055
- existingKnowledgeBodyHashes: loadExistingKnowledgeBodyHashes(opts.writeTarget.source.path),
1061
+ existingKnowledgeBodyHashes,
1056
1062
  promotionFailures,
1057
1063
  warnings,
1058
1064
  pushSkipReason: accounting.pushSkipReason,
@@ -1077,6 +1083,7 @@ async function akmConsolidateInner(opts, config, stashDir, startMs, warnings, sh
1077
1083
  planned: allOps,
1078
1084
  warnings,
1079
1085
  durationMs: Date.now() - startMs,
1086
+ prefilteredAlreadyPromoted,
1080
1087
  perfTelemetry: {
1081
1088
  dedupPoolSize,
1082
1089
  llmPoolSize,
@@ -1203,7 +1210,13 @@ export async function emitPromotionProposal(op, ctx) {
1203
1210
  // different knowledgeRef slug.
1204
1211
  // Use cacheHash (case-preserving stripped body) to match the canonical
1205
1212
  // hash domain used by the body-embedding cache and pending-proposal set.
1206
- const bodyHash = cacheHash(sourceBody);
1213
+ // H1: hash `memoryContent` (one frontmatter strip, inside cacheHash)
1214
+ // rather than the already-stripped `sourceBody` — hashing sourceBody here
1215
+ // double-stripped (parseFrontmatter ran once above to produce sourceBody,
1216
+ // then cacheHash's internal stripFrontmatterBody ran again), diverging from
1217
+ // the single-strip domain `loadExistingKnowledgeBodyHashes` and the
1218
+ // pre-filter use whenever a body starts with its own `---` block.
1219
+ const bodyHash = cacheHash(memoryContent);
1207
1220
  if (shouldSkipPromotionBodyDuplicate({ bodyHash, op, knowledgeRef, ctx }))
1208
1221
  return;
1209
1222
  try {