akm-cli 0.9.26 → 0.9.27-alpha.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (39) hide show
  1. package/CHANGELOG.md +237 -0
  2. package/LICENSE +3 -4
  3. package/dist/assets/prompts/consolidate-system.md +2 -2
  4. package/dist/assets/prompts/distill-lesson-system.md +29 -7
  5. package/dist/assets/prompts/extract-session.md +2 -2
  6. package/dist/commands/improve/consolidate/coverage.js +71 -17
  7. package/dist/commands/improve/consolidate/pair-pass.js +12 -9
  8. package/dist/commands/improve/consolidate.js +77 -5
  9. package/dist/commands/improve/distill-guards.js +9 -10
  10. package/dist/commands/improve/distill.js +70 -22
  11. package/dist/commands/improve/extract-prompt.js +61 -39
  12. package/dist/commands/improve/extract.js +2 -1
  13. package/dist/commands/improve/preparation.js +14 -1
  14. package/dist/commands/improve/reflect.js +36 -4
  15. package/dist/commands/improve/retrieval-gate.js +1 -1
  16. package/dist/commands/improve/session-asset.js +3 -2
  17. package/dist/commands/improve/stage.js +35 -53
  18. package/dist/commands/proposal/drain.js +85 -21
  19. package/dist/commands/proposal/proposal-types.js +1 -1
  20. package/dist/core/config/schema/improve-processes.js +3 -3
  21. package/dist/core/paths.js +0 -9
  22. package/dist/indexer/indexer.js +35 -19
  23. package/dist/integrations/harnesses/codex/agent-builder.js +23 -15
  24. package/dist/llm/client.js +44 -14
  25. package/dist/llm/memory-infer.js +1 -1
  26. package/dist/scripts/akm-migrate-node.js +25 -20
  27. package/dist/scripts/akm-migrate.js +25 -20
  28. package/dist/storage/repositories/improve-ledger-repository.js +4 -1
  29. package/dist/storage/repositories/index-entry-schema.js +20 -4
  30. package/dist/storage/repositories/index-fts-repository.js +44 -3
  31. package/dist/storage/repositories/index-schema.js +14 -8
  32. package/docs/README.md +1 -2
  33. package/docs/integration/bundling-akm.md +1 -1
  34. package/docs/migration/v0.8-to-v0.9.md +3 -1
  35. package/docs/reference/README.md +1 -1
  36. package/docs/reference/cli.md +8 -4
  37. package/docs/reference/configuration.md +5 -2
  38. package/docs/reference/data-and-telemetry.md +1 -1
  39. package/package.json +1 -1
@@ -26,7 +26,7 @@ const MAX_QUERIES = 5;
26
26
  const MAX_QUERY_CHARS = 2000;
27
27
  /** The judge sees this much of the body, as in the retrieval eval. */
28
28
  const MAX_DOC_CHARS = 1500;
29
- const GRADE_SCHEMA = {
29
+ export const GRADE_SCHEMA = {
30
30
  type: "object",
31
31
  required: ["grade", "reason"],
32
32
  additionalProperties: false,
@@ -14,9 +14,10 @@ import { assembleAsset } from "../../core/asset/asset-serialize.js";
14
14
  import { conceptIdFromTypeName } from "../../core/asset/resolve-ref.js";
15
15
  import { parseEmbeddedJsonResponse } from "../../core/parse.js";
16
16
  import { recordWrittenPath } from "../../core/write-provenance.js";
17
+ // Every property is required, as a strict structured-output provider needs; an empty `tags` array is none.
17
18
  export const SESSION_SUMMARY_JSON_SCHEMA = {
18
19
  type: "object",
19
- required: ["summary", "key_topics"],
20
+ required: ["summary", "key_topics", "tags"],
20
21
  additionalProperties: false,
21
22
  properties: {
22
23
  summary: { type: "string" },
@@ -61,7 +62,7 @@ export function buildSessionSummaryPrompt(data) {
61
62
  "Transcript:",
62
63
  renderTranscriptForSummary(data.events),
63
64
  "",
64
- 'Respond as JSON: {"summary": string, "key_topics": string[], "tags"?: string[]}.',
65
+ 'Respond as JSON: {"summary": string, "key_topics": string[], "tags": string[]} (an empty array for no tags).',
65
66
  ].join("\n");
66
67
  }
67
68
  /** The summary JSON, tolerating prose around it; `undefined` when nothing usable parses. */
@@ -216,28 +216,31 @@ export function resolveQualityGateJudge(config, profile, processName, onNotices)
216
216
  onNotices?.(resolved.notices);
217
217
  return resolved.runner;
218
218
  }
219
- /** Lesson judge prompt; similar existing lessons let it mark near-duplicates down. */
220
- export function buildJudgePrompt(lessonContent, sourceContent, similarLessons) {
219
+ /** Lesson judge prompt: what the writer was given (the source and its feedback), the assets nearest the new lesson, the lesson. */
220
+ export function buildJudgePrompt(lessonContent, sourceContent, related, feedback) {
221
221
  const lines = [
222
- "You are evaluating a proposed lesson asset for an akm knowledge base.",
222
+ "You are evaluating a lesson an agent wrote from a memory and the feedback about it, for an akm knowledge base.",
223
223
  "",
224
224
  "Score this lesson on each criterion from 1 (poor) to 5 (excellent):",
225
- "1. NOVELTY: Does the lesson add information not already present in the source asset?",
226
- "2. NON-REDUNDANCY: Is this lesson meaningfully different from what the source already says?",
227
- "3. GROUNDING: Is the lesson about what the source asset is about? Score 1-2 only if it is about a different subject than the source; 3 if it is on the source's subject but goes beyond or corrects what the source says (it may draw on feedback you are not shown); 4-5 if the source supports it. A lesson may generalize the source's point.",
225
+ "1. REUSABLE: Does the lesson state a rule an agent can use on another occasion, with the reason it holds? Score 1-2 when it only records what was done, shipped, decided, found or is pending, on a date or for one build, machine or project, or how a system is set up now, however it is phrased. Score 4-5 for a rule with its reason.",
226
+ "2. NON-REDUNDANCY: Is the lesson new next to the existing assets shown below? Score 1-2 only when one of them already states the same rule. Assets on other subjects change nothing: score 4-5 when none is shown or none is on the same subject.",
227
+ "3. GROUNDING: Is every statement in the lesson stated by the source or its feedback, in any words? Check each cause, step, number, rule and limit in the lesson against them. Score 4-5 when each is stated. Score 3 when one stretches what the source says. Score 1-2 when any is in neither, when the lesson drops a limit the source states (one place checked, not confirmed, a guess) and says more than it, or when it is about another subject than the source.",
228
228
  "",
229
- "Source asset content:",
229
+ "Source memory:",
230
230
  "```",
231
231
  // The window distill generates from (buildDistillPrompt): grounding can reject, so the judge reads all of it.
232
232
  sourceContent.slice(0, 3000),
233
233
  "```",
234
234
  ];
235
- if (similarLessons && similarLessons.length > 0) {
236
- lines.push("", "Existing similar lessons (top-3 by similarity). Rate NOVELTY and NON-REDUNDANCY lower if the proposed lesson is substantially similar to any of these:");
237
- for (const sl of similarLessons)
238
- lines.push(`\nExisting lesson ref: ${sl.ref}`, "```", sl.content.slice(0, 500), "```");
235
+ if (feedback && feedback.length > 0) {
236
+ lines.push("", "Feedback recorded about the memory (the writer saw it too):", "```", feedback.join("\n").slice(0, 1500), "```");
237
+ }
238
+ if (related && related.length > 0) {
239
+ lines.push("", "Existing assets nearest the new lesson (they may be on another subject):");
240
+ for (const asset of related)
241
+ lines.push(`\nExisting asset ref: ${asset.ref}`, "```", asset.content.slice(0, 600), "```");
239
242
  }
240
- lines.push("", "Proposed lesson content:", "```", lessonContent.slice(0, 1000), "```", "", 'Return ONLY valid JSON, no prose: {"scores": {"novelty": <1-5 integer>, "nonRedundancy": <1-5 integer>, "grounding": <1-5 integer>}, "reason": "<one sentence>"}');
243
+ lines.push("", "Proposed lesson:", "```", lessonContent.slice(0, 2000), "```", "", 'Return ONLY valid JSON, no prose: {"scores": {"reusable": <1-5 integer>, "nonRedundancy": <1-5 integer>, "grounding": <1-5 integer>}, "reason": "<one sentence naming the weakest criterion>"}');
241
244
  return lines.join("\n");
242
245
  }
243
246
  function boundedDocument(content, maxChars = 6000) {
@@ -317,23 +320,19 @@ export function buildReflectJudgePrompt(candidateContent, sourceContent, feedbac
317
320
  }
318
321
  /**
319
322
  * `grounding` is scored with the other lesson criteria but left out of their
320
- * mean: a lesson about a different subject than its source reads as novel and
321
- * non-redundant, so they would pass it (or, in the review band, mint it as
322
- * a pending proposal). The rubric reserves 1-2 for a different subject. A score
323
- * of {@link UNGROUNDED_MAX_SCORE} or less is a rejection whatever the mean says
324
- * (#999). A higher score up to {@link BORDERLINE_GROUNDING_MAX_SCORE} is only
325
- * borderline: a lesson on its source's subject that advises beyond it has scored
326
- * 2, and a score can move a point between runs (see `runQualityJudge`), so it
327
- * goes to a person unless the mean alone already rejects it. A lesson that goes
328
- * beyond or corrects its source is on its subject: distill folds feedback into
329
- * the lesson, and the judge is never shown it. A contradiction of the source is
330
- * the optional fidelity check's to send to a human (`judgeAndQueue` in
331
- * distill.ts), so the rubric must not pre-empt it.
323
+ * mean. A lesson criterion scored {@link LESSON_REJECT_MAX_SCORE} or less is a
324
+ * rejection whatever the mean says: the mean would hide it (4 and 1 average
325
+ * 2.5, a review), and a reviewer was reading every lesson that was not rejected,
326
+ * 17 of 19 of them bad on 2026-10-05. The rubric reserves 1-2 for a lesson that
327
+ * records what was done instead of a rule, repeats an asset the library holds,
328
+ * or states what neither its source nor its feedback does. The judge is shown
329
+ * the feedback the writer saw, so a statement it supports is not an invention. A
330
+ * contradiction of the source is the optional fidelity check's to send to a
331
+ * human (`judgeAndQueue` in distill.ts).
332
332
  */
333
333
  const GROUNDING_CRITERION = "grounding";
334
- const UNGROUNDED_MAX_SCORE = 1;
335
- const BORDERLINE_GROUNDING_MAX_SCORE = 2;
336
- const LESSON_JUDGE_CRITERIA = ["novelty", "nonRedundancy", GROUNDING_CRITERION];
334
+ const LESSON_REJECT_MAX_SCORE = 2;
335
+ const LESSON_JUDGE_CRITERIA = ["reusable", "nonRedundancy", GROUNDING_CRITERION];
337
336
  const REFLECT_JUDGE_CRITERIA = ["need", "preservation", "quality"];
338
337
  /**
339
338
  * Read a judge response: the per-criterion shape (averaged here, `grounding`
@@ -370,7 +369,7 @@ function parseJudgeResponse(raw, keys) {
370
369
  }
371
370
  return inRange(parsed.score) ? { score: parsed.score, lowest: parsed.score, reason } : undefined;
372
371
  }
373
- function judgeResponseSchema(keys) {
372
+ export function judgeResponseSchema(keys) {
374
373
  return {
375
374
  type: "object",
376
375
  required: ["scores", "reason"],
@@ -390,9 +389,8 @@ function judgeResponseSchema(keys) {
390
389
  * The quality judge. Fails closed: no runner, an unparseable verdict or a
391
390
  * provider failure never passes content. Bands: every criterion in the mean
392
391
  * >= 4 passes, otherwise a mean >= 2.5 is review and a lower one reject; a
393
- * `grounding` score of {@link UNGROUNDED_MAX_SCORE} or less rejects whatever
394
- * the mean is, and one of {@link BORDERLINE_GROUNDING_MAX_SCORE} routes a lesson
395
- * that would pass to review (a mean that rejects stays a rejection).
392
+ * lesson criterion (`grounding` included) of {@link LESSON_REJECT_MAX_SCORE}
393
+ * or less rejects whatever the mean is.
396
394
  * Temperature is set to 0, which reduces run-to-run variation but does not
397
395
  * remove it: on some servers (llama.cpp batching, for one) the same request can
398
396
  * score a point apart, so the routing rules are chosen with that margin in mind.
@@ -430,34 +428,18 @@ async function runQualityJudge(feature, config, prompt, keys, chat, options) {
430
428
  if (!parsed)
431
429
  return { pass: false, score: -1, reason: "judge parse failed — routed to review", reviewNeeded: true };
432
430
  const { score, lowest, reason, criteria } = parsed;
433
- const grounding = criteria?.[GROUNDING_CRITERION];
434
- if (criteria && grounding !== undefined && grounding <= UNGROUNDED_MAX_SCORE) {
435
- return {
436
- pass: false,
437
- score,
438
- reason: `Off-subject for its source (grounding ${grounding}/5): ${reason}`,
439
- criteria,
440
- };
431
+ // A lesson criterion at 2 or below is a defect the mean would hide (4 and 1 average 2.5, a review): it rejects.
432
+ if (criteria && criteria[GROUNDING_CRITERION] !== undefined) {
433
+ const [weakest, low] = Object.entries(criteria).sort((x, y) => x[1] - y[1])[0];
434
+ if (low <= LESSON_REJECT_MAX_SCORE)
435
+ return { pass: false, score, reason: `${weakest} ${low}/5: ${reason}`, criteria };
441
436
  }
442
437
  const verdict = lowest >= 4 ? { pass: true } : score >= 2.5 ? { pass: false, reviewNeeded: true } : { pass: false };
443
- // Borderline grounding is a person's call even when the lesson would pass; a mean that rejects stays rejected.
444
- if (criteria &&
445
- grounding !== undefined &&
446
- grounding <= BORDERLINE_GROUNDING_MAX_SCORE &&
447
- (verdict.pass || verdict.reviewNeeded)) {
448
- return {
449
- pass: false,
450
- reviewNeeded: true,
451
- score,
452
- reason: `Borderline on grounding (${grounding}/5), routed to review: ${reason}`,
453
- criteria,
454
- };
455
- }
456
438
  return { ...verdict, score, reason, ...(criteria ? { criteria } : {}) };
457
439
  }
458
440
  /** Judge a proposed lesson (or knowledge promotion) against its source. */
459
441
  export function runLessonQualityJudge(config, lessonContent, sourceContent, chat, options = {}) {
460
- const prompt = buildJudgePrompt(lessonContent, sourceContent, options.similarLessons);
442
+ const prompt = buildJudgePrompt(lessonContent, sourceContent, options.related, options.feedback);
461
443
  return runQualityJudge("lesson_quality_gate", config, prompt, LESSON_JUDGE_CRITERIA, chat, options);
462
444
  }
463
445
  /** Judge an in-place reflect revision without new-lesson novelty criteria. */
@@ -28,6 +28,7 @@ import { info, warn } from "../../core/warn.js";
28
28
  import { DEFAULT_LLM_TIMEOUT_MS } from "../../integrations/agent/config.js";
29
29
  import { buildExecution, resolveExecution } from "../../integrations/agent/execution.js";
30
30
  import { assertRunnerCredentials, runExecution, } from "../../integrations/agent/runner-dispatch.js";
31
+ import { nearestKnowledgeNotes } from "../improve/consolidate/coverage.js";
31
32
  import { errMessage, noticeSet } from "../improve/stage.js";
32
33
  import { akmProposalAccept, akmProposalReject } from "./proposal.js";
33
34
  import { isRetireProposal, PAIR_PASS_GATE, STALE_TARGET_GATE_REASON } from "./proposal-types.js";
@@ -57,8 +58,13 @@ function categorizeDrainFailure(message, fallback) {
57
58
  * rejection, so instead of failing identically every run it is auto-rejected
58
59
  * once; the ledger records `failed`, keeping the ref re-proposable.
59
60
  */
60
- async function acceptProposal(opts, proposal, id, reason, promoteFn, rejectFn) {
61
- const gateDecision = { outcome: "auto-accepted", reason, gate: DRAIN_GATE };
61
+ async function acceptProposal(opts, proposal, id, reason, promoteFn, rejectFn, judgeReason) {
62
+ const gateDecision = {
63
+ outcome: "auto-accepted",
64
+ reason,
65
+ gate: DRAIN_GATE,
66
+ ...(judgeReason ? { judgeReason } : {}),
67
+ };
62
68
  try {
63
69
  if (!opts.dryRun) {
64
70
  await promoteFn({
@@ -102,7 +108,7 @@ async function acceptProposal(opts, proposal, id, reason, promoteFn, rejectFn) {
102
108
  }
103
109
  }
104
110
  /** Reject one proposal (nothing in a dry run); the error message on failure. */
105
- async function rejectProposal(opts, id, reason, gateReason, rejectFn) {
111
+ async function rejectProposal(opts, id, reason, gateReason, rejectFn, judgeReason) {
106
112
  if (opts.dryRun)
107
113
  return undefined;
108
114
  try {
@@ -110,7 +116,12 @@ async function rejectProposal(opts, id, reason, gateReason, rejectFn) {
110
116
  stashDir: opts.stashDir,
111
117
  id,
112
118
  reason,
113
- gateDecision: { outcome: "auto-rejected", reason: gateReason, gate: DRAIN_GATE },
119
+ gateDecision: {
120
+ outcome: "auto-rejected",
121
+ reason: gateReason,
122
+ gate: DRAIN_GATE,
123
+ ...(judgeReason ? { judgeReason } : {}),
124
+ },
114
125
  });
115
126
  return undefined;
116
127
  }
@@ -118,7 +129,20 @@ async function rejectProposal(opts, id, reason, gateReason, rejectFn) {
118
129
  return errMessage(err);
119
130
  }
120
131
  }
121
- /** The judgment prompt: the proposal, the live asset it would overwrite, and same-ref siblings. */
132
+ /**
133
+ * `text` as a fenced block whose fence is longer than any backtick run inside
134
+ * it (the CommonMark rule), so a note holding its own code block is not cut
135
+ * short at the first inner fence.
136
+ */
137
+ export function fencedBlock(text) {
138
+ let longest = 2;
139
+ for (const run of text.match(/`+/g) ?? [])
140
+ if (run.length > longest)
141
+ longest = run.length;
142
+ const fence = "`".repeat(longest + 1);
143
+ return [fence, text, fence];
144
+ }
145
+ /** The judgment prompt: the proposal, the live asset it would overwrite, same-ref siblings, and for a promotion the nearest knowledge notes. */
122
146
  export function buildJudgmentPrompt(proposal, reason, ctx) {
123
147
  const sections = [
124
148
  "You are adjudicating a pending knowledge-base proposal no quality judge has",
@@ -129,12 +153,10 @@ export function buildJudgmentPrompt(proposal, reason, ctx) {
129
153
  `Left for judgment because: ${reason === "needs-judgment" ? "no quality judge has passed this content yet" : reason}`,
130
154
  "",
131
155
  "## Proposed content",
132
- "```",
133
- proposalContent(proposal),
134
- "```",
156
+ ...fencedBlock(proposalContent(proposal)),
135
157
  ];
136
158
  if (ctx.liveAsset !== undefined) {
137
- sections.push("", "## Current live asset (would be overwritten on accept)", "```", ctx.liveAsset, "```");
159
+ sections.push("", "## Current live asset (would be overwritten on accept)", ...fencedBlock(ctx.liveAsset));
138
160
  }
139
161
  else {
140
162
  sections.push("", "## Current live asset", "(none — this proposal would create a new asset)");
@@ -142,8 +164,15 @@ export function buildJudgmentPrompt(proposal, reason, ctx) {
142
164
  if (ctx.siblings.length > 0) {
143
165
  sections.push("", "## Other pending proposals for the same ref (dedup context)");
144
166
  for (const sib of ctx.siblings) {
145
- sections.push("", `### Sibling ${sib.id} (source: ${sib.source})`, "```", proposalContent(sib), "```");
167
+ sections.push("", `### Sibling ${sib.id} (source: ${sib.source})`, ...fencedBlock(proposalContent(sib)));
168
+ }
169
+ }
170
+ if (ctx.neighbours && ctx.neighbours.length > 0) {
171
+ sections.push("", "## Existing knowledge notes nearest to this promotion's source memory");
172
+ for (const note of ctx.neighbours) {
173
+ sections.push("", `### ${note.ref}`, note.description, ...fencedBlock(note.excerpt));
146
174
  }
175
+ sections.push("", "Reject the promotion if these notes already cover what it says, even in other words.");
147
176
  }
148
177
  sections.push("", "## Your task", 'Return ONLY a JSON object: {"decision": "accept" | "reject" | "defer", "reason": "<short reason>"}.', "- accept: the proposed content is a correct, valuable update worth committing.", "- reject: the proposal is wrong, a duplicate, or contradicts the live asset.", "- defer: you cannot decide from the provided context (leave it pending).", "Output the JSON object and nothing else.");
149
178
  return sections.join("\n");
@@ -205,7 +234,7 @@ async function dispatchJudgment(runner, prompt, seams) {
205
234
  * (under `applyMode` and the remaining accept budget) or the reject. A defer, an
206
235
  * unparseable verdict or a runner error leaves the item undecided.
207
236
  */
208
- async function runJudgmentTier(opts, result, pending, acceptBudget, promoteFn, rejectFn, seams) {
237
+ async function runJudgmentTier(opts, result, pending, acceptBudget, promoteFn, rejectFn, seams, deferNotes) {
209
238
  const byId = new Map(pending.map((p) => [p.id, p]));
210
239
  const notices = noticeSet();
211
240
  const stillDeferred = [];
@@ -216,21 +245,33 @@ async function runJudgmentTier(opts, result, pending, acceptBudget, promoteFn, r
216
245
  stillDeferred.push(item);
217
246
  continue;
218
247
  }
248
+ const liveAsset = readLiveAssetContent(opts.stashDir, proposal.ref);
219
249
  const prompt = buildJudgmentPrompt(proposal, item.reason, {
220
- liveAsset: readLiveAssetContent(opts.stashDir, proposal.ref),
250
+ liveAsset,
221
251
  siblings: pending.filter((p) => p.ref === proposal.ref && p.id !== proposal.id),
252
+ // A create the model otherwise judges blind: the model never sees knowledge/.
253
+ ...(liveAsset === undefined ? { neighbours: promotionNeighbours(opts.stashDir, proposal) } : {}),
222
254
  });
223
255
  const dispatch = await dispatchJudgment(opts.judgment, prompt, seams);
224
256
  notices.add(dispatch.notices);
225
257
  if (dispatch.error)
226
258
  warn(`[triage] judgment dispatch failed for ${item.id}: ${dispatch.error}`);
227
259
  const verdict = dispatch.error ? null : dispatch.verdict;
228
- if (!verdict || verdict.decision === "defer") {
260
+ if (!verdict) {
261
+ deferNotes.set(item.id, { reason: dispatch.error ? "judgment-error" : "judgment-parse-failure" });
262
+ stillDeferred.push(item);
263
+ continue;
264
+ }
265
+ if (verdict.decision === "defer") {
266
+ deferNotes.set(item.id, {
267
+ reason: "judgment-deferred",
268
+ ...(verdict.reason ? { judgeReason: verdict.reason } : {}),
269
+ });
229
270
  stillDeferred.push(item);
230
271
  continue;
231
272
  }
232
273
  if (verdict.decision === "reject") {
233
- const failure = await rejectProposal(opts, item.id, verdict.reason || "judgment: reject", "judgment-reject", rejectFn);
274
+ const failure = await rejectProposal(opts, item.id, verdict.reason || "judgment: reject", "judgment-reject", rejectFn, verdict.reason);
234
275
  if (failure === undefined) {
235
276
  result.rejected.push(item.id);
236
277
  }
@@ -252,6 +293,7 @@ async function runJudgmentTier(opts, result, pending, acceptBudget, promoteFn, r
252
293
  reason: "judgment-accept",
253
294
  contentHash: proposalContentHash(proposal),
254
295
  gate: DRAIN_GATE,
296
+ ...(verdict.reason ? { judgeReason: verdict.reason } : {}),
255
297
  });
256
298
  result.staged.push(item.id);
257
299
  }
@@ -265,7 +307,7 @@ async function runJudgmentTier(opts, result, pending, acceptBudget, promoteFn, r
265
307
  result.skippedByCap.push(item.id);
266
308
  continue;
267
309
  }
268
- const outcome = await acceptProposal(opts, proposal, item.id, "judgment-accept", promoteFn, rejectFn);
310
+ const outcome = await acceptProposal(opts, proposal, item.id, "judgment-accept", promoteFn, rejectFn, verdict.reason);
269
311
  if (outcome === "promoted") {
270
312
  result.promoted.push(item.id);
271
313
  acceptBudget -= 1;
@@ -286,6 +328,21 @@ async function runJudgmentTier(opts, result, pending, acceptBudget, promoteFn, r
286
328
  result.notices = notices.list();
287
329
  result.deferred = stillDeferred;
288
330
  }
331
+ /** The knowledge notes nearest to a consolidate promotion's source memory; none for any other proposal. */
332
+ function promotionNeighbours(stashDir, proposal) {
333
+ if (proposal.source !== "consolidate" || proposal.promotionSource === undefined)
334
+ return [];
335
+ try {
336
+ const parsed = parseRefInput(proposal.promotionSource);
337
+ const typeDir = stashDirFor(parsed.type);
338
+ if (!typeDir)
339
+ return [];
340
+ return nearestKnowledgeNotes(assetPathForName(parsed.type, path.join(stashDir, typeDir), parsed.name));
341
+ }
342
+ catch {
343
+ return [];
344
+ }
345
+ }
289
346
  /** The live asset a proposal would overwrite, if any. */
290
347
  function readLiveAssetContent(stashDir, ref) {
291
348
  try {
@@ -314,8 +371,8 @@ export async function drainProposals(opts, promoteFn = akmProposalAccept, reject
314
371
  const empties = [];
315
372
  for (const proposal of pending) {
316
373
  // A consolidate pair-pass `retire` proposal is auto-accepted only when the
317
- // pair judge staged it as a duplicate with nothing unique on either side
318
- // (spec §25.9, equivalent content); every other one waits for a direct
374
+ // pair pass staged it: nothing unique on either side, confirmed by a
375
+ // second look, no continuity risk; every other one waits for a direct
319
376
  // `akm proposal accept` (spec §25.6). Checked before isEmptyDiff, which
320
377
  // has nothing meaningful to read on a delete-primary change.
321
378
  if (isRetireProposal(proposal)) {
@@ -324,7 +381,7 @@ export async function drainProposals(opts, promoteFn = akmProposalAccept, reject
324
381
  staged.gate === PAIR_PASS_GATE &&
325
382
  staged.contentHash === proposalContentHash(proposal) &&
326
383
  !proposal.retirement?.continuityRisk) {
327
- accepts.push({ id: proposal.id, reason: "duplicate" });
384
+ accepts.push({ id: proposal.id, reason: staged.reason });
328
385
  }
329
386
  continue;
330
387
  }
@@ -394,15 +451,22 @@ export async function drainProposals(opts, promoteFn = akmProposalAccept, reject
394
451
  }
395
452
  }
396
453
  }
454
+ const deferNotes = new Map();
397
455
  if (opts.judgment && result.deferred.length > 0) {
398
- await runJudgmentTier({ ...opts, judgment: opts.judgment }, result, pending, cap - promotedHere, promoteFn, rejectFn, judgmentSeams);
456
+ await runJudgmentTier({ ...opts, judgment: opts.judgment }, result, pending, cap - promotedHere, promoteFn, rejectFn, judgmentSeams, deferNotes);
399
457
  }
400
458
  // #577: whatever stays undecided is left for review (`review_needed` in the ledger).
401
459
  if (!opts.dryRun) {
402
- const reviewReason = opts.judgment ? "judgment-deferred" : "no-judge-configured";
403
460
  for (const item of result.deferred) {
461
+ const note = deferNotes.get(item.id);
462
+ const reviewReason = note?.reason ?? (opts.judgment ? "judgment-deferred" : "no-judge-configured");
404
463
  try {
405
- recordGateDecision(opts.stashDir, item.id, { outcome: "deferred", reason: reviewReason, gate: DRAIN_GATE });
464
+ recordGateDecision(opts.stashDir, item.id, {
465
+ outcome: "deferred",
466
+ reason: reviewReason,
467
+ gate: DRAIN_GATE,
468
+ ...(note?.judgeReason ? { judgeReason: note.judgeReason } : {}),
469
+ });
406
470
  }
407
471
  catch (err) {
408
472
  warn(`[triage] failed to record gate decision for ${item.id}: ${errMessage(err)}`);
@@ -53,7 +53,7 @@ export function isRetireProposal(proposal) {
53
53
  }
54
54
  /** A promote refused because the target changed after mint (STALE, R20) — not a merit judgement. */
55
55
  export const STALE_TARGET_GATE_REASON = "stale-target";
56
- /** The gate on a retire proposal the triage drain may accept unattended: a pair-judged duplicate. */
56
+ /** The gate on a retire proposal the triage drain may accept unattended: a staged pair-judged retirement. */
57
57
  export const PAIR_PASS_GATE = "consolidate-pair";
58
58
  export const EXPIRED_GATE_REASON = "expired";
59
59
  export const ASSET_MISSING_GATE_REASON = "asset-missing";
@@ -78,9 +78,9 @@ const qualityGateField = z
78
78
  .optional();
79
79
  /**
80
80
  * WS-3b: CLS (Complementary Learning System) interleaving (step 9).
81
- * distill/memoryInference prompts include embedding-retrieved existing adjacent
82
- * lessons/knowledge to prevent catastrophic interference with prior generalizations.
83
- * Default OFF. Only meaningful on `distill` and `memoryInference` processes.
81
+ * The distill prompt includes the lessons, knowledge notes and skills the library already holds near the
82
+ * memory, so the writer answers NONE for a rule one of them states and does not overwrite a prior
83
+ * generalization. Default ON; `enabled: false` turns it off. Only meaningful on the `distill` process.
84
84
  */
85
85
  const clsField = z
86
86
  .object({
@@ -332,15 +332,6 @@ export function getStashStateKey(stashDir) {
332
332
  function stashScopedDir(base, stashDir) {
333
333
  return path.join(base, getStashStateKey(stashDir));
334
334
  }
335
- /**
336
- * `$STATE/improve/measurement/verdicts/<stash>/` — `akm-eval-proactive-verdict`
337
- * reports. Moved out of `$STASH/.akm/measurement/verdicts/` (itlackey/akm#890);
338
- * the pilot treatment file at `$STASH/.akm/measurement/` is manually-authored
339
- * measurement input and stays put.
340
- */
341
- export function getMeasurementVerdictsDir(stashDir) {
342
- return stashScopedDir(path.join(getStateDir(), "improve", "measurement", "verdicts"), stashDir);
343
- }
344
335
  /**
345
336
  * `$CACHE/index/unresolved-sources/<stash>/` — synthetic placeholder path for
346
337
  * a configured source whose content root did not resolve this run. Never
@@ -5,7 +5,7 @@ import fs from "node:fs";
5
5
  import path from "node:path";
6
6
  import { detectAdapterId } from "../core/adapter/detect-adapter.js";
7
7
  import { adapterForId } from "../core/adapter/registry.js";
8
- import { isHttpUrl } from "../core/common.js";
8
+ import { compareCodePoints, isHttpUrl } from "../core/common.js";
9
9
  import { classifyPathAccess, describeInaccessiblePath } from "../core/path-access.js";
10
10
  import { getDbPath } from "../core/paths.js";
11
11
  import { SCRIPT_EXTENSIONS } from "../core/recognition-util.js";
@@ -14,6 +14,7 @@ import { isVerbose, warn, warnOnce, warnVerbose } from "../core/warn.js";
14
14
  import { resolveSourcesForOrigin } from "../registry/origin-resolve.js";
15
15
  import { closeDatabase, openExistingDatabase, openIndexDatabase, openReadonlyExistingDatabase, } from "../storage/repositories/index-connection.js";
16
16
  import { deleteEntriesByBundle, deleteEntriesByDirAndBundle, deleteEntriesByDirExceptRefs, deleteEntriesByIds, deleteUsageEventsByEntryIds, findEntryIdByRef, getAllEntries, getEmbeddableEntryCount, getEntryCount, getIndexedBundleIdsByDir, getIndexedDirPathsByBundleId, relinkUsageEvents, upsertEntry, } from "../storage/repositories/index-entries-repository.js";
17
+ import { rebuildFtsIfTotalsStale } from "../storage/repositories/index-fts-repository.js";
17
18
  import { clearStaleCacheEntries } from "../storage/repositories/index-llm-cache-repository.js";
18
19
  import { deleteIndexDirState, deleteMeta, getIndexDirState, getMeta, setMeta, upsertIndexDirState, } from "../storage/repositories/index-meta-repository.js";
19
20
  import { VACUUM_PENDING_META } from "../storage/repositories/index-schema.js";
@@ -125,7 +126,14 @@ export async function runEmbeddingPass(params) {
125
126
  */
126
127
  function finalizeIndex(args) {
127
128
  const { db, sources, sourceDirs, stashDir, signal, onProgress } = args;
128
- onProgress({ phase: "fts", message: "Full-text search index is current." });
129
+ // Rows replaced or removed since its BM25 totals were taken leave them off; recompute them (#1048).
130
+ const rebuilt = rebuildFtsIfTotalsStale(db);
131
+ onProgress({
132
+ phase: "fts",
133
+ message: rebuilt
134
+ ? "Rebuilt the full-text search index to recompute its totals."
135
+ : "Full-text search index is current.",
136
+ });
129
137
  const tFtsEnd = Date.now();
130
138
  // Re-link state.db usage events to the regenerated index and recompute the
131
139
  // derived utility cache. Stored refs already use the current item-ref grammar,
@@ -907,11 +915,6 @@ function requiresWorkflowSourcePreflight(ctxs) {
907
915
  }
908
916
  /** Phase 2 (sync): write all pre-generated scan records inside a single transaction. */
909
917
  function persistDirRecords(db, dirRecords, warnings, bundleByRoot) {
910
- // Per-source dedup: the same logical asset can appear more than once within
911
- // one owning source, where source order still makes the first occurrence win.
912
- // The owner is part of the key so identical concepts in different bundles
913
- // remain distinct indexed rows.
914
- const indexedAssetIdentities = new Set();
915
918
  const deletedUsageEntryIds = new Set();
916
919
  const findPersisted = db.prepare("SELECT id, content_hash, file_path, adapter_id FROM entries WHERE item_ref = ?");
917
920
  const insertTransaction = db.transaction(() => {
@@ -961,7 +964,7 @@ function persistDirRecords(db, dirRecords, warnings, bundleByRoot) {
961
964
  let persistedRows = 0;
962
965
  let dedupedRows = 0;
963
966
  if (stash) {
964
- const ownerIdentity = bundle.bundleId;
967
+ const sourceRoot = `${path.resolve(currentStashDir)}${path.sep}`;
965
968
  for (const entry of stash.entries) {
966
969
  const entryPath = entry.filename ? path.join(dirPath, entry.filename) : null;
967
970
  if (!entryPath) {
@@ -973,23 +976,36 @@ function persistDirRecords(db, dirRecords, warnings, bundleByRoot) {
973
976
  warn(`Skipping entry without adapter-owned concept identity: ${entryPath}`);
974
977
  continue;
975
978
  }
976
- // Adapter-owned concept identity is path-based and cannot be replaced
977
- // by presentation fields such as type/title.
978
- const identityKey = `${ownerIdentity}\0${adapterConceptId}`;
979
- if (indexedAssetIdentities.has(identityKey)) {
980
- dedupedRows++;
981
- continue;
982
- }
983
- indexedAssetIdentities.add(identityKey);
984
979
  // content_hash = doc.hash from the drain, keyed by the recognized
985
980
  // file's path. A missing hash preserves the existing value on upsert.
986
981
  const contentHash = hashByFile?.get(entryPath);
982
+ // Adapter-owned concept identity is path-based and cannot be replaced
983
+ // by presentation fields such as type/title.
987
984
  const provenance = deriveEntryProvenance(bundle, entry.type, entry.name, adapterConceptId);
985
+ // Two files of one source can claim one ref (a skill's references/a.md and
986
+ // knowledge/skills/x/references/a.md are both knowledge/skills/x/references/a). The file with
987
+ // the smaller path holds it, whichever directories this run drains and in whatever order the
988
+ // walk met them (#1050): an entry yields to a row held by a file with a smaller path and
989
+ // otherwise takes the ref over. The same concept in another bundle is a different row, and a
990
+ // holder whose file is gone is no claim.
991
+ const holder = findPersisted.get(provenance.itemRef) ?? undefined;
992
+ if (holder &&
993
+ holder.file_path !== entryPath &&
994
+ holder.file_path.startsWith(sourceRoot) &&
995
+ fs.existsSync(holder.file_path)) {
996
+ const yields = compareCodePoints(holder.file_path, entryPath) < 0;
997
+ const [indexed, skipped] = yields ? [holder.file_path, entryPath] : [entryPath, holder.file_path];
998
+ warnings.push(`Two files claim ${provenance.itemRef}: indexed ${indexed}, skipped ${skipped}.`);
999
+ if (yields) {
1000
+ dedupedRows++;
1001
+ continue;
1002
+ }
1003
+ // The directory the ref comes from now has a row fewer than its files: drain it again.
1004
+ deleteIndexDirState(db, path.dirname(holder.file_path));
1005
+ }
988
1006
  keptItemRefs.add(provenance.itemRef);
989
1007
  persistedRows++;
990
- const previous = sameVariant
991
- ? (findPersisted.get(provenance.itemRef) ?? undefined)
992
- : undefined;
1008
+ const previous = sameVariant ? holder : undefined;
993
1009
  const unchanged = previous !== undefined &&
994
1010
  contentHash !== undefined &&
995
1011
  previous.content_hash === contentHash &&
@@ -19,12 +19,13 @@
19
19
  * `./result-extractor.ts`. Always emitted, mirroring how the Claude builder
20
20
  * always emits `--print`: dispatch is the captured, non-interactive path.
21
21
  * - Codex is the NATIVE-SCHEMA tier (plan §"Structured-output
22
- * normalization"): `req.schema` is written to a temp file and passed via
23
- * `--output-schema <file>`. The file is tiny, uniquely named under the OS
24
- * temp dir, and intentionally NOT cleaned up here — `BuiltCommand` has no
25
- * post-run hook, and the spawned process reads the file after `build()`
26
- * returns. OS temp reaping owns the lifecycle. The engine still validates
27
- * the output defensively (the constrained output is trusted but verified).
22
+ * normalization"): `req.schema` is written to a file and passed via
23
+ * `--output-schema <file>`. The file is named by the hash of the schema
24
+ * and lives in akm's cache dir, so a schema is written once however many
25
+ * units dispatch with it: `BuiltCommand` has no post-run hook, and the
26
+ * spawned process reads the file after `build()` returns, so a file per
27
+ * dispatch could only leak (#1051). The engine still validates the output
28
+ * defensively (the constrained output is trusted but verified).
28
29
  * - `codex exec` has no system-prompt flag; `req.systemPrompt` is folded
29
30
  * into the prompt payload (system text first, blank line, then the task),
30
31
  * after the `--` end-of-options separator so it can never be parsed as a
@@ -52,22 +53,29 @@
52
53
  * that registry, so this builder is reachable under the `"codex"` platform
53
54
  * name without any further wiring.
54
55
  */
55
- import { mkdtempSync, writeFileSync } from "node:fs";
56
- import { tmpdir } from "node:os";
56
+ import { mkdirSync } from "node:fs";
57
57
  import { join } from "node:path";
58
+ import { writeFileAtomic } from "../../../core/common.js";
59
+ import { getCacheDir } from "../../../core/paths.js";
60
+ import { sha256Hex } from "../../../runtime.js";
58
61
  import { resolveDispatchModel } from "../../agent/builder-shared.js";
59
62
  import { createAgentRequestLowerer } from "../../agent/request-lowering.js";
60
63
  /**
61
- * Write a node's JSON Schema to a fresh temp file for `--output-schema`.
64
+ * Write a node's JSON Schema to a file for `--output-schema`, named by the hash
65
+ * of its text, in akm's cache dir. Returns the absolute file path (the value
66
+ * handed to the flag).
62
67
  *
63
- * A unique `mkdtemp` directory per build avoids collisions between concurrent
64
- * fan-out units dispatching in the same process. Returns the absolute file
65
- * path (the value handed to the flag).
68
+ * The same schema is the same file, so concurrent fan-out units share it, and
69
+ * the atomic rename makes a write of identical bytes safe under a reader: no
70
+ * unit can see a half-written schema, and none removes it from another. Nothing
71
+ * is cleaned up, because nothing accumulates but one file per distinct schema.
66
72
  */
67
73
  export function writeCodexOutputSchemaFile(schema) {
68
- const dir = mkdtempSync(join(tmpdir(), "akm-codex-schema-"));
69
- const file = join(dir, "output-schema.json");
70
- writeFileSync(file, `${JSON.stringify(schema, null, 2)}\n`, "utf8");
74
+ const text = `${JSON.stringify(schema, null, 2)}\n`;
75
+ const dir = join(getCacheDir(), "codex-output-schemas");
76
+ mkdirSync(dir, { recursive: true });
77
+ const file = join(dir, `${sha256Hex(text)}.json`);
78
+ writeFileAtomic(file, text);
71
79
  return file;
72
80
  }
73
81
  /**