akm-cli 0.9.16-alpha.2 → 0.9.17-alpha.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (49) hide show
  1. package/CHANGELOG.md +495 -1
  2. package/dist/assets/prompts/consolidate-system.md +4 -11
  3. package/dist/assets/prompts/graph-extract-user-prompt.md +5 -5
  4. package/dist/commands/health/accept-rate.js +6 -0
  5. package/dist/commands/health/checks.js +54 -0
  6. package/dist/commands/health/improve-metrics.js +1 -5
  7. package/dist/commands/health/report-view-model.js +0 -1
  8. package/dist/commands/health.js +10 -0
  9. package/dist/commands/improve/consolidate/chunking.js +19 -35
  10. package/dist/commands/improve/consolidate/merge.js +6 -9
  11. package/dist/commands/improve/consolidate.js +104 -91
  12. package/dist/commands/improve/distill/promote-memory.js +40 -2
  13. package/dist/commands/improve/distill/quality-gate.js +186 -23
  14. package/dist/commands/improve/distill.js +42 -8
  15. package/dist/commands/improve/eligibility.js +13 -3
  16. package/dist/commands/improve/improve-cli.js +32 -9
  17. package/dist/commands/improve/improve-strategies.js +23 -1
  18. package/dist/commands/improve/improve.js +121 -84
  19. package/dist/commands/improve/loop-stages.js +241 -108
  20. package/dist/commands/improve/preparation.js +50 -17
  21. package/dist/commands/improve/reflect.js +16 -5
  22. package/dist/commands/improve/shared.js +0 -10
  23. package/dist/commands/proposal/drain.js +79 -10
  24. package/dist/commands/proposal/proposal-types.js +21 -0
  25. package/dist/commands/proposal/repository.js +108 -29
  26. package/dist/core/asset/frontmatter.js +106 -1
  27. package/dist/core/config/schema/improve-processes.js +29 -2
  28. package/dist/core/improve-result.js +9 -0
  29. package/dist/core/paths.js +7 -0
  30. package/dist/indexer/ensure-index.js +52 -7
  31. package/dist/indexer/graph/graph-extraction.js +82 -8
  32. package/dist/indexer/passes/memory-inference.js +16 -1
  33. package/dist/llm/client.js +16 -2
  34. package/dist/llm/graph-extract.js +162 -18
  35. package/dist/output/html-render.js +2 -1
  36. package/dist/output/stdout.js +24 -0
  37. package/dist/output/text.js +4 -3
  38. package/dist/scripts/akm-migrate-node.js +20 -4
  39. package/dist/scripts/akm-migrate.js +20 -4
  40. package/dist/storage/repositories/index-entries-repository.js +43 -0
  41. package/dist/storage/repositories/proposals-repository.js +4 -1
  42. package/dist/storage/state-db-integrity.js +123 -0
  43. package/dist/workflows/program/schema.js +1 -0
  44. package/docs/reference/cli.md +4 -3
  45. package/docs/reference/data-and-telemetry.md +1 -0
  46. package/package.json +1 -1
  47. package/schemas/akm-config.json +44 -0
  48. package/schemas/akm-workflow.json +1 -0
  49. package/dist/commands/improve/eval-cases.js +0 -52
@@ -54,6 +54,39 @@ export async function planMemoryKnowledgePromotion(ctx) {
54
54
  }
55
55
  return { promotion: promotion, existingKnowledgeContent };
56
56
  }
57
+ /**
58
+ * PRECHECK (tier3-0917): read-only classification of whether distill would
59
+ * promote this memory to knowledge — used by the improve loop's distill
60
+ * pre-generation proposal guard (`loop-stages.ts`) to pre-check the SAME ref
61
+ * {@link planMemoryKnowledgePromotion} will target at dispatch time
62
+ * (knowledgeRef when this resolves `true`, the derived lesson ref
63
+ * otherwise), rather than guessing which of the two a guard hit applies to.
64
+ * Delegates to {@link planMemoryKnowledgePromotion} itself so the
65
+ * classification can never drift from the real dispatch decision — no LLM
66
+ * call, no side effect. Every {@link PromoteMemoryContext} field this
67
+ * classification does not read (chat, fetchSimilarLessonsFn,
68
+ * existingRefVocabulary, …) is a safe placeholder here.
69
+ */
70
+ export async function wouldPromoteMemoryToKnowledge(args) {
71
+ const plan = await planMemoryKnowledgePromotion({
72
+ targetKind: "auto",
73
+ inputRef: args.inputRef,
74
+ durableInputRef: args.durableInputRef,
75
+ assetContent: args.assetContent,
76
+ filteredEvents: args.feedbackEvents,
77
+ config: args.config,
78
+ stash: args.stash,
79
+ lookup: args.lookup,
80
+ fetchSimilarLessonsFn: () => Promise.resolve([]),
81
+ existingRefVocabulary: new Set(),
82
+ outcomeWeightEnabled: false,
83
+ eligMeta: {},
84
+ exclusionSetSize: 0,
85
+ filteredFeedbackCount: 0,
86
+ feedbackFullyFiltered: false,
87
+ });
88
+ return plan !== null;
89
+ }
57
90
  /** Whether a classified promotion will actually call merge generation and/or the judge. */
58
91
  export function memoryKnowledgePromotionRequiresDispatch(ctx, plan) {
59
92
  const hasRunner = Boolean(ctx.llmRunner);
@@ -205,11 +238,16 @@ export async function promoteMemoryToKnowledge(ctx, planned) {
205
238
  ...(ctx.onNotices ? { onNotices: ctx.onNotices } : {}),
206
239
  });
207
240
  if (!judgeResult.pass) {
241
+ const proposalOpts = {
242
+ ...(ctx.proposalsCtx ? { proposalsCtx: ctx.proposalsCtx } : {}),
243
+ ...(ctx.sourceRun !== undefined ? { sourceRun: ctx.sourceRun } : {}),
244
+ ...(ctx.llmRunner?.connection.model ? { modelId: ctx.llmRunner.connection.model } : {}),
245
+ };
208
246
  if (judgeResult.reviewNeeded) {
209
247
  // Uncertainty band (2.5–3.5): queue as review_needed instead of rejecting.
210
- return writeQualityRejection(stash, inputRef, promotion.knowledgeRef, resolvedPromotionContent, judgeResult.score, judgeResult.reason, { reviewNeeded: true }, ctx.eligibilitySource, ctx.eventsCtx);
248
+ return writeQualityRejection(stash, inputRef, promotion.knowledgeRef, resolvedPromotionContent, judgeResult.score, judgeResult.reason, { reviewNeeded: true, ...(judgeResult.criteria ? { criteria: judgeResult.criteria } : {}) }, ctx.eligibilitySource, ctx.eventsCtx, proposalOpts);
211
249
  }
212
- return writeQualityRejection(stash, inputRef, promotion.knowledgeRef, resolvedPromotionContent, judgeResult.score, judgeResult.reason, {}, ctx.eligibilitySource, ctx.eventsCtx);
250
+ return writeQualityRejection(stash, inputRef, promotion.knowledgeRef, resolvedPromotionContent, judgeResult.score, judgeResult.reason, judgeResult.criteria ? { criteria: judgeResult.criteria } : {}, ctx.eligibilitySource, ctx.eventsCtx, proposalOpts);
213
251
  }
214
252
  // Normalize 1-5 judge score to [0, 1]. Only a real passing verdict reaches
215
253
  // here (07 P0-2: the judge now fails CLOSED on no-LLM / timeout / parse
@@ -17,11 +17,14 @@ import { appendEvent } from "../../../core/events.js";
17
17
  import { parseEmbeddedJsonResponse } from "../../../core/parse.js";
18
18
  import { getDistillRejectedDir } from "../../../core/paths.js";
19
19
  import { withStateDb } from "../../../core/state-db.js";
20
+ import { warn } from "../../../core/warn.js";
20
21
  import { recordWrittenPath } from "../../../core/write-provenance.js";
21
22
  import { callStructured } from "../../../llm/structured-call.js";
23
+ import { archiveProposal, isProposalSkipped, recordGateDecision, } from "../../proposal/repository.js";
22
24
  import { akmSearch } from "../../read/search.js";
23
25
  import { scoreEncodingSalience } from "../encoding-salience.js";
24
26
  import { resolveImproveLlmExecution } from "../execution.js";
27
+ import { emitProposal } from "../proposal-envelope.js";
25
28
  import { computeSalience, upsertAssetSalience } from "../salience.js";
26
29
  // ── D-4 / #390: Top-3 similar lessons retrieval ──────────────────────────────
27
30
  /**
@@ -75,8 +78,7 @@ export function buildJudgePrompt(lessonContent, sourceContent, similarLessons) {
75
78
  "",
76
79
  "Score this lesson on each criterion from 1 (poor) to 5 (excellent):",
77
80
  "1. NOVELTY: Does the lesson add information not already present in the source asset?",
78
- "2. ACTIONABILITY: Can an agent follow this lesson without additional context?",
79
- "3. NON-REDUNDANCY: Is this lesson meaningfully different from what the source already says?",
81
+ "2. NON-REDUNDANCY: Is this lesson meaningfully different from what the source already says?",
80
82
  "",
81
83
  "Source asset content:",
82
84
  "```",
@@ -99,7 +101,7 @@ export function buildJudgePrompt(lessonContent, sourceContent, similarLessons) {
99
101
  lines.push(lessonContent.slice(0, 1000));
100
102
  lines.push("```");
101
103
  lines.push("");
102
- lines.push('Return ONLY valid JSON, no prose: {"score": <average score 1-5 as float>, "reason": "<one sentence>"}');
104
+ lines.push('Return ONLY valid JSON, no prose: {"scores": {"novelty": <1-5 integer>, "nonRedundancy": <1-5 integer>}, "reason": "<one sentence>"}');
103
105
  return lines.join("\n");
104
106
  }
105
107
  function boundedDocument(content, maxChars = 6000) {
@@ -156,10 +158,89 @@ export function buildReflectJudgePrompt(candidateContent, sourceContent, feedbac
156
158
  buildChangedRegion(sourceContent, candidateContent),
157
159
  "```",
158
160
  "",
159
- 'Return ONLY valid JSON, no prose: {"score": <average score 1-5 as float>, "reason": "<one sentence>"}',
161
+ 'Return ONLY valid JSON, no prose: {"scores": {"feedbackAlignment": <1-5 integer>, "preservation": <1-5 integer>, "quality": <1-5 integer>}, "reason": "<one sentence>"}',
160
162
  ].join("\n");
161
163
  }
162
- async function runQualityJudge(feature, config, prompt, chat, options = {}) {
164
+ /**
165
+ * Criterion keys `buildJudgePrompt` asks the lesson judge to score.
166
+ * R16: ACTIONABILITY dropped (splinter measured AUC 0.46 against accept/reject
167
+ * outcomes — no signal — and averaging it pulled scores toward the review band).
168
+ */
169
+ const LESSON_JUDGE_CRITERIA_KEYS = ["novelty", "nonRedundancy"];
170
+ /** Criterion keys `buildReflectJudgePrompt` asks the reflect judge to score. */
171
+ const REFLECT_JUDGE_CRITERIA_KEYS = ["feedbackAlignment", "preservation", "quality"];
172
+ /**
173
+ * R16 / r2-2 / JUDGE2: parse the judge's JSON response, accepting either the
174
+ * current per-criterion shape (`{"scores": {...}, "reason"}`, averaged in
175
+ * code) or the old averaged-float shape (`{"score": 1-5, "reason"}`) a model
176
+ * may still return. `expectedCriteriaKeys` names the criteria this judge's
177
+ * prompt asked for; only those keys are read, validated, and averaged — a
178
+ * `scores` object missing any of them is a parse failure (a truncated or
179
+ * partial response can't auto-pass on whatever keys happened to arrive), and
180
+ * any OTHER key present (e.g. a model spelling a key differently, or echoing
181
+ * a criterion the prompt didn't ask for) is silently ignored rather than
182
+ * changing the score or failing the parse. Each expected criterion (or the
183
+ * bare score) must be a finite number in 1..5; anything else — an
184
+ * out-of-range or non-finite value, a non-string `reason` — is a parse
185
+ * failure so the caller routes to review exactly as before.
186
+ */
187
+ function parseJudgeResponse(raw, expectedCriteriaKeys) {
188
+ const parsed = parseEmbeddedJsonResponse(raw);
189
+ if (!parsed || typeof parsed.reason !== "string")
190
+ return undefined;
191
+ const reason = parsed.reason;
192
+ if (parsed.scores !== undefined) {
193
+ if (typeof parsed.scores !== "object" || parsed.scores === null || Array.isArray(parsed.scores))
194
+ return undefined;
195
+ const scores = parsed.scores;
196
+ const criteria = {};
197
+ let sum = 0;
198
+ for (const key of expectedCriteriaKeys) {
199
+ const value = scores[key];
200
+ if (typeof value !== "number" || !Number.isFinite(value) || value < 1 || value > 5)
201
+ return undefined;
202
+ criteria[key] = value;
203
+ sum += value;
204
+ }
205
+ const score = sum / expectedCriteriaKeys.length;
206
+ return { score, reason, criteria };
207
+ }
208
+ if (typeof parsed.score === "number" && Number.isFinite(parsed.score) && parsed.score >= 1 && parsed.score <= 5) {
209
+ return { score: parsed.score, reason };
210
+ }
211
+ return undefined;
212
+ }
213
+ /**
214
+ * JUDGE2: strict JSON Schema for a judge response, sent through the same
215
+ * `supportsJsonSchema`-gated `request.responseSchema` path
216
+ * `src/llm/graph-extract.ts` (`GRAPH_EXTRACTION_JSON_SCHEMA`) uses — a
217
+ * provider that doesn't opt in (`runner.connection.supportsJsonSchema`) sees
218
+ * no change. Built from `expectedCriteriaKeys` so each judge's schema matches
219
+ * exactly the criteria its own prompt asks for; `additionalProperties: false`
220
+ * at both levels means a model that spells a key differently is rejected by
221
+ * a schema-enforcing provider rather than silently producing a parse failure.
222
+ */
223
+ function buildJudgeResponseSchema(expectedCriteriaKeys) {
224
+ const properties = {};
225
+ for (const key of expectedCriteriaKeys) {
226
+ properties[key] = { type: "integer", minimum: 1, maximum: 5 };
227
+ }
228
+ return {
229
+ type: "object",
230
+ required: ["scores", "reason"],
231
+ additionalProperties: false,
232
+ properties: {
233
+ scores: {
234
+ type: "object",
235
+ required: [...expectedCriteriaKeys],
236
+ additionalProperties: false,
237
+ properties,
238
+ },
239
+ reason: { type: "string" },
240
+ },
241
+ };
242
+ }
243
+ async function runQualityJudge(feature, config, prompt, expectedCriteriaKeys, chat, options = {}) {
163
244
  const resolvedDefault = !options.runnerSelectionFrozen && !options.llmRunner
164
245
  ? resolveImproveLlmExecution({ config, processName: `${feature}-judge` })
165
246
  : null;
@@ -183,6 +264,13 @@ async function runQualityJudge(feature, config, prompt, chat, options = {}) {
183
264
  ],
184
265
  request: {
185
266
  enableThinking: false,
267
+ // R13: the judge must not inherit the generation runner's temperature
268
+ // (measured: 10/16 verdict flips at 0.3, 0/16 at 0). Pinned regardless
269
+ // of what `engines.<name>.temperature` the runner resolves.
270
+ temperature: 0,
271
+ // JUDGE2: bounds the response to exactly this judge's criteria on
272
+ // providers that opt into structured output; a no-op otherwise.
273
+ responseSchema: buildJudgeResponseSchema(expectedCriteriaKeys),
186
274
  ...(Object.hasOwn(options, "timeoutMs") ? { timeoutMs: options.timeoutMs } : {}),
187
275
  ...(options.signal ? { signal: options.signal } : {}),
188
276
  ...(chat ? { chat } : {}),
@@ -193,26 +281,20 @@ async function runQualityJudge(feature, config, prompt, chat, options = {}) {
193
281
  fallback: "",
194
282
  ...(options.onNotices ? { onNotices: options.onNotices } : {}),
195
283
  });
196
- const parsed = parseEmbeddedJsonResponse(raw);
197
- if (!parsed ||
198
- typeof parsed.score !== "number" ||
199
- !Number.isFinite(parsed.score) ||
200
- parsed.score < 1 ||
201
- parsed.score > 5 ||
202
- typeof parsed.reason !== "string") {
284
+ const parsed = parseJudgeResponse(raw, expectedCriteriaKeys);
285
+ if (!parsed) {
203
286
  return { pass: false, score: -1, reason: "judge parse failed — routed to review", reviewNeeded: true };
204
287
  }
205
288
  // D-5 / #388: Three-band system (MT-Bench arXiv:2306.05685 — ~±0.5 judge variance).
206
289
  // >= 3.5: auto-queue as pending (pass: true)
207
290
  // 2.5–3.5: review-needed band — uncertain, escalate to human (reviewNeeded: true)
208
291
  // < 2.5: auto-reject (pass: false)
209
- const score = parsed.score;
210
- const reason = parsed.reason ?? "";
292
+ const { score, reason, criteria } = parsed;
211
293
  if (score >= 3.5)
212
- return { pass: true, score, reason };
294
+ return { pass: true, score, reason, ...(criteria ? { criteria } : {}) };
213
295
  if (score >= 2.5)
214
- return { pass: false, score, reason, reviewNeeded: true };
215
- return { pass: false, score, reason };
296
+ return { pass: false, score, reason, reviewNeeded: true, ...(criteria ? { criteria } : {}) };
297
+ return { pass: false, score, reason, ...(criteria ? { criteria } : {}) };
216
298
  }
217
299
  catch (error) {
218
300
  // Invalid symbolic credentials are configuration failures, not a negative
@@ -235,17 +317,37 @@ async function runQualityJudge(feature, config, prompt, chat, options = {}) {
235
317
  * stash. The rejection is `quality_rejected`, not `review_needed`.
236
318
  */
237
319
  export async function runLessonQualityJudge(config, lessonContent, sourceContent, chat, options = {}) {
238
- return runQualityJudge("lesson_quality_gate", config, buildJudgePrompt(lessonContent, sourceContent, options.similarLessons), chat, options);
320
+ return runQualityJudge("lesson_quality_gate", config, buildJudgePrompt(lessonContent, sourceContent, options.similarLessons), LESSON_JUDGE_CRITERIA_KEYS, chat, options);
239
321
  }
240
322
  /** Judge an in-place reflect revision without applying new-lesson novelty criteria. */
241
323
  export async function runReflectQualityJudge(config, candidateContent, sourceContent, feedback, chat, options = {}) {
242
- return runQualityJudge("proposal_quality_gate", config, buildReflectJudgePrompt(candidateContent, sourceContent, feedback), chat, options);
324
+ return runQualityJudge("proposal_quality_gate", config, buildReflectJudgePrompt(candidateContent, sourceContent, feedback), REFLECT_JUDGE_CRITERIA_KEYS, chat, options);
243
325
  }
244
326
  // ── Quality-rejection helper ─────────────────────────────────────────────────
245
327
  /**
246
328
  * Write a rejected lesson to `$STATE/improve/distill-rejected/<stash>/`
247
- * (itlackey/akm#890), append a `distill_invoked` quality-rejected event, and
248
- * return the `quality_rejected` envelope.
329
+ * (itlackey/akm#890), persist it as a real `proposals` row, append a
330
+ * `distill_invoked` quality-rejected event, and return the `quality_rejected`
331
+ * envelope.
332
+ *
333
+ * R10: the proposal row is minted through the same `createProposal`
334
+ * (`emitProposal`) path every other distill proposal takes, so `source:
335
+ * "distill"` fingerprint/backoff bookkeeping (proposal/repository.ts
336
+ * `checkFingerprintAndBackoff`) and the Reflexion "previously rejected"
337
+ * context (distill.ts's `buildDistillMessages`, reflect.ts's
338
+ * `readRejectedProposals`) can see it — before this, a quality rejection
339
+ * left only an event and a `$STATE`-side file nothing read, so the same ref
340
+ * was re-selected and re-rejected on every run. `review_needed` stays
341
+ * `pending` for a human to triage in the normal queue (matching what
342
+ * promote-memory.ts's comment always claimed) and is stamped with a
343
+ * `quality-gate` gate decision so the triage drain's `classifyPendingProposals`
344
+ * (proposal/drain.ts) leaves it pending instead of deferring it to the
345
+ * judgment tier, which could auto-accept it with no human in the loop;
346
+ * `quality_rejected` is minted pending, then immediately archived to
347
+ * `rejected` with the judge's reason.
348
+ * A fingerprint/backoff guard hit here (rare pre-R9; the pre-generation
349
+ * guard is item R9) just means no new row — the envelope + event below are
350
+ * written either way.
249
351
  *
250
352
  * @param stash - Root stash directory.
251
353
  * @param inputRef - The original input ref (for the event).
@@ -255,15 +357,75 @@ export async function runReflectQualityJudge(config, candidateContent, sourceCon
255
357
  * @param reason - Human-readable rejection reason.
256
358
  * @param extraMeta - Optional additional metadata for the event.
257
359
  * @param eventsCtx - Events context so the emit takes appendEvent's fast path (R25).
360
+ * @param proposalOpts - Test seam / attribution passthrough for the minted proposal row.
258
361
  */
259
- export function writeQualityRejection(stash, inputRef, proposalRef, content, score, reason, extraMeta = {}, eligibilitySource, eventsCtx) {
362
+ export function writeQualityRejection(stash, inputRef, proposalRef, content, score, reason, extraMeta = {}, eligibilitySource, eventsCtx, proposalOpts = {}) {
260
363
  // D-5 / #388: reviewNeeded flag selects "review_needed" vs "quality_rejected" outcome.
261
364
  const outcome = extraMeta.reviewNeeded ? "review_needed" : "quality_rejected";
365
+ // r2-1: the mint-time canonical validator inside createProposal (via
366
+ // emitProposal) throws UsageError for structurally-invalid content (e.g. a
367
+ // lessons/ ref missing description/when_to_use). The proposal row here is
368
+ // bookkeeping for backoff/Reflexion, never the authoritative record of the
369
+ // rejection, so a validator throw degrades to "no row minted" — the same
370
+ // bucket as the fingerprint/backoff skip below, not a caller-visible error.
371
+ // r3-1: the archiveProposal call below is guarded the same way, for the
372
+ // same reason.
373
+ let mintedProposal;
374
+ try {
375
+ mintedProposal = emitProposal({ stashDir: stash, ...(proposalOpts.proposalsCtx ? { proposalsCtx: proposalOpts.proposalsCtx } : {}) }, {
376
+ ref: proposalRef,
377
+ source: "distill",
378
+ ...(proposalOpts.sourceRun !== undefined ? { sourceRun: proposalOpts.sourceRun } : {}),
379
+ ...(proposalOpts.modelId !== undefined ? { modelId: proposalOpts.modelId } : {}),
380
+ payload: { content },
381
+ ...(eligibilitySource ? { eligibilitySource } : {}),
382
+ });
383
+ }
384
+ catch {
385
+ mintedProposal = undefined;
386
+ }
387
+ let proposal;
388
+ if (mintedProposal && !isProposalSkipped(mintedProposal)) {
389
+ if (outcome === "quality_rejected") {
390
+ try {
391
+ proposal = archiveProposal(stash, mintedProposal.id, "rejected", reason, proposalOpts.proposalsCtx);
392
+ }
393
+ catch (error) {
394
+ warn(`[akm] writeQualityRejection: failed to archive proposal ${mintedProposal.id} as rejected: ${error instanceof Error ? error.message : String(error)}`);
395
+ }
396
+ }
397
+ else {
398
+ proposal = mintedProposal;
399
+ // REVIEW: stamp the mint so the triage drain's `classifyPendingProposals`
400
+ // skips it instead of deferring it to the judgment tier, which could
401
+ // auto-accept it under `applyMode: promote` with no human ever seeing
402
+ // the review-band content the gate explicitly refused to auto-queue.
403
+ // Best-effort like the mint/archive tolerance above: a stamp failure
404
+ // warns and continues rather than blocking the rejection envelope.
405
+ try {
406
+ proposal =
407
+ recordGateDecision(stash, mintedProposal.id, { outcome: "deferred", reason: "quality-review", gate: "quality-gate" }, proposalOpts.proposalsCtx) ?? proposal;
408
+ }
409
+ catch (error) {
410
+ warn(`[akm] writeQualityRejection: failed to stamp gate decision for ${mintedProposal.id}: ${error instanceof Error ? error.message : String(error)}`);
411
+ }
412
+ }
413
+ }
262
414
  const rejectDir = getDistillRejectedDir(stash);
263
415
  fs.mkdirSync(rejectDir, { recursive: true });
264
416
  const ts = timestampForFilename();
265
417
  const rejectPath = path.join(rejectDir, `${ts}-${proposalRef.replace(/[:/\\]/g, "-")}.md`);
266
- fs.writeFileSync(rejectPath, `---\nscore: ${score}\nreason: ${reason}\noutcome: ${outcome}\n---\n\n${content}`, "utf8");
418
+ // R16: surface the judge's per-criterion scores in the envelope frontmatter
419
+ // when the caller supplied them (a judge-based rejection), same as the event.
420
+ const criteria = extraMeta.criteria && typeof extraMeta.criteria === "object" && !Array.isArray(extraMeta.criteria)
421
+ ? extraMeta.criteria
422
+ : undefined;
423
+ const criteriaFrontmatter = criteria
424
+ ? `criteria:\n${Object.entries(criteria)
425
+ .map(([key, value]) => ` ${key}: ${value}`)
426
+ .join("\n")}\n`
427
+ : "";
428
+ fs.writeFileSync(rejectPath, `---\nscore: ${score}\nreason: ${reason}\noutcome: ${outcome}\n${criteriaFrontmatter}---\n\n${content}`, "utf8");
267
429
  // #652 / itlackey/akm#890: journal it even though it now lands under
268
430
  // `$STATE`, outside the stash's git repo — `result.writtenPaths` reports
269
431
  // every path a run touched, in or out of the stash (describeRunWrittenPaths
@@ -295,6 +457,7 @@ export function writeQualityRejection(stash, inputRef, proposalRef, content, sco
295
457
  proposalRef,
296
458
  score,
297
459
  reason,
460
+ ...(proposal ? { proposalId: proposal.id, proposal } : {}),
298
461
  ...extraMeta,
299
462
  };
300
463
  }
@@ -68,7 +68,8 @@ import { disposeLoweredExecutionDispatchLease, } from "../../integrations/agent/
68
68
  import { callStructured, preflightStructuredLlmRunner } from "../../llm/structured-call.js";
69
69
  import { closeDatabase, openReadonlyExistingDatabase } from "../../storage/repositories/index-connection.js";
70
70
  import { getAllEntries } from "../../storage/repositories/index-entries-repository.js";
71
- import { isProposalSkipped, listProposals, listProposalsReadOnly, proposalContent, } from "../proposal/repository.js";
71
+ import { isStaleTargetRejection } from "../proposal/proposal-types.js";
72
+ import { isProposalSkipped, listProposals, listProposalsReadOnly, } from "../proposal/repository.js";
72
73
  import { stripFrontmatterBody as stripBodyForFidelity } from "./content-hash.js";
73
74
  import { autoRepairLessonFrontmatter, autoSwapDescriptionWhenToUse, collectLessonQualityFindings, repairLessonDescriptionTruncation, } from "./distill/content-repair.js";
74
75
  import { memoryKnowledgePromotionRequiresDispatch, planMemoryKnowledgePromotion, promoteMemoryToKnowledge, } from "./distill/promote-memory.js";
@@ -790,6 +791,8 @@ export async function akmDistill(options) {
790
791
  eligMeta,
791
792
  eventsCtx: options.eventsCtx,
792
793
  stash,
794
+ ...(options.ctx ? { proposalsCtx: options.ctx } : {}),
795
+ ...(options.sourceRun !== undefined ? { sourceRun: options.sourceRun } : {}),
793
796
  });
794
797
  if ("rejection" in assembled)
795
798
  return withNotices(assembled.rejection);
@@ -868,7 +871,11 @@ async function emitDistillLessonProposal(args) {
868
871
  reviewNeeded: true,
869
872
  fidelityContradiction: true,
870
873
  ...(exclusionSet.size > 0 ? { filteredFeedbackCount, feedbackFullyFiltered } : {}),
871
- }, options.eligibilitySource, options.eventsCtx);
874
+ }, options.eligibilitySource, options.eventsCtx, {
875
+ ...(options.ctx ? { proposalsCtx: options.ctx } : {}),
876
+ ...(options.sourceRun !== undefined ? { sourceRun: options.sourceRun } : {}),
877
+ ...(distillRunner?.connection.model ? { modelId: distillRunner.connection.model } : {}),
878
+ });
872
879
  }
873
880
  }
874
881
  catch {
@@ -972,7 +979,7 @@ async function emitDistillLessonProposal(args) {
972
979
  * throwing a `UsageError` on any finding. Extracted verbatim from `akmDistill`.
973
980
  */
974
981
  function assembleAndValidateDistillContent(args) {
975
- const { raw, effectiveProposalKind, inputRef, durableInputRef, itemRef, effectiveLessonRef, exclusionSet, filteredFeedbackCount, eligMeta, eventsCtx, stash, } = args;
982
+ const { raw, effectiveProposalKind, inputRef, durableInputRef, itemRef, effectiveLessonRef, exclusionSet, filteredFeedbackCount, eligMeta, eventsCtx, stash, proposalsCtx, sourceRun, } = args;
976
983
  // Structured-output path: when the provider honoured the JSON schema, `raw`
977
984
  // is a JSON object string (not a markdown blob). Try to parse it and assemble
978
985
  // the canonical `---\nfm\n---\n\nbody` form before using the markdown
@@ -1046,7 +1053,10 @@ function assembleAndValidateDistillContent(args) {
1046
1053
  proposalKind: effectiveProposalKind,
1047
1054
  findingKinds: qualityFindings.map((f) => f.kind),
1048
1055
  ...(exclusionSet.size > 0 ? { filteredFeedbackCount } : {}),
1049
- }, eligMeta.eligibilitySource, eventsCtx),
1056
+ }, eligMeta.eligibilitySource, eventsCtx, {
1057
+ ...(proposalsCtx ? { proposalsCtx } : {}),
1058
+ ...(sourceRun !== undefined ? { sourceRun } : {}),
1059
+ }),
1050
1060
  };
1051
1061
  }
1052
1062
  return { content, descriptionSwapped };
@@ -1185,16 +1195,25 @@ async function applyDistillQualityGate(args) {
1185
1195
  onNotices,
1186
1196
  });
1187
1197
  if (!judgeResult.pass) {
1198
+ const proposalOpts = {
1199
+ ...(options.ctx ? { proposalsCtx: options.ctx } : {}),
1200
+ ...(options.sourceRun !== undefined ? { sourceRun: options.sourceRun } : {}),
1201
+ ...(distillRunner?.connection.model ? { modelId: distillRunner.connection.model } : {}),
1202
+ };
1188
1203
  if (judgeResult.reviewNeeded) {
1189
1204
  return {
1190
1205
  rejection: writeQualityRejection(stash, inputRef, effectiveLessonRef, content, judgeResult.score, judgeResult.reason, {
1191
1206
  reviewNeeded: true,
1207
+ ...(judgeResult.criteria ? { criteria: judgeResult.criteria } : {}),
1192
1208
  ...(exclusionSet.size > 0 ? { filteredFeedbackCount, feedbackFullyFiltered } : {}),
1193
- }, options.eligibilitySource, options.eventsCtx),
1209
+ }, options.eligibilitySource, options.eventsCtx, proposalOpts),
1194
1210
  };
1195
1211
  }
1196
1212
  return {
1197
- rejection: writeQualityRejection(stash, inputRef, effectiveLessonRef, content, judgeResult.score, judgeResult.reason, exclusionSet.size > 0 ? { filteredFeedbackCount, feedbackFullyFiltered } : {}, options.eligibilitySource, options.eventsCtx),
1213
+ rejection: writeQualityRejection(stash, inputRef, effectiveLessonRef, content, judgeResult.score, judgeResult.reason, {
1214
+ ...(judgeResult.criteria ? { criteria: judgeResult.criteria } : {}),
1215
+ ...(exclusionSet.size > 0 ? { filteredFeedbackCount, feedbackFullyFiltered } : {}),
1216
+ }, options.eligibilitySource, options.eventsCtx, proposalOpts),
1198
1217
  };
1199
1218
  }
1200
1219
  // Normalize 1-5 judge score to [0, 1]. Only a real passing verdict
@@ -1240,12 +1259,21 @@ async function buildDistillMessages(args) {
1240
1259
  const { options, stash, inputRef, assetContent, feedback, effectiveProposalKind, effectiveLessonRef, fetchSimilarLessonsFn, } = args;
1241
1260
  // Inject last 1–3 rejected proposals for this ref as Reflexion-style
1242
1261
  // verbal-RL context so the LLM avoids regenerating refused proposals.
1262
+ // Exclude the drain's stale-target auto-rejects (STALE, R20): those are a
1263
+ // procedural refusal (the target changed after mint), not a judgement on
1264
+ // the content, and would mislead this Reflexion-style "don't repeat this"
1265
+ // context.
1243
1266
  const rejectedForRef = listProposalsReadOnly(stash, { ref: inputRef, status: "rejected", includeArchive: true }, options.ctx)
1267
+ .filter((p) => !isStaleTargetRejection(p))
1244
1268
  .sort((a, b) => new Date(b.updatedAt ?? 0).getTime() - new Date(a.updatedAt ?? 0).getTime())
1245
1269
  .slice(0, MAX_REJECTED_PROPOSALS)
1246
1270
  .map((p) => ({
1247
1271
  reason: p.review?.reason ?? "no reason given",
1248
- contentPreview: proposalContent(p).slice(0, 500),
1272
+ // #legacy: `changes` is empty for pre-existing rows (storedToChanges),
1273
+ // which makes `proposalContent` throw before reflect dispatch even
1274
+ // runs. `payload.content` is populated for every row regardless, so
1275
+ // read the preview from there instead.
1276
+ contentPreview: p.payload.content.slice(0, 500),
1249
1277
  }));
1250
1278
  // WS-3b CLS interleaving (step 9).
1251
1279
  // When cls.enabled, inject embedding-retrieved adjacent lessons/knowledge
@@ -1282,7 +1310,13 @@ async function buildDistillMessages(args) {
1282
1310
  { role: "user", content: userPrompt },
1283
1311
  ];
1284
1312
  }
1285
- async function defaultLookup(ref, stashDir) {
1313
+ /**
1314
+ * Exported (PRECHECK, tier3-0917) so the improve loop's distill
1315
+ * pre-generation guard (`loop-stages.ts`) can resolve the same asset path
1316
+ * `akmDistill` would when checking whether a memory promotes to knowledge —
1317
+ * without duplicating the resolution logic.
1318
+ */
1319
+ export async function defaultLookup(ref, stashDir) {
1286
1320
  return resolveAssetPath(ref, {
1287
1321
  stashDir,
1288
1322
  mode: "disk-only",
@@ -467,12 +467,22 @@ export function buildLatestProposalTsMap(refs, source, itemRefByRef, eventsCtx)
467
467
  if (!ref)
468
468
  continue;
469
469
  // For distill_invoked we only count attempts that produced (or attempted
470
- // to produce) a real proposal — config_disabled / parse-error outcomes
471
- // should not move the signal-delta cursor forward.
470
+ // to produce) a real proposal — config_disabled outcomes (no LLM work was
471
+ // actually invoked) and llm_failed (transport/timeout — nothing to show
472
+ // for the attempt) should not move the signal-delta cursor forward.
473
+ // R10: quality_rejected and review_needed now also mint a real proposal
474
+ // row (see quality-gate.ts's `writeQualityRejection`), so — like queued —
475
+ // a real attempt was made and the cursor must advance, or the same
476
+ // already-rejected ref is re-selected and re-rejected on every run.
472
477
  if (eventType === "distill_invoked") {
473
478
  const outcome = e.metadata?.outcome;
474
- if (outcome !== "queued" && outcome !== "skipped" && outcome !== "validation_failed")
479
+ if (outcome !== "queued" &&
480
+ outcome !== "skipped" &&
481
+ outcome !== "validation_failed" &&
482
+ outcome !== "quality_rejected" &&
483
+ outcome !== "review_needed") {
475
484
  continue;
485
+ }
476
486
  }
477
487
  const ts = e.ts ?? "";
478
488
  if (ts > (out.get(ref) ?? ""))
@@ -128,26 +128,47 @@ function collectRequiredEngineTargets(plan) {
128
128
  * insufficient here: a gateway can list a model while its upstream completion
129
129
  * route is dead (#980). Deduplicate by endpoint + model, not endpoint alone,
130
130
  * because model backends behind one gateway can fail independently.
131
+ *
132
+ * R17: a probe that PASSES used to leave no trace — a slow or flapping
133
+ * gateway was invisible in the improve result. On success, return one
134
+ * {@link EngineProbeOutcome} per target (process, engine, endpoint,
135
+ * reachable, latencyMs) so the caller can record it on the run result;
136
+ * targets sharing a deduplicated probe share its measured latency.
137
+ *
138
+ * Exported for unit tests, which inject a fake `probeReachable` (the
139
+ * "probe seam") instead of hitting a real endpoint.
131
140
  */
132
- async function assertRequiredEnginesReachable(plan, probeReachable = (connection) => probeLlmReachable(connection, 3_000)) {
141
+ export async function assertRequiredEnginesReachable(plan, probeReachable = (connection) => probeLlmReachable(connection, 3_000)) {
133
142
  const targets = collectRequiredEngineTargets(plan);
134
143
  if (targets.length === 0)
135
- return;
144
+ return [];
136
145
  const probesByConnection = new Map();
137
146
  const probed = await Promise.all(targets.map(async (target) => {
138
147
  const key = `${target.connection.endpoint.replace(/\/+$/, "")}|${target.connection.model}`;
139
148
  let pending = probesByConnection.get(key);
140
149
  if (!pending) {
141
- pending = probeReachable(target.connection);
150
+ const probeStartedAt = Date.now();
151
+ pending = probeReachable(target.connection).then((reach) => ({
152
+ reach,
153
+ latencyMs: Date.now() - probeStartedAt,
154
+ }));
142
155
  probesByConnection.set(key, pending);
143
156
  }
144
- return { ...target, reach: await pending };
157
+ const { reach, latencyMs } = await pending;
158
+ return { ...target, reach, latencyMs };
145
159
  }));
146
160
  const unreachable = probed.filter((item) => !item.reach.reachable);
147
- if (unreachable.length === 0)
148
- return;
149
- const lines = unreachable.map((item) => ` - ${item.process} (engine "${item.engine}", ${item.connection.endpoint}): ${item.reach.error ?? "did not respond"}`);
150
- throw new ConfigError(`--require-engines: ${unreachable.length} improve process${unreachable.length === 1 ? "" : "es"} cannot run because ${unreachable.length === 1 ? "its" : "their"} engine completion path is not reachable:\n${lines.join("\n")}`, "LLM_NOT_CONFIGURED");
161
+ if (unreachable.length > 0) {
162
+ const lines = unreachable.map((item) => ` - ${item.process} (engine "${item.engine}", ${item.connection.endpoint}): ${item.reach.error ?? "did not respond"}`);
163
+ throw new ConfigError(`--require-engines: ${unreachable.length} improve process${unreachable.length === 1 ? "" : "es"} cannot run because ${unreachable.length === 1 ? "its" : "their"} engine completion path is not reachable:\n${lines.join("\n")}`, "LLM_NOT_CONFIGURED");
164
+ }
165
+ return probed.map((item) => ({
166
+ process: item.process,
167
+ engine: item.engine,
168
+ endpoint: item.connection.endpoint,
169
+ reachable: item.reach.reachable,
170
+ latencyMs: item.latencyMs,
171
+ }));
151
172
  }
152
173
  /**
153
174
  * `--show-prompt` (#952): render the composed reflect prompt for one asset ref
@@ -343,9 +364,10 @@ export const improveCommand = defineCommand({
343
364
  await runShowPromptCli(scopeArg, scopeRef, taskArg, targetArg, resolvedPlan);
344
365
  return;
345
366
  }
367
+ let engineProbe;
346
368
  if (args["require-engines"]) {
347
369
  assertRequiredEnginesAvailable(resolvedPlan);
348
- await assertRequiredEnginesReachable(resolvedPlan);
370
+ engineProbe = await assertRequiredEnginesReachable(resolvedPlan);
349
371
  }
350
372
  const selectedStrategyName = resolvedPlan.strategy.name;
351
373
  const sensitiveValues = collectEngineCredentialValues(effectiveConfig);
@@ -422,6 +444,7 @@ export const improveCommand = defineCommand({
422
444
  ...(requireFeedbackSignal ? { requireFeedbackSignal } : {}),
423
445
  ...(skipIfLocked ? { skipIfLocked } : {}),
424
446
  ...(strategyArg !== undefined ? { strategy: strategyArg } : {}),
447
+ ...(engineProbe !== undefined ? { engineProbe } : {}),
425
448
  ...(Object.keys(syncOverride).length > 0 ? { sync: syncOverride } : {}),
426
449
  consolidateOptions: {
427
450
  target: targetArg,
@@ -9,7 +9,7 @@ import proactiveMaintenance from "../../assets/improve-strategies/proactive-main
9
9
  import quick from "../../assets/improve-strategies/quick.json" with { type: "json" };
10
10
  import reflectDistill from "../../assets/improve-strategies/reflect-distill.json" with { type: "json" };
11
11
  import thorough from "../../assets/improve-strategies/thorough.json" with { type: "json" };
12
- import { parseRefInput } from "../../core/asset/resolve-ref.js";
12
+ import { conceptIdFromTypeName, parseRefInput } from "../../core/asset/resolve-ref.js";
13
13
  import { ImproveProfileConfigSchema } from "../../core/config/config-schema.js";
14
14
  import { deepMergeConfig } from "../../core/config/deep-merge.js";
15
15
  import { BUILTIN_IMPROVE_STRATEGY_NAMES, IMPROVE_PROCESS_ENGINE_CAPABILITIES, } from "../../core/config/engine-semantics.js";
@@ -27,6 +27,14 @@ export function resolveProcessEnabled(processName, strategy) {
27
27
  const processes = strategy.processes;
28
28
  return processes?.[processName]?.enabled === true;
29
29
  }
30
+ /** `bundle//conceptId` -> bare `conceptId`, for an `excludeRefPrefixes` entry. */
31
+ function stripBundlePrefix(value) {
32
+ const boundary = value.indexOf("//");
33
+ const stripped = boundary >= 0 ? value.slice(boundary + 2) : value;
34
+ // A trailing `/` (e.g. "knowledge/wikis/articles/raw/") would otherwise turn the
35
+ // segment-boundary check below into `startsWith(".../raw//")`, which never matches.
36
+ return stripped.replace(/\/+$/, "");
37
+ }
30
38
  export function shouldSkipRef(ref, processName, strategy) {
31
39
  const process = strategy.processes?.[processName];
32
40
  if (process?.enabled === false)
@@ -35,6 +43,20 @@ export function shouldSkipRef(ref, processName, strategy) {
35
43
  const allowed = process?.allowedTypes ?? DEFAULT_ALLOWED_TYPES[processName];
36
44
  if (!allowed.includes(parsed.type))
37
45
  return { skip: true, reason: "type-filter" };
46
+ // R12: reflect only — raw wiki-ingest snapshots are type `knowledge`, so
47
+ // allowedTypes alone can't exclude them (distill/consolidate are memory-only).
48
+ if (processName === "reflect") {
49
+ const excludePrefixes = strategy.processes?.reflect?.excludeRefPrefixes;
50
+ if (excludePrefixes && excludePrefixes.length > 0) {
51
+ const conceptId = conceptIdFromTypeName(parsed.type, parsed.name);
52
+ const excluded = excludePrefixes.some((prefix) => {
53
+ const stripped = stripBundlePrefix(prefix);
54
+ return conceptId === stripped || conceptId.startsWith(`${stripped}/`);
55
+ });
56
+ if (excluded)
57
+ return { skip: true, reason: "exclude-filter" };
58
+ }
59
+ }
38
60
  return { skip: false, reason: "" };
39
61
  }
40
62
  export function isStrategyFilteredForAllPasses(ref, strategy) {