akm-cli 0.9.27-alpha.1 → 0.9.27-alpha.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +131 -0
- package/LICENSE +3 -4
- package/dist/assets/prompts/distill-lesson-system.md +29 -7
- package/dist/commands/improve/consolidate/coverage.js +71 -17
- package/dist/commands/improve/consolidate/pair-pass.js +11 -8
- package/dist/commands/improve/consolidate.js +63 -1
- package/dist/commands/improve/distill-guards.js +9 -10
- package/dist/commands/improve/distill.js +62 -18
- package/dist/commands/improve/preparation.js +14 -1
- package/dist/commands/improve/stage.js +34 -52
- package/dist/commands/proposal/drain.js +90 -22
- package/dist/commands/proposal/proposal-types.js +1 -1
- package/dist/core/config/schema/improve-processes.js +3 -3
- package/dist/core/paths.js +0 -9
- package/dist/storage/repositories/improve-ledger-repository.js +4 -1
- package/docs/README.md +1 -2
- package/docs/integration/bundling-akm.md +1 -1
- package/docs/migration/v0.8-to-v0.9.md +3 -1
- package/docs/reference/README.md +1 -1
- package/docs/reference/cli.md +7 -4
- package/docs/reference/data-and-telemetry.md +1 -1
- package/package.json +1 -1
|
@@ -78,23 +78,29 @@ export function deriveLessonRef(inputRef) {
|
|
|
78
78
|
// properties, so every property is required and "none" is an empty array (#1046).
|
|
79
79
|
export const DISTILL_LESSON_JSON_SCHEMA = {
|
|
80
80
|
type: "object",
|
|
81
|
-
required: ["description", "when_to_use", "body", "tags"],
|
|
81
|
+
required: ["reason", "decision", "description", "when_to_use", "body", "tags"],
|
|
82
82
|
additionalProperties: false,
|
|
83
83
|
properties: {
|
|
84
|
+
reason: {
|
|
85
|
+
type: "string",
|
|
86
|
+
description: "One sentence, written first: the cause and the fix the memory states, or why it states none.",
|
|
87
|
+
},
|
|
88
|
+
decision: {
|
|
89
|
+
type: "string",
|
|
90
|
+
enum: ["lesson", "none"],
|
|
91
|
+
description: "`none` when the memory holds no lesson (it records what was done, a design, or steps already written elsewhere): leave the other fields empty. Otherwise `lesson`.",
|
|
92
|
+
},
|
|
84
93
|
description: {
|
|
85
94
|
type: "string",
|
|
86
|
-
|
|
87
|
-
description: "Single complete sentence (80-200 chars) summarising what the lesson teaches. No markdown, no leading 'When'/'If'.",
|
|
95
|
+
description: "Single complete sentence summarising what the lesson teaches. No markdown, no leading 'When'/'If'. Empty for `none`.",
|
|
88
96
|
},
|
|
89
97
|
when_to_use: {
|
|
90
98
|
type: "string",
|
|
91
|
-
|
|
92
|
-
description: "Single complete sentence describing the concrete trigger condition for the lesson.",
|
|
99
|
+
description: "Single complete sentence describing the concrete trigger condition for the lesson. Empty for `none`.",
|
|
93
100
|
},
|
|
94
101
|
body: {
|
|
95
102
|
type: "string",
|
|
96
|
-
|
|
97
|
-
description: "Lesson body — plain markdown, 1-3 short paragraphs of practical guidance.",
|
|
103
|
+
description: "Lesson body: plain markdown, shorter than the memory, stating only what it and its feedback say. Empty for `none`.",
|
|
98
104
|
},
|
|
99
105
|
tags: {
|
|
100
106
|
type: "array",
|
|
@@ -154,6 +160,19 @@ export function assembleStructuredDistillMarkdown(payload, kind) {
|
|
|
154
160
|
fm.xrefs = sources;
|
|
155
161
|
return assembleAssetFromString(serializeFrontmatterQuoted(fm), body);
|
|
156
162
|
}
|
|
163
|
+
/**
|
|
164
|
+
* The writer's answer when it found no lesson: the word NONE, or `decision: "none"` in a reply bound to the schema,
|
|
165
|
+
* with the reason it gave (`""` for the bare word). `null` for any other reply.
|
|
166
|
+
*/
|
|
167
|
+
function answeredNone(raw) {
|
|
168
|
+
if (/^[\s"'`*_]*none[\s.!"'`*_]*$/i.test(stripMarkdownFences(raw)))
|
|
169
|
+
return { reason: "" };
|
|
170
|
+
const payload = parseEmbeddedJsonResponse(raw);
|
|
171
|
+
if (payload === null || typeof payload !== "object" || Array.isArray(payload) || payload.decision !== "none") {
|
|
172
|
+
return null;
|
|
173
|
+
}
|
|
174
|
+
return { reason: typeof payload.reason === "string" ? payload.reason.trim() : "" };
|
|
175
|
+
}
|
|
157
176
|
function validateKnowledgeContent(content, inputRef) {
|
|
158
177
|
const findings = [];
|
|
159
178
|
const parsed = parseFrontmatter(content);
|
|
@@ -251,7 +270,7 @@ export function buildDistillPrompt(input) {
|
|
|
251
270
|
}
|
|
252
271
|
lines.push(input.proposalKind === "knowledge"
|
|
253
272
|
? "Produce the knowledge markdown file now. Start your response with `---` on the first line, followed by a `description:` field whose value is a 1-sentence summary (20–400 chars). Never use placeholder values like `---`, `tbd`, `n/a`, or a single dash. If the source has nothing meaningful to summarize, do NOT produce a proposal — return an empty response instead. The frontmatter block ends with a second `---` line; do not emit any additional `---` fences in the body."
|
|
254
|
-
: "Produce the lesson markdown file now. Start your response with `---` on the first line, followed by `description:` and `when_to_use:` fields. Both must be real one-sentence summaries (20–400 chars) — never placeholder values like `---`, `tbd`, or `n/a`. The frontmatter block ends with a second `---` line; do not emit any additional `---` fences in the body.");
|
|
273
|
+
: "Produce the lesson markdown file now. Start your response with `---` on the first line, followed by `description:` and `when_to_use:` fields. Both must be real one-sentence summaries (20–400 chars) — never placeholder values like `---`, `tbd`, or `n/a`. The frontmatter block ends with a second `---` line; do not emit any additional `---` fences in the body. If the memory holds no lesson, answer NONE instead.");
|
|
255
274
|
return lines.join("\n");
|
|
256
275
|
}
|
|
257
276
|
// ── Invocation ───────────────────────────────────────────────────────────────
|
|
@@ -343,7 +362,7 @@ export async function akmDistill(options) {
|
|
|
343
362
|
asset,
|
|
344
363
|
vocabulary: loadRefVocabulary(),
|
|
345
364
|
outcomeWeightEnabled: config.improve?.salience?.outcomeWeightEnabled !== false,
|
|
346
|
-
|
|
365
|
+
related: options.fetchRelatedFn ?? fetchRelatedAssets,
|
|
347
366
|
lookup,
|
|
348
367
|
};
|
|
349
368
|
const feedbackEvents = readDistillFeedback(run);
|
|
@@ -381,6 +400,8 @@ async function distill(run, targetKind, kind, outputRef, feedbackEvents) {
|
|
|
381
400
|
system,
|
|
382
401
|
prompt,
|
|
383
402
|
gate: { config: run.config, enabled: true },
|
|
403
|
+
// NONE is an answer: a parser that rejected it would ask for a lesson again.
|
|
404
|
+
parse: (raw) => (answeredNone(raw) ? raw : parseEmbeddedJsonResponse(raw)),
|
|
384
405
|
// The injected test transport never sees the schema.
|
|
385
406
|
request: {
|
|
386
407
|
...(run.options.chat === undefined
|
|
@@ -413,6 +434,10 @@ async function distill(run, targetKind, kind, outputRef, feedbackEvents) {
|
|
|
413
434
|
...exclusionMeta(run, true),
|
|
414
435
|
};
|
|
415
436
|
}
|
|
437
|
+
const none = answeredNone(call.raw);
|
|
438
|
+
if (none) {
|
|
439
|
+
return skipDistill(run, outputRef, kind, "nothing_reusable", `The writer found no lesson in ${run.inputRef}${none.reason ? `: ${none.reason}` : "."}`);
|
|
440
|
+
}
|
|
416
441
|
const assembled = assembleDistilledContent(run, call.raw, kind, outputRef);
|
|
417
442
|
if ("rejection" in assembled)
|
|
418
443
|
return assembled.rejection;
|
|
@@ -422,8 +447,20 @@ async function distill(run, targetKind, kind, outputRef, feedbackEvents) {
|
|
|
422
447
|
content: assembled.content,
|
|
423
448
|
source: run.asset.content,
|
|
424
449
|
descriptionSwapped: assembled.descriptionSwapped,
|
|
450
|
+
feedback: feedbackLines(feedback),
|
|
425
451
|
});
|
|
426
452
|
}
|
|
453
|
+
/** The feedback that says something, one line each, for the judge. A bare signal says nothing the writer could use. */
|
|
454
|
+
function feedbackLines(feedback) {
|
|
455
|
+
const lines = [];
|
|
456
|
+
for (const event of feedback) {
|
|
457
|
+
const meta = event.metadata ?? {};
|
|
458
|
+
const detail = (typeof meta.reason === "string" ? meta.reason : "") || (typeof meta.note === "string" ? meta.note : "");
|
|
459
|
+
if (detail.trim())
|
|
460
|
+
lines.push(`- [${typeof meta.signal === "string" ? meta.signal : event.eventType}] ${detail.trim()}`);
|
|
461
|
+
}
|
|
462
|
+
return lines;
|
|
463
|
+
}
|
|
427
464
|
/** Whether a file already holds the lesson `ref` in the stash the proposal would be filed in. */
|
|
428
465
|
function lessonExists(run, ref) {
|
|
429
466
|
const { type, name } = parseRefInput(ref);
|
|
@@ -489,11 +526,12 @@ async function judgeAndQueue(run, out) {
|
|
|
489
526
|
let confidence;
|
|
490
527
|
let judged;
|
|
491
528
|
if (qualityGateEnabled(run)) {
|
|
492
|
-
const
|
|
529
|
+
const related = await run.related(content.slice(0, 500), RELATED_COUNT);
|
|
493
530
|
// The judge reads what the generator read: the source body, without its frontmatter (buildDistillPrompt).
|
|
494
531
|
const source = out.source ? parseFrontmatter(out.source).content.trim() : "";
|
|
495
532
|
const verdict = await runLessonQualityJudge(run.config, content, source, run.options.chat, {
|
|
496
|
-
...(
|
|
533
|
+
...(related.length > 0 ? { related } : {}),
|
|
534
|
+
...(out.feedback && out.feedback.length > 0 ? { feedback: out.feedback } : {}),
|
|
497
535
|
...((run.judgeRunner ?? run.runner) ? { llmRunner: run.judgeRunner ?? run.runner } : {}),
|
|
498
536
|
...(run.options.signal ? { signal: run.options.signal } : {}),
|
|
499
537
|
onNotices: run.notices.add,
|
|
@@ -873,13 +911,13 @@ function readDistillFeedback(run) {
|
|
|
873
911
|
/** System + user prompt: rejected-proposal context, optional CLS neighbours, stash standards. */
|
|
874
912
|
async function buildDistillMessages(run, feedback, kind, outputRef) {
|
|
875
913
|
const rejectedProposals = rejectedProposalContext(run.stash, run.inputRef, run.options.ctx, run.options.eventsCtx);
|
|
876
|
-
// CLS interleaving (default
|
|
914
|
+
// CLS interleaving (default on): show the related lessons, knowledge notes and skills, so the writer neither repeats nor overwrites them.
|
|
877
915
|
const cls = getImproveProcessConfig("distill", run.profile)?.cls ?? {};
|
|
878
916
|
let clsContext = "";
|
|
879
|
-
if (cls.enabled) {
|
|
917
|
+
if (cls.enabled !== false) {
|
|
880
918
|
try {
|
|
881
919
|
const query = run.asset.content ? run.asset.content.slice(0, 500) : run.inputRef;
|
|
882
|
-
clsContext = buildClsContext(await run.
|
|
920
|
+
clsContext = buildClsContext(await run.related(query, cls.adjacentCount ?? DEFAULT_CLS_ADJACENT_COUNT), cls);
|
|
883
921
|
}
|
|
884
922
|
catch {
|
|
885
923
|
// CLS context is supplemental.
|
|
@@ -908,12 +946,18 @@ async function defaultLookup(ref, stashDir) {
|
|
|
908
946
|
honorOrigin: false,
|
|
909
947
|
});
|
|
910
948
|
}
|
|
911
|
-
/**
|
|
912
|
-
|
|
949
|
+
/** What the library already holds on a subject: lessons and knowledge notes say it, a skill is how to do it. */
|
|
950
|
+
const RELATED_TYPES = ["lesson", "knowledge", "skill"];
|
|
951
|
+
const RELATED_COUNT = 3;
|
|
952
|
+
/** The top-N lessons, knowledge notes and skills related to `query`, best first (empty when search is unavailable). */
|
|
953
|
+
async function fetchRelatedAssets(query, n) {
|
|
913
954
|
try {
|
|
914
|
-
|
|
915
|
-
|
|
955
|
+
// One search per type: memories outnumber the rest and would fill an untyped list.
|
|
956
|
+
const results = await Promise.all(RELATED_TYPES.map((type) => akmSearch({ query, type, limit: n, skipLogging: true, eventSource: "improve" })));
|
|
957
|
+
return results
|
|
958
|
+
.flatMap((result) => result?.hits ?? [])
|
|
916
959
|
.filter((h) => "path" in h && typeof h.path === "string")
|
|
960
|
+
.sort((a, b) => (b.score ?? 0) - (a.score ?? 0))
|
|
917
961
|
.slice(0, n)
|
|
918
962
|
.map((h) => {
|
|
919
963
|
let content = "";
|
|
@@ -601,6 +601,16 @@ export function buildSnapshotManifest(args) {
|
|
|
601
601
|
const latestNegativeTs = new Map();
|
|
602
602
|
const feedback = new Map(candidates.map((r) => [r.ref, { hasSignal: false, positive: 0, negative: 0 }]));
|
|
603
603
|
if (candidates.length > 0) {
|
|
604
|
+
// When each ref's accepted feedback proposals were created: a fix event is acted on once one exists at or after it.
|
|
605
|
+
const fixedAt = new Map();
|
|
606
|
+
if (stashDir) {
|
|
607
|
+
withRunState(eventsCtx, args.readOnly !== true, (db) => {
|
|
608
|
+
for (const p of listStateProposals(db, { stashDir, status: "accepted" })) {
|
|
609
|
+
if (p.source === "feedback")
|
|
610
|
+
fixedAt.set(p.ref, [...(fixedAt.get(p.ref) ?? []), p.createdAt]);
|
|
611
|
+
}
|
|
612
|
+
});
|
|
613
|
+
}
|
|
604
614
|
for (const e of readEvents({ type: "feedback" }, eventsCtx).events) {
|
|
605
615
|
const ref = e.ref ? refByKey.get(e.ref) : undefined;
|
|
606
616
|
const entry = ref ? feedback.get(ref) : undefined;
|
|
@@ -612,8 +622,11 @@ export function buildSnapshotManifest(args) {
|
|
|
612
622
|
entry.hasSignal = true;
|
|
613
623
|
if (ts > (latestFeedbackTs.get(ref) ?? ""))
|
|
614
624
|
latestFeedbackTs.set(ref, ts);
|
|
615
|
-
|
|
625
|
+
const fixApplied = e.metadata?.fix !== undefined &&
|
|
626
|
+
(fixedAt.get(e.ref ?? "") ?? []).some((createdAt) => createdAt >= ts);
|
|
627
|
+
if (signal === "negative" && !fixApplied && ts > (latestNegativeTs.get(ref) ?? "")) {
|
|
616
628
|
latestNegativeTs.set(ref, ts);
|
|
629
|
+
}
|
|
617
630
|
}
|
|
618
631
|
if (signal === "positive")
|
|
619
632
|
entry.positive++;
|
|
@@ -216,28 +216,31 @@ export function resolveQualityGateJudge(config, profile, processName, onNotices)
|
|
|
216
216
|
onNotices?.(resolved.notices);
|
|
217
217
|
return resolved.runner;
|
|
218
218
|
}
|
|
219
|
-
/** Lesson judge prompt
|
|
220
|
-
export function buildJudgePrompt(lessonContent, sourceContent,
|
|
219
|
+
/** Lesson judge prompt: what the writer was given (the source and its feedback), the assets nearest the new lesson, the lesson. */
|
|
220
|
+
export function buildJudgePrompt(lessonContent, sourceContent, related, feedback) {
|
|
221
221
|
const lines = [
|
|
222
|
-
"You are evaluating a
|
|
222
|
+
"You are evaluating a lesson an agent wrote from a memory and the feedback about it, for an akm knowledge base.",
|
|
223
223
|
"",
|
|
224
224
|
"Score this lesson on each criterion from 1 (poor) to 5 (excellent):",
|
|
225
|
-
"1.
|
|
226
|
-
"2. NON-REDUNDANCY: Is
|
|
227
|
-
"3. GROUNDING: Is the lesson
|
|
225
|
+
"1. REUSABLE: Does the lesson state a rule an agent can use on another occasion, with the reason it holds? Score 1-2 when it only records what was done, shipped, decided, found or is pending, on a date or for one build, machine or project, or how a system is set up now, however it is phrased. Score 4-5 for a rule with its reason.",
|
|
226
|
+
"2. NON-REDUNDANCY: Is the lesson new next to the existing assets shown below? Score 1-2 only when one of them already states the same rule. Assets on other subjects change nothing: score 4-5 when none is shown or none is on the same subject.",
|
|
227
|
+
"3. GROUNDING: Is every statement in the lesson stated by the source or its feedback, in any words? Check each cause, step, number, rule and limit in the lesson against them. Score 4-5 when each is stated. Score 3 when one stretches what the source says. Score 1-2 when any is in neither, when the lesson drops a limit the source states (one place checked, not confirmed, a guess) and says more than it, or when it is about another subject than the source.",
|
|
228
228
|
"",
|
|
229
|
-
"Source
|
|
229
|
+
"Source memory:",
|
|
230
230
|
"```",
|
|
231
231
|
// The window distill generates from (buildDistillPrompt): grounding can reject, so the judge reads all of it.
|
|
232
232
|
sourceContent.slice(0, 3000),
|
|
233
233
|
"```",
|
|
234
234
|
];
|
|
235
|
-
if (
|
|
236
|
-
lines.push("", "
|
|
237
|
-
|
|
238
|
-
|
|
235
|
+
if (feedback && feedback.length > 0) {
|
|
236
|
+
lines.push("", "Feedback recorded about the memory (the writer saw it too):", "```", feedback.join("\n").slice(0, 1500), "```");
|
|
237
|
+
}
|
|
238
|
+
if (related && related.length > 0) {
|
|
239
|
+
lines.push("", "Existing assets nearest the new lesson (they may be on another subject):");
|
|
240
|
+
for (const asset of related)
|
|
241
|
+
lines.push(`\nExisting asset ref: ${asset.ref}`, "```", asset.content.slice(0, 600), "```");
|
|
239
242
|
}
|
|
240
|
-
lines.push("", "Proposed lesson
|
|
243
|
+
lines.push("", "Proposed lesson:", "```", lessonContent.slice(0, 2000), "```", "", 'Return ONLY valid JSON, no prose: {"scores": {"reusable": <1-5 integer>, "nonRedundancy": <1-5 integer>, "grounding": <1-5 integer>}, "reason": "<one sentence naming the weakest criterion>"}');
|
|
241
244
|
return lines.join("\n");
|
|
242
245
|
}
|
|
243
246
|
function boundedDocument(content, maxChars = 6000) {
|
|
@@ -317,23 +320,19 @@ export function buildReflectJudgePrompt(candidateContent, sourceContent, feedbac
|
|
|
317
320
|
}
|
|
318
321
|
/**
|
|
319
322
|
* `grounding` is scored with the other lesson criteria but left out of their
|
|
320
|
-
* mean
|
|
321
|
-
*
|
|
322
|
-
* a
|
|
323
|
-
* of
|
|
324
|
-
*
|
|
325
|
-
*
|
|
326
|
-
*
|
|
327
|
-
*
|
|
328
|
-
*
|
|
329
|
-
* the lesson, and the judge is never shown it. A contradiction of the source is
|
|
330
|
-
* the optional fidelity check's to send to a human (`judgeAndQueue` in
|
|
331
|
-
* distill.ts), so the rubric must not pre-empt it.
|
|
323
|
+
* mean. A lesson criterion scored {@link LESSON_REJECT_MAX_SCORE} or less is a
|
|
324
|
+
* rejection whatever the mean says: the mean would hide it (4 and 1 average
|
|
325
|
+
* 2.5, a review), and a reviewer was reading every lesson that was not rejected,
|
|
326
|
+
* 17 of 19 of them bad on 2026-10-05. The rubric reserves 1-2 for a lesson that
|
|
327
|
+
* records what was done instead of a rule, repeats an asset the library holds,
|
|
328
|
+
* or states what neither its source nor its feedback does. The judge is shown
|
|
329
|
+
* the feedback the writer saw, so a statement it supports is not an invention. A
|
|
330
|
+
* contradiction of the source is the optional fidelity check's to send to a
|
|
331
|
+
* human (`judgeAndQueue` in distill.ts).
|
|
332
332
|
*/
|
|
333
333
|
const GROUNDING_CRITERION = "grounding";
|
|
334
|
-
const
|
|
335
|
-
const
|
|
336
|
-
const LESSON_JUDGE_CRITERIA = ["novelty", "nonRedundancy", GROUNDING_CRITERION];
|
|
334
|
+
const LESSON_REJECT_MAX_SCORE = 2;
|
|
335
|
+
const LESSON_JUDGE_CRITERIA = ["reusable", "nonRedundancy", GROUNDING_CRITERION];
|
|
337
336
|
const REFLECT_JUDGE_CRITERIA = ["need", "preservation", "quality"];
|
|
338
337
|
/**
|
|
339
338
|
* Read a judge response: the per-criterion shape (averaged here, `grounding`
|
|
@@ -390,9 +389,8 @@ export function judgeResponseSchema(keys) {
|
|
|
390
389
|
* The quality judge. Fails closed: no runner, an unparseable verdict or a
|
|
391
390
|
* provider failure never passes content. Bands: every criterion in the mean
|
|
392
391
|
* >= 4 passes, otherwise a mean >= 2.5 is review and a lower one reject; a
|
|
393
|
-
* `grounding`
|
|
394
|
-
*
|
|
395
|
-
* that would pass to review (a mean that rejects stays a rejection).
|
|
392
|
+
* lesson criterion (`grounding` included) of {@link LESSON_REJECT_MAX_SCORE}
|
|
393
|
+
* or less rejects whatever the mean is.
|
|
396
394
|
* Temperature is set to 0, which reduces run-to-run variation but does not
|
|
397
395
|
* remove it: on some servers (llama.cpp batching, for one) the same request can
|
|
398
396
|
* score a point apart, so the routing rules are chosen with that margin in mind.
|
|
@@ -430,34 +428,18 @@ async function runQualityJudge(feature, config, prompt, keys, chat, options) {
|
|
|
430
428
|
if (!parsed)
|
|
431
429
|
return { pass: false, score: -1, reason: "judge parse failed — routed to review", reviewNeeded: true };
|
|
432
430
|
const { score, lowest, reason, criteria } = parsed;
|
|
433
|
-
|
|
434
|
-
if (criteria &&
|
|
435
|
-
|
|
436
|
-
|
|
437
|
-
score,
|
|
438
|
-
reason: `Off-subject for its source (grounding ${grounding}/5): ${reason}`,
|
|
439
|
-
criteria,
|
|
440
|
-
};
|
|
431
|
+
// A lesson criterion at 2 or below is a defect the mean would hide (4 and 1 average 2.5, a review): it rejects.
|
|
432
|
+
if (criteria && criteria[GROUNDING_CRITERION] !== undefined) {
|
|
433
|
+
const [weakest, low] = Object.entries(criteria).sort((x, y) => x[1] - y[1])[0];
|
|
434
|
+
if (low <= LESSON_REJECT_MAX_SCORE)
|
|
435
|
+
return { pass: false, score, reason: `${weakest} ${low}/5: ${reason}`, criteria };
|
|
441
436
|
}
|
|
442
437
|
const verdict = lowest >= 4 ? { pass: true } : score >= 2.5 ? { pass: false, reviewNeeded: true } : { pass: false };
|
|
443
|
-
// Borderline grounding is a person's call even when the lesson would pass; a mean that rejects stays rejected.
|
|
444
|
-
if (criteria &&
|
|
445
|
-
grounding !== undefined &&
|
|
446
|
-
grounding <= BORDERLINE_GROUNDING_MAX_SCORE &&
|
|
447
|
-
(verdict.pass || verdict.reviewNeeded)) {
|
|
448
|
-
return {
|
|
449
|
-
pass: false,
|
|
450
|
-
reviewNeeded: true,
|
|
451
|
-
score,
|
|
452
|
-
reason: `Borderline on grounding (${grounding}/5), routed to review: ${reason}`,
|
|
453
|
-
criteria,
|
|
454
|
-
};
|
|
455
|
-
}
|
|
456
438
|
return { ...verdict, score, reason, ...(criteria ? { criteria } : {}) };
|
|
457
439
|
}
|
|
458
440
|
/** Judge a proposed lesson (or knowledge promotion) against its source. */
|
|
459
441
|
export function runLessonQualityJudge(config, lessonContent, sourceContent, chat, options = {}) {
|
|
460
|
-
const prompt = buildJudgePrompt(lessonContent, sourceContent, options.
|
|
442
|
+
const prompt = buildJudgePrompt(lessonContent, sourceContent, options.related, options.feedback);
|
|
461
443
|
return runQualityJudge("lesson_quality_gate", config, prompt, LESSON_JUDGE_CRITERIA, chat, options);
|
|
462
444
|
}
|
|
463
445
|
/** Judge an in-place reflect revision without new-lesson novelty criteria. */
|
|
@@ -28,6 +28,7 @@ import { info, warn } from "../../core/warn.js";
|
|
|
28
28
|
import { DEFAULT_LLM_TIMEOUT_MS } from "../../integrations/agent/config.js";
|
|
29
29
|
import { buildExecution, resolveExecution } from "../../integrations/agent/execution.js";
|
|
30
30
|
import { assertRunnerCredentials, runExecution, } from "../../integrations/agent/runner-dispatch.js";
|
|
31
|
+
import { nearestKnowledgeNotes } from "../improve/consolidate/coverage.js";
|
|
31
32
|
import { errMessage, noticeSet } from "../improve/stage.js";
|
|
32
33
|
import { akmProposalAccept, akmProposalReject } from "./proposal.js";
|
|
33
34
|
import { isRetireProposal, PAIR_PASS_GATE, STALE_TARGET_GATE_REASON } from "./proposal-types.js";
|
|
@@ -57,8 +58,13 @@ function categorizeDrainFailure(message, fallback) {
|
|
|
57
58
|
* rejection, so instead of failing identically every run it is auto-rejected
|
|
58
59
|
* once; the ledger records `failed`, keeping the ref re-proposable.
|
|
59
60
|
*/
|
|
60
|
-
async function acceptProposal(opts, proposal, id, reason, promoteFn, rejectFn) {
|
|
61
|
-
const gateDecision = {
|
|
61
|
+
async function acceptProposal(opts, proposal, id, reason, promoteFn, rejectFn, judgeReason) {
|
|
62
|
+
const gateDecision = {
|
|
63
|
+
outcome: "auto-accepted",
|
|
64
|
+
reason,
|
|
65
|
+
gate: DRAIN_GATE,
|
|
66
|
+
...(judgeReason ? { judgeReason } : {}),
|
|
67
|
+
};
|
|
62
68
|
try {
|
|
63
69
|
if (!opts.dryRun) {
|
|
64
70
|
await promoteFn({
|
|
@@ -102,7 +108,7 @@ async function acceptProposal(opts, proposal, id, reason, promoteFn, rejectFn) {
|
|
|
102
108
|
}
|
|
103
109
|
}
|
|
104
110
|
/** Reject one proposal (nothing in a dry run); the error message on failure. */
|
|
105
|
-
async function rejectProposal(opts, id, reason, gateReason, rejectFn) {
|
|
111
|
+
async function rejectProposal(opts, id, reason, gateReason, rejectFn, judgeReason) {
|
|
106
112
|
if (opts.dryRun)
|
|
107
113
|
return undefined;
|
|
108
114
|
try {
|
|
@@ -110,7 +116,12 @@ async function rejectProposal(opts, id, reason, gateReason, rejectFn) {
|
|
|
110
116
|
stashDir: opts.stashDir,
|
|
111
117
|
id,
|
|
112
118
|
reason,
|
|
113
|
-
gateDecision: {
|
|
119
|
+
gateDecision: {
|
|
120
|
+
outcome: "auto-rejected",
|
|
121
|
+
reason: gateReason,
|
|
122
|
+
gate: DRAIN_GATE,
|
|
123
|
+
...(judgeReason ? { judgeReason } : {}),
|
|
124
|
+
},
|
|
114
125
|
});
|
|
115
126
|
return undefined;
|
|
116
127
|
}
|
|
@@ -118,8 +129,23 @@ async function rejectProposal(opts, id, reason, gateReason, rejectFn) {
|
|
|
118
129
|
return errMessage(err);
|
|
119
130
|
}
|
|
120
131
|
}
|
|
121
|
-
/**
|
|
132
|
+
/**
|
|
133
|
+
* `text` as a fenced block whose fence is longer than any backtick run inside
|
|
134
|
+
* it (the CommonMark rule), so a note holding its own code block is not cut
|
|
135
|
+
* short at the first inner fence.
|
|
136
|
+
*/
|
|
137
|
+
export function fencedBlock(text) {
|
|
138
|
+
let longest = 2;
|
|
139
|
+
for (const run of text.match(/`+/g) ?? [])
|
|
140
|
+
if (run.length > longest)
|
|
141
|
+
longest = run.length;
|
|
142
|
+
const fence = "`".repeat(longest + 1);
|
|
143
|
+
return [fence, text, fence];
|
|
144
|
+
}
|
|
145
|
+
/** The judgment prompt: the proposal, the live asset it would overwrite, same-ref siblings, and for a promotion the nearest knowledge notes. */
|
|
122
146
|
export function buildJudgmentPrompt(proposal, reason, ctx) {
|
|
147
|
+
// A promotion (a memory proposed as a new knowledge note) must also be durable.
|
|
148
|
+
const promotion = proposal.source === "consolidate" && ctx.liveAsset === undefined;
|
|
123
149
|
const sections = [
|
|
124
150
|
"You are adjudicating a pending knowledge-base proposal no quality judge has",
|
|
125
151
|
"passed yet. Decide whether to accept, reject, or defer it.",
|
|
@@ -129,12 +155,10 @@ export function buildJudgmentPrompt(proposal, reason, ctx) {
|
|
|
129
155
|
`Left for judgment because: ${reason === "needs-judgment" ? "no quality judge has passed this content yet" : reason}`,
|
|
130
156
|
"",
|
|
131
157
|
"## Proposed content",
|
|
132
|
-
|
|
133
|
-
proposalContent(proposal),
|
|
134
|
-
"```",
|
|
158
|
+
...fencedBlock(proposalContent(proposal)),
|
|
135
159
|
];
|
|
136
160
|
if (ctx.liveAsset !== undefined) {
|
|
137
|
-
sections.push("", "## Current live asset (would be overwritten on accept)",
|
|
161
|
+
sections.push("", "## Current live asset (would be overwritten on accept)", ...fencedBlock(ctx.liveAsset));
|
|
138
162
|
}
|
|
139
163
|
else {
|
|
140
164
|
sections.push("", "## Current live asset", "(none — this proposal would create a new asset)");
|
|
@@ -142,10 +166,19 @@ export function buildJudgmentPrompt(proposal, reason, ctx) {
|
|
|
142
166
|
if (ctx.siblings.length > 0) {
|
|
143
167
|
sections.push("", "## Other pending proposals for the same ref (dedup context)");
|
|
144
168
|
for (const sib of ctx.siblings) {
|
|
145
|
-
sections.push("", `### Sibling ${sib.id} (source: ${sib.source})`,
|
|
169
|
+
sections.push("", `### Sibling ${sib.id} (source: ${sib.source})`, ...fencedBlock(proposalContent(sib)));
|
|
170
|
+
}
|
|
171
|
+
}
|
|
172
|
+
if (ctx.neighbours && ctx.neighbours.length > 0) {
|
|
173
|
+
sections.push("", "## Existing knowledge notes nearest to this promotion's source memory");
|
|
174
|
+
for (const note of ctx.neighbours) {
|
|
175
|
+
sections.push("", `### ${note.ref}`, note.description, ...fencedBlock(note.excerpt));
|
|
146
176
|
}
|
|
177
|
+
sections.push("", "Reject the promotion if these notes already cover what it says, even in other words.");
|
|
147
178
|
}
|
|
148
|
-
sections.push("", "## Your task", 'Return ONLY a JSON object: {"decision": "accept" | "reject" | "defer", "reason": "<short reason>"}.', "- accept: the proposed content is a correct, valuable update worth committing.",
|
|
179
|
+
sections.push("", "## Your task", 'Return ONLY a JSON object: {"decision": "accept" | "reject" | "defer", "reason": "<short reason>"}.', "- accept: the proposed content is a correct, valuable update worth committing.", promotion
|
|
180
|
+
? '- reject: the proposal is wrong, a duplicate, contradicts the live asset, or is not durable: it reports the state of something that changes (a status, rollout, branch, commit, version, test count, "as of <date>"), is a plan not yet carried out, or describes something already retired, replaced or superseded. A lesson drawn from an incident is durable.'
|
|
181
|
+
: "- reject: the proposal is wrong, a duplicate, or contradicts the live asset.", "- defer: you cannot decide from the provided context (leave it pending).", "Output the JSON object and nothing else.");
|
|
149
182
|
return sections.join("\n");
|
|
150
183
|
}
|
|
151
184
|
/** A verdict from raw runner output (the first JSON object), or null. */
|
|
@@ -205,7 +238,7 @@ async function dispatchJudgment(runner, prompt, seams) {
|
|
|
205
238
|
* (under `applyMode` and the remaining accept budget) or the reject. A defer, an
|
|
206
239
|
* unparseable verdict or a runner error leaves the item undecided.
|
|
207
240
|
*/
|
|
208
|
-
async function runJudgmentTier(opts, result, pending, acceptBudget, promoteFn, rejectFn, seams) {
|
|
241
|
+
async function runJudgmentTier(opts, result, pending, acceptBudget, promoteFn, rejectFn, seams, deferNotes) {
|
|
209
242
|
const byId = new Map(pending.map((p) => [p.id, p]));
|
|
210
243
|
const notices = noticeSet();
|
|
211
244
|
const stillDeferred = [];
|
|
@@ -216,21 +249,33 @@ async function runJudgmentTier(opts, result, pending, acceptBudget, promoteFn, r
|
|
|
216
249
|
stillDeferred.push(item);
|
|
217
250
|
continue;
|
|
218
251
|
}
|
|
252
|
+
const liveAsset = readLiveAssetContent(opts.stashDir, proposal.ref);
|
|
219
253
|
const prompt = buildJudgmentPrompt(proposal, item.reason, {
|
|
220
|
-
liveAsset
|
|
254
|
+
liveAsset,
|
|
221
255
|
siblings: pending.filter((p) => p.ref === proposal.ref && p.id !== proposal.id),
|
|
256
|
+
// A create the model otherwise judges blind: the model never sees knowledge/.
|
|
257
|
+
...(liveAsset === undefined ? { neighbours: promotionNeighbours(opts.stashDir, proposal) } : {}),
|
|
222
258
|
});
|
|
223
259
|
const dispatch = await dispatchJudgment(opts.judgment, prompt, seams);
|
|
224
260
|
notices.add(dispatch.notices);
|
|
225
261
|
if (dispatch.error)
|
|
226
262
|
warn(`[triage] judgment dispatch failed for ${item.id}: ${dispatch.error}`);
|
|
227
263
|
const verdict = dispatch.error ? null : dispatch.verdict;
|
|
228
|
-
if (!verdict
|
|
264
|
+
if (!verdict) {
|
|
265
|
+
deferNotes.set(item.id, { reason: dispatch.error ? "judgment-error" : "judgment-parse-failure" });
|
|
266
|
+
stillDeferred.push(item);
|
|
267
|
+
continue;
|
|
268
|
+
}
|
|
269
|
+
if (verdict.decision === "defer") {
|
|
270
|
+
deferNotes.set(item.id, {
|
|
271
|
+
reason: "judgment-deferred",
|
|
272
|
+
...(verdict.reason ? { judgeReason: verdict.reason } : {}),
|
|
273
|
+
});
|
|
229
274
|
stillDeferred.push(item);
|
|
230
275
|
continue;
|
|
231
276
|
}
|
|
232
277
|
if (verdict.decision === "reject") {
|
|
233
|
-
const failure = await rejectProposal(opts, item.id, verdict.reason || "judgment: reject", "judgment-reject", rejectFn);
|
|
278
|
+
const failure = await rejectProposal(opts, item.id, verdict.reason || "judgment: reject", "judgment-reject", rejectFn, verdict.reason);
|
|
234
279
|
if (failure === undefined) {
|
|
235
280
|
result.rejected.push(item.id);
|
|
236
281
|
}
|
|
@@ -252,6 +297,7 @@ async function runJudgmentTier(opts, result, pending, acceptBudget, promoteFn, r
|
|
|
252
297
|
reason: "judgment-accept",
|
|
253
298
|
contentHash: proposalContentHash(proposal),
|
|
254
299
|
gate: DRAIN_GATE,
|
|
300
|
+
...(verdict.reason ? { judgeReason: verdict.reason } : {}),
|
|
255
301
|
});
|
|
256
302
|
result.staged.push(item.id);
|
|
257
303
|
}
|
|
@@ -265,7 +311,7 @@ async function runJudgmentTier(opts, result, pending, acceptBudget, promoteFn, r
|
|
|
265
311
|
result.skippedByCap.push(item.id);
|
|
266
312
|
continue;
|
|
267
313
|
}
|
|
268
|
-
const outcome = await acceptProposal(opts, proposal, item.id, "judgment-accept", promoteFn, rejectFn);
|
|
314
|
+
const outcome = await acceptProposal(opts, proposal, item.id, "judgment-accept", promoteFn, rejectFn, verdict.reason);
|
|
269
315
|
if (outcome === "promoted") {
|
|
270
316
|
result.promoted.push(item.id);
|
|
271
317
|
acceptBudget -= 1;
|
|
@@ -286,6 +332,21 @@ async function runJudgmentTier(opts, result, pending, acceptBudget, promoteFn, r
|
|
|
286
332
|
result.notices = notices.list();
|
|
287
333
|
result.deferred = stillDeferred;
|
|
288
334
|
}
|
|
335
|
+
/** The knowledge notes nearest to a consolidate promotion's source memory; none for any other proposal. */
|
|
336
|
+
function promotionNeighbours(stashDir, proposal) {
|
|
337
|
+
if (proposal.source !== "consolidate" || proposal.promotionSource === undefined)
|
|
338
|
+
return [];
|
|
339
|
+
try {
|
|
340
|
+
const parsed = parseRefInput(proposal.promotionSource);
|
|
341
|
+
const typeDir = stashDirFor(parsed.type);
|
|
342
|
+
if (!typeDir)
|
|
343
|
+
return [];
|
|
344
|
+
return nearestKnowledgeNotes(assetPathForName(parsed.type, path.join(stashDir, typeDir), parsed.name));
|
|
345
|
+
}
|
|
346
|
+
catch {
|
|
347
|
+
return [];
|
|
348
|
+
}
|
|
349
|
+
}
|
|
289
350
|
/** The live asset a proposal would overwrite, if any. */
|
|
290
351
|
function readLiveAssetContent(stashDir, ref) {
|
|
291
352
|
try {
|
|
@@ -314,8 +375,8 @@ export async function drainProposals(opts, promoteFn = akmProposalAccept, reject
|
|
|
314
375
|
const empties = [];
|
|
315
376
|
for (const proposal of pending) {
|
|
316
377
|
// A consolidate pair-pass `retire` proposal is auto-accepted only when the
|
|
317
|
-
// pair
|
|
318
|
-
//
|
|
378
|
+
// pair pass staged it: nothing unique on either side, confirmed by a
|
|
379
|
+
// second look, no continuity risk; every other one waits for a direct
|
|
319
380
|
// `akm proposal accept` (spec §25.6). Checked before isEmptyDiff, which
|
|
320
381
|
// has nothing meaningful to read on a delete-primary change.
|
|
321
382
|
if (isRetireProposal(proposal)) {
|
|
@@ -324,7 +385,7 @@ export async function drainProposals(opts, promoteFn = akmProposalAccept, reject
|
|
|
324
385
|
staged.gate === PAIR_PASS_GATE &&
|
|
325
386
|
staged.contentHash === proposalContentHash(proposal) &&
|
|
326
387
|
!proposal.retirement?.continuityRisk) {
|
|
327
|
-
accepts.push({ id: proposal.id, reason:
|
|
388
|
+
accepts.push({ id: proposal.id, reason: staged.reason });
|
|
328
389
|
}
|
|
329
390
|
continue;
|
|
330
391
|
}
|
|
@@ -394,15 +455,22 @@ export async function drainProposals(opts, promoteFn = akmProposalAccept, reject
|
|
|
394
455
|
}
|
|
395
456
|
}
|
|
396
457
|
}
|
|
458
|
+
const deferNotes = new Map();
|
|
397
459
|
if (opts.judgment && result.deferred.length > 0) {
|
|
398
|
-
await runJudgmentTier({ ...opts, judgment: opts.judgment }, result, pending, cap - promotedHere, promoteFn, rejectFn, judgmentSeams);
|
|
460
|
+
await runJudgmentTier({ ...opts, judgment: opts.judgment }, result, pending, cap - promotedHere, promoteFn, rejectFn, judgmentSeams, deferNotes);
|
|
399
461
|
}
|
|
400
462
|
// #577: whatever stays undecided is left for review (`review_needed` in the ledger).
|
|
401
463
|
if (!opts.dryRun) {
|
|
402
|
-
const reviewReason = opts.judgment ? "judgment-deferred" : "no-judge-configured";
|
|
403
464
|
for (const item of result.deferred) {
|
|
465
|
+
const note = deferNotes.get(item.id);
|
|
466
|
+
const reviewReason = note?.reason ?? (opts.judgment ? "judgment-deferred" : "no-judge-configured");
|
|
404
467
|
try {
|
|
405
|
-
recordGateDecision(opts.stashDir, item.id, {
|
|
468
|
+
recordGateDecision(opts.stashDir, item.id, {
|
|
469
|
+
outcome: "deferred",
|
|
470
|
+
reason: reviewReason,
|
|
471
|
+
gate: DRAIN_GATE,
|
|
472
|
+
...(note?.judgeReason ? { judgeReason: note.judgeReason } : {}),
|
|
473
|
+
});
|
|
406
474
|
}
|
|
407
475
|
catch (err) {
|
|
408
476
|
warn(`[triage] failed to record gate decision for ${item.id}: ${errMessage(err)}`);
|
|
@@ -53,7 +53,7 @@ export function isRetireProposal(proposal) {
|
|
|
53
53
|
}
|
|
54
54
|
/** A promote refused because the target changed after mint (STALE, R20) — not a merit judgement. */
|
|
55
55
|
export const STALE_TARGET_GATE_REASON = "stale-target";
|
|
56
|
-
/** The gate on a retire proposal the triage drain may accept unattended: a pair-judged
|
|
56
|
+
/** The gate on a retire proposal the triage drain may accept unattended: a staged pair-judged retirement. */
|
|
57
57
|
export const PAIR_PASS_GATE = "consolidate-pair";
|
|
58
58
|
export const EXPIRED_GATE_REASON = "expired";
|
|
59
59
|
export const ASSET_MISSING_GATE_REASON = "asset-missing";
|
|
@@ -78,9 +78,9 @@ const qualityGateField = z
|
|
|
78
78
|
.optional();
|
|
79
79
|
/**
|
|
80
80
|
* WS-3b: CLS (Complementary Learning System) interleaving (step 9).
|
|
81
|
-
* distill
|
|
82
|
-
*
|
|
83
|
-
* Default
|
|
81
|
+
* The distill prompt includes the lessons, knowledge notes and skills the library already holds near the
|
|
82
|
+
* memory, so the writer answers NONE for a rule one of them states and does not overwrite a prior
|
|
83
|
+
* generalization. Default ON; `enabled: false` turns it off. Only meaningful on the `distill` process.
|
|
84
84
|
*/
|
|
85
85
|
const clsField = z
|
|
86
86
|
.object({
|
package/dist/core/paths.js
CHANGED
|
@@ -332,15 +332,6 @@ export function getStashStateKey(stashDir) {
|
|
|
332
332
|
function stashScopedDir(base, stashDir) {
|
|
333
333
|
return path.join(base, getStashStateKey(stashDir));
|
|
334
334
|
}
|
|
335
|
-
/**
|
|
336
|
-
* `$STATE/improve/measurement/verdicts/<stash>/` — `akm-eval-proactive-verdict`
|
|
337
|
-
* reports. Moved out of `$STASH/.akm/measurement/verdicts/` (itlackey/akm#890);
|
|
338
|
-
* the pilot treatment file at `$STASH/.akm/measurement/` is manually-authored
|
|
339
|
-
* measurement input and stays put.
|
|
340
|
-
*/
|
|
341
|
-
export function getMeasurementVerdictsDir(stashDir) {
|
|
342
|
-
return stashScopedDir(path.join(getStateDir(), "improve", "measurement", "verdicts"), stashDir);
|
|
343
|
-
}
|
|
344
335
|
/**
|
|
345
336
|
* `$CACHE/index/unresolved-sources/<stash>/` — synthetic placeholder path for
|
|
346
337
|
* a configured source whose content root did not resolve this run. Never
|
|
@@ -50,11 +50,14 @@ export const CONSOLIDATE_LEDGER_SOURCE = "consolidate";
|
|
|
50
50
|
* promotion is a verdict on that memory's text; asking the model about the
|
|
51
51
|
* same text again can only reproduce the proposal (accepted used to be
|
|
52
52
|
* eligible at once, rejected after 7 days), so the memory waits for an edit.
|
|
53
|
+
* So does a memory the model judged and left alone (`judged_no_action`):
|
|
54
|
+
* the same text would be judged weekly with the same answer.
|
|
53
55
|
* The clock stays for a row with no recorded hash — one decided before the
|
|
54
56
|
* hash was recorded — see {@link nextEligibleAt} and {@link isContentDrivenRow}.
|
|
55
57
|
*/
|
|
56
58
|
export function isContentDrivenDecision(source, outcome) {
|
|
57
|
-
return source === CONSOLIDATE_LEDGER_SOURCE &&
|
|
59
|
+
return (source === CONSOLIDATE_LEDGER_SOURCE &&
|
|
60
|
+
(outcome === "accepted" || outcome === "rejected" || outcome === "judged_no_action"));
|
|
58
61
|
}
|
|
59
62
|
/**
|
|
60
63
|
* Whether this row is held by its content hash: a decided consolidate
|