akm-cli 0.9.26 → 0.9.27-alpha.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +237 -0
- package/LICENSE +3 -4
- package/dist/assets/prompts/consolidate-system.md +2 -2
- package/dist/assets/prompts/distill-lesson-system.md +29 -7
- package/dist/assets/prompts/extract-session.md +2 -2
- package/dist/commands/improve/consolidate/coverage.js +71 -17
- package/dist/commands/improve/consolidate/pair-pass.js +12 -9
- package/dist/commands/improve/consolidate.js +77 -5
- package/dist/commands/improve/distill-guards.js +9 -10
- package/dist/commands/improve/distill.js +70 -22
- package/dist/commands/improve/extract-prompt.js +61 -39
- package/dist/commands/improve/extract.js +2 -1
- package/dist/commands/improve/preparation.js +14 -1
- package/dist/commands/improve/reflect.js +36 -4
- package/dist/commands/improve/retrieval-gate.js +1 -1
- package/dist/commands/improve/session-asset.js +3 -2
- package/dist/commands/improve/stage.js +35 -53
- package/dist/commands/proposal/drain.js +85 -21
- package/dist/commands/proposal/proposal-types.js +1 -1
- package/dist/core/config/schema/improve-processes.js +3 -3
- package/dist/core/paths.js +0 -9
- package/dist/indexer/indexer.js +35 -19
- package/dist/integrations/harnesses/codex/agent-builder.js +23 -15
- package/dist/llm/client.js +44 -14
- package/dist/llm/memory-infer.js +1 -1
- package/dist/scripts/akm-migrate-node.js +25 -20
- package/dist/scripts/akm-migrate.js +25 -20
- package/dist/storage/repositories/improve-ledger-repository.js +4 -1
- package/dist/storage/repositories/index-entry-schema.js +20 -4
- package/dist/storage/repositories/index-fts-repository.js +44 -3
- package/dist/storage/repositories/index-schema.js +14 -8
- package/docs/README.md +1 -2
- package/docs/integration/bundling-akm.md +1 -1
- package/docs/migration/v0.8-to-v0.9.md +3 -1
- package/docs/reference/README.md +1 -1
- package/docs/reference/cli.md +8 -4
- package/docs/reference/configuration.md +5 -2
- package/docs/reference/data-and-telemetry.md +1 -1
- package/package.json +1 -1
|
@@ -26,7 +26,7 @@ const MAX_QUERIES = 5;
|
|
|
26
26
|
const MAX_QUERY_CHARS = 2000;
|
|
27
27
|
/** The judge sees this much of the body, as in the retrieval eval. */
|
|
28
28
|
const MAX_DOC_CHARS = 1500;
|
|
29
|
-
const GRADE_SCHEMA = {
|
|
29
|
+
export const GRADE_SCHEMA = {
|
|
30
30
|
type: "object",
|
|
31
31
|
required: ["grade", "reason"],
|
|
32
32
|
additionalProperties: false,
|
|
@@ -14,9 +14,10 @@ import { assembleAsset } from "../../core/asset/asset-serialize.js";
|
|
|
14
14
|
import { conceptIdFromTypeName } from "../../core/asset/resolve-ref.js";
|
|
15
15
|
import { parseEmbeddedJsonResponse } from "../../core/parse.js";
|
|
16
16
|
import { recordWrittenPath } from "../../core/write-provenance.js";
|
|
17
|
+
// Every property is required, as a strict structured-output provider needs; an empty `tags` array is none.
|
|
17
18
|
export const SESSION_SUMMARY_JSON_SCHEMA = {
|
|
18
19
|
type: "object",
|
|
19
|
-
required: ["summary", "key_topics"],
|
|
20
|
+
required: ["summary", "key_topics", "tags"],
|
|
20
21
|
additionalProperties: false,
|
|
21
22
|
properties: {
|
|
22
23
|
summary: { type: "string" },
|
|
@@ -61,7 +62,7 @@ export function buildSessionSummaryPrompt(data) {
|
|
|
61
62
|
"Transcript:",
|
|
62
63
|
renderTranscriptForSummary(data.events),
|
|
63
64
|
"",
|
|
64
|
-
'Respond as JSON: {"summary": string, "key_topics": string[], "tags"
|
|
65
|
+
'Respond as JSON: {"summary": string, "key_topics": string[], "tags": string[]} (an empty array for no tags).',
|
|
65
66
|
].join("\n");
|
|
66
67
|
}
|
|
67
68
|
/** The summary JSON, tolerating prose around it; `undefined` when nothing usable parses. */
|
|
@@ -216,28 +216,31 @@ export function resolveQualityGateJudge(config, profile, processName, onNotices)
|
|
|
216
216
|
onNotices?.(resolved.notices);
|
|
217
217
|
return resolved.runner;
|
|
218
218
|
}
|
|
219
|
-
/** Lesson judge prompt
|
|
220
|
-
export function buildJudgePrompt(lessonContent, sourceContent,
|
|
219
|
+
/** Lesson judge prompt: what the writer was given (the source and its feedback), the assets nearest the new lesson, the lesson. */
|
|
220
|
+
export function buildJudgePrompt(lessonContent, sourceContent, related, feedback) {
|
|
221
221
|
const lines = [
|
|
222
|
-
"You are evaluating a
|
|
222
|
+
"You are evaluating a lesson an agent wrote from a memory and the feedback about it, for an akm knowledge base.",
|
|
223
223
|
"",
|
|
224
224
|
"Score this lesson on each criterion from 1 (poor) to 5 (excellent):",
|
|
225
|
-
"1.
|
|
226
|
-
"2. NON-REDUNDANCY: Is
|
|
227
|
-
"3. GROUNDING: Is the lesson
|
|
225
|
+
"1. REUSABLE: Does the lesson state a rule an agent can use on another occasion, with the reason it holds? Score 1-2 when it only records what was done, shipped, decided, found or is pending, on a date or for one build, machine or project, or how a system is set up now, however it is phrased. Score 4-5 for a rule with its reason.",
|
|
226
|
+
"2. NON-REDUNDANCY: Is the lesson new next to the existing assets shown below? Score 1-2 only when one of them already states the same rule. Assets on other subjects change nothing: score 4-5 when none is shown or none is on the same subject.",
|
|
227
|
+
"3. GROUNDING: Is every statement in the lesson stated by the source or its feedback, in any words? Check each cause, step, number, rule and limit in the lesson against them. Score 4-5 when each is stated. Score 3 when one stretches what the source says. Score 1-2 when any is in neither, when the lesson drops a limit the source states (one place checked, not confirmed, a guess) and says more than it, or when it is about another subject than the source.",
|
|
228
228
|
"",
|
|
229
|
-
"Source
|
|
229
|
+
"Source memory:",
|
|
230
230
|
"```",
|
|
231
231
|
// The window distill generates from (buildDistillPrompt): grounding can reject, so the judge reads all of it.
|
|
232
232
|
sourceContent.slice(0, 3000),
|
|
233
233
|
"```",
|
|
234
234
|
];
|
|
235
|
-
if (
|
|
236
|
-
lines.push("", "
|
|
237
|
-
|
|
238
|
-
|
|
235
|
+
if (feedback && feedback.length > 0) {
|
|
236
|
+
lines.push("", "Feedback recorded about the memory (the writer saw it too):", "```", feedback.join("\n").slice(0, 1500), "```");
|
|
237
|
+
}
|
|
238
|
+
if (related && related.length > 0) {
|
|
239
|
+
lines.push("", "Existing assets nearest the new lesson (they may be on another subject):");
|
|
240
|
+
for (const asset of related)
|
|
241
|
+
lines.push(`\nExisting asset ref: ${asset.ref}`, "```", asset.content.slice(0, 600), "```");
|
|
239
242
|
}
|
|
240
|
-
lines.push("", "Proposed lesson
|
|
243
|
+
lines.push("", "Proposed lesson:", "```", lessonContent.slice(0, 2000), "```", "", 'Return ONLY valid JSON, no prose: {"scores": {"reusable": <1-5 integer>, "nonRedundancy": <1-5 integer>, "grounding": <1-5 integer>}, "reason": "<one sentence naming the weakest criterion>"}');
|
|
241
244
|
return lines.join("\n");
|
|
242
245
|
}
|
|
243
246
|
function boundedDocument(content, maxChars = 6000) {
|
|
@@ -317,23 +320,19 @@ export function buildReflectJudgePrompt(candidateContent, sourceContent, feedbac
|
|
|
317
320
|
}
|
|
318
321
|
/**
|
|
319
322
|
* `grounding` is scored with the other lesson criteria but left out of their
|
|
320
|
-
* mean
|
|
321
|
-
*
|
|
322
|
-
* a
|
|
323
|
-
* of
|
|
324
|
-
*
|
|
325
|
-
*
|
|
326
|
-
*
|
|
327
|
-
*
|
|
328
|
-
*
|
|
329
|
-
* the lesson, and the judge is never shown it. A contradiction of the source is
|
|
330
|
-
* the optional fidelity check's to send to a human (`judgeAndQueue` in
|
|
331
|
-
* distill.ts), so the rubric must not pre-empt it.
|
|
323
|
+
* mean. A lesson criterion scored {@link LESSON_REJECT_MAX_SCORE} or less is a
|
|
324
|
+
* rejection whatever the mean says: the mean would hide it (4 and 1 average
|
|
325
|
+
* 2.5, a review), and a reviewer was reading every lesson that was not rejected,
|
|
326
|
+
* 17 of 19 of them bad on 2026-10-05. The rubric reserves 1-2 for a lesson that
|
|
327
|
+
* records what was done instead of a rule, repeats an asset the library holds,
|
|
328
|
+
* or states what neither its source nor its feedback does. The judge is shown
|
|
329
|
+
* the feedback the writer saw, so a statement it supports is not an invention. A
|
|
330
|
+
* contradiction of the source is the optional fidelity check's to send to a
|
|
331
|
+
* human (`judgeAndQueue` in distill.ts).
|
|
332
332
|
*/
|
|
333
333
|
const GROUNDING_CRITERION = "grounding";
|
|
334
|
-
const
|
|
335
|
-
const
|
|
336
|
-
const LESSON_JUDGE_CRITERIA = ["novelty", "nonRedundancy", GROUNDING_CRITERION];
|
|
334
|
+
const LESSON_REJECT_MAX_SCORE = 2;
|
|
335
|
+
const LESSON_JUDGE_CRITERIA = ["reusable", "nonRedundancy", GROUNDING_CRITERION];
|
|
337
336
|
const REFLECT_JUDGE_CRITERIA = ["need", "preservation", "quality"];
|
|
338
337
|
/**
|
|
339
338
|
* Read a judge response: the per-criterion shape (averaged here, `grounding`
|
|
@@ -370,7 +369,7 @@ function parseJudgeResponse(raw, keys) {
|
|
|
370
369
|
}
|
|
371
370
|
return inRange(parsed.score) ? { score: parsed.score, lowest: parsed.score, reason } : undefined;
|
|
372
371
|
}
|
|
373
|
-
function judgeResponseSchema(keys) {
|
|
372
|
+
export function judgeResponseSchema(keys) {
|
|
374
373
|
return {
|
|
375
374
|
type: "object",
|
|
376
375
|
required: ["scores", "reason"],
|
|
@@ -390,9 +389,8 @@ function judgeResponseSchema(keys) {
|
|
|
390
389
|
* The quality judge. Fails closed: no runner, an unparseable verdict or a
|
|
391
390
|
* provider failure never passes content. Bands: every criterion in the mean
|
|
392
391
|
* >= 4 passes, otherwise a mean >= 2.5 is review and a lower one reject; a
|
|
393
|
-
* `grounding`
|
|
394
|
-
*
|
|
395
|
-
* that would pass to review (a mean that rejects stays a rejection).
|
|
392
|
+
* lesson criterion (`grounding` included) of {@link LESSON_REJECT_MAX_SCORE}
|
|
393
|
+
* or less rejects whatever the mean is.
|
|
396
394
|
* Temperature is set to 0, which reduces run-to-run variation but does not
|
|
397
395
|
* remove it: on some servers (llama.cpp batching, for one) the same request can
|
|
398
396
|
* score a point apart, so the routing rules are chosen with that margin in mind.
|
|
@@ -430,34 +428,18 @@ async function runQualityJudge(feature, config, prompt, keys, chat, options) {
|
|
|
430
428
|
if (!parsed)
|
|
431
429
|
return { pass: false, score: -1, reason: "judge parse failed — routed to review", reviewNeeded: true };
|
|
432
430
|
const { score, lowest, reason, criteria } = parsed;
|
|
433
|
-
|
|
434
|
-
if (criteria &&
|
|
435
|
-
|
|
436
|
-
|
|
437
|
-
score,
|
|
438
|
-
reason: `Off-subject for its source (grounding ${grounding}/5): ${reason}`,
|
|
439
|
-
criteria,
|
|
440
|
-
};
|
|
431
|
+
// A lesson criterion at 2 or below is a defect the mean would hide (4 and 1 average 2.5, a review): it rejects.
|
|
432
|
+
if (criteria && criteria[GROUNDING_CRITERION] !== undefined) {
|
|
433
|
+
const [weakest, low] = Object.entries(criteria).sort((x, y) => x[1] - y[1])[0];
|
|
434
|
+
if (low <= LESSON_REJECT_MAX_SCORE)
|
|
435
|
+
return { pass: false, score, reason: `${weakest} ${low}/5: ${reason}`, criteria };
|
|
441
436
|
}
|
|
442
437
|
const verdict = lowest >= 4 ? { pass: true } : score >= 2.5 ? { pass: false, reviewNeeded: true } : { pass: false };
|
|
443
|
-
// Borderline grounding is a person's call even when the lesson would pass; a mean that rejects stays rejected.
|
|
444
|
-
if (criteria &&
|
|
445
|
-
grounding !== undefined &&
|
|
446
|
-
grounding <= BORDERLINE_GROUNDING_MAX_SCORE &&
|
|
447
|
-
(verdict.pass || verdict.reviewNeeded)) {
|
|
448
|
-
return {
|
|
449
|
-
pass: false,
|
|
450
|
-
reviewNeeded: true,
|
|
451
|
-
score,
|
|
452
|
-
reason: `Borderline on grounding (${grounding}/5), routed to review: ${reason}`,
|
|
453
|
-
criteria,
|
|
454
|
-
};
|
|
455
|
-
}
|
|
456
438
|
return { ...verdict, score, reason, ...(criteria ? { criteria } : {}) };
|
|
457
439
|
}
|
|
458
440
|
/** Judge a proposed lesson (or knowledge promotion) against its source. */
|
|
459
441
|
export function runLessonQualityJudge(config, lessonContent, sourceContent, chat, options = {}) {
|
|
460
|
-
const prompt = buildJudgePrompt(lessonContent, sourceContent, options.
|
|
442
|
+
const prompt = buildJudgePrompt(lessonContent, sourceContent, options.related, options.feedback);
|
|
461
443
|
return runQualityJudge("lesson_quality_gate", config, prompt, LESSON_JUDGE_CRITERIA, chat, options);
|
|
462
444
|
}
|
|
463
445
|
/** Judge an in-place reflect revision without new-lesson novelty criteria. */
|
|
@@ -28,6 +28,7 @@ import { info, warn } from "../../core/warn.js";
|
|
|
28
28
|
import { DEFAULT_LLM_TIMEOUT_MS } from "../../integrations/agent/config.js";
|
|
29
29
|
import { buildExecution, resolveExecution } from "../../integrations/agent/execution.js";
|
|
30
30
|
import { assertRunnerCredentials, runExecution, } from "../../integrations/agent/runner-dispatch.js";
|
|
31
|
+
import { nearestKnowledgeNotes } from "../improve/consolidate/coverage.js";
|
|
31
32
|
import { errMessage, noticeSet } from "../improve/stage.js";
|
|
32
33
|
import { akmProposalAccept, akmProposalReject } from "./proposal.js";
|
|
33
34
|
import { isRetireProposal, PAIR_PASS_GATE, STALE_TARGET_GATE_REASON } from "./proposal-types.js";
|
|
@@ -57,8 +58,13 @@ function categorizeDrainFailure(message, fallback) {
|
|
|
57
58
|
* rejection, so instead of failing identically every run it is auto-rejected
|
|
58
59
|
* once; the ledger records `failed`, keeping the ref re-proposable.
|
|
59
60
|
*/
|
|
60
|
-
async function acceptProposal(opts, proposal, id, reason, promoteFn, rejectFn) {
|
|
61
|
-
const gateDecision = {
|
|
61
|
+
async function acceptProposal(opts, proposal, id, reason, promoteFn, rejectFn, judgeReason) {
|
|
62
|
+
const gateDecision = {
|
|
63
|
+
outcome: "auto-accepted",
|
|
64
|
+
reason,
|
|
65
|
+
gate: DRAIN_GATE,
|
|
66
|
+
...(judgeReason ? { judgeReason } : {}),
|
|
67
|
+
};
|
|
62
68
|
try {
|
|
63
69
|
if (!opts.dryRun) {
|
|
64
70
|
await promoteFn({
|
|
@@ -102,7 +108,7 @@ async function acceptProposal(opts, proposal, id, reason, promoteFn, rejectFn) {
|
|
|
102
108
|
}
|
|
103
109
|
}
|
|
104
110
|
/** Reject one proposal (nothing in a dry run); the error message on failure. */
|
|
105
|
-
async function rejectProposal(opts, id, reason, gateReason, rejectFn) {
|
|
111
|
+
async function rejectProposal(opts, id, reason, gateReason, rejectFn, judgeReason) {
|
|
106
112
|
if (opts.dryRun)
|
|
107
113
|
return undefined;
|
|
108
114
|
try {
|
|
@@ -110,7 +116,12 @@ async function rejectProposal(opts, id, reason, gateReason, rejectFn) {
|
|
|
110
116
|
stashDir: opts.stashDir,
|
|
111
117
|
id,
|
|
112
118
|
reason,
|
|
113
|
-
gateDecision: {
|
|
119
|
+
gateDecision: {
|
|
120
|
+
outcome: "auto-rejected",
|
|
121
|
+
reason: gateReason,
|
|
122
|
+
gate: DRAIN_GATE,
|
|
123
|
+
...(judgeReason ? { judgeReason } : {}),
|
|
124
|
+
},
|
|
114
125
|
});
|
|
115
126
|
return undefined;
|
|
116
127
|
}
|
|
@@ -118,7 +129,20 @@ async function rejectProposal(opts, id, reason, gateReason, rejectFn) {
|
|
|
118
129
|
return errMessage(err);
|
|
119
130
|
}
|
|
120
131
|
}
|
|
121
|
-
/**
|
|
132
|
+
/**
|
|
133
|
+
* `text` as a fenced block whose fence is longer than any backtick run inside
|
|
134
|
+
* it (the CommonMark rule), so a note holding its own code block is not cut
|
|
135
|
+
* short at the first inner fence.
|
|
136
|
+
*/
|
|
137
|
+
export function fencedBlock(text) {
|
|
138
|
+
let longest = 2;
|
|
139
|
+
for (const run of text.match(/`+/g) ?? [])
|
|
140
|
+
if (run.length > longest)
|
|
141
|
+
longest = run.length;
|
|
142
|
+
const fence = "`".repeat(longest + 1);
|
|
143
|
+
return [fence, text, fence];
|
|
144
|
+
}
|
|
145
|
+
/** The judgment prompt: the proposal, the live asset it would overwrite, same-ref siblings, and for a promotion the nearest knowledge notes. */
|
|
122
146
|
export function buildJudgmentPrompt(proposal, reason, ctx) {
|
|
123
147
|
const sections = [
|
|
124
148
|
"You are adjudicating a pending knowledge-base proposal no quality judge has",
|
|
@@ -129,12 +153,10 @@ export function buildJudgmentPrompt(proposal, reason, ctx) {
|
|
|
129
153
|
`Left for judgment because: ${reason === "needs-judgment" ? "no quality judge has passed this content yet" : reason}`,
|
|
130
154
|
"",
|
|
131
155
|
"## Proposed content",
|
|
132
|
-
|
|
133
|
-
proposalContent(proposal),
|
|
134
|
-
"```",
|
|
156
|
+
...fencedBlock(proposalContent(proposal)),
|
|
135
157
|
];
|
|
136
158
|
if (ctx.liveAsset !== undefined) {
|
|
137
|
-
sections.push("", "## Current live asset (would be overwritten on accept)",
|
|
159
|
+
sections.push("", "## Current live asset (would be overwritten on accept)", ...fencedBlock(ctx.liveAsset));
|
|
138
160
|
}
|
|
139
161
|
else {
|
|
140
162
|
sections.push("", "## Current live asset", "(none — this proposal would create a new asset)");
|
|
@@ -142,8 +164,15 @@ export function buildJudgmentPrompt(proposal, reason, ctx) {
|
|
|
142
164
|
if (ctx.siblings.length > 0) {
|
|
143
165
|
sections.push("", "## Other pending proposals for the same ref (dedup context)");
|
|
144
166
|
for (const sib of ctx.siblings) {
|
|
145
|
-
sections.push("", `### Sibling ${sib.id} (source: ${sib.source})`,
|
|
167
|
+
sections.push("", `### Sibling ${sib.id} (source: ${sib.source})`, ...fencedBlock(proposalContent(sib)));
|
|
168
|
+
}
|
|
169
|
+
}
|
|
170
|
+
if (ctx.neighbours && ctx.neighbours.length > 0) {
|
|
171
|
+
sections.push("", "## Existing knowledge notes nearest to this promotion's source memory");
|
|
172
|
+
for (const note of ctx.neighbours) {
|
|
173
|
+
sections.push("", `### ${note.ref}`, note.description, ...fencedBlock(note.excerpt));
|
|
146
174
|
}
|
|
175
|
+
sections.push("", "Reject the promotion if these notes already cover what it says, even in other words.");
|
|
147
176
|
}
|
|
148
177
|
sections.push("", "## Your task", 'Return ONLY a JSON object: {"decision": "accept" | "reject" | "defer", "reason": "<short reason>"}.', "- accept: the proposed content is a correct, valuable update worth committing.", "- reject: the proposal is wrong, a duplicate, or contradicts the live asset.", "- defer: you cannot decide from the provided context (leave it pending).", "Output the JSON object and nothing else.");
|
|
149
178
|
return sections.join("\n");
|
|
@@ -205,7 +234,7 @@ async function dispatchJudgment(runner, prompt, seams) {
|
|
|
205
234
|
* (under `applyMode` and the remaining accept budget) or the reject. A defer, an
|
|
206
235
|
* unparseable verdict or a runner error leaves the item undecided.
|
|
207
236
|
*/
|
|
208
|
-
async function runJudgmentTier(opts, result, pending, acceptBudget, promoteFn, rejectFn, seams) {
|
|
237
|
+
async function runJudgmentTier(opts, result, pending, acceptBudget, promoteFn, rejectFn, seams, deferNotes) {
|
|
209
238
|
const byId = new Map(pending.map((p) => [p.id, p]));
|
|
210
239
|
const notices = noticeSet();
|
|
211
240
|
const stillDeferred = [];
|
|
@@ -216,21 +245,33 @@ async function runJudgmentTier(opts, result, pending, acceptBudget, promoteFn, r
|
|
|
216
245
|
stillDeferred.push(item);
|
|
217
246
|
continue;
|
|
218
247
|
}
|
|
248
|
+
const liveAsset = readLiveAssetContent(opts.stashDir, proposal.ref);
|
|
219
249
|
const prompt = buildJudgmentPrompt(proposal, item.reason, {
|
|
220
|
-
liveAsset
|
|
250
|
+
liveAsset,
|
|
221
251
|
siblings: pending.filter((p) => p.ref === proposal.ref && p.id !== proposal.id),
|
|
252
|
+
// A create the model otherwise judges blind: the model never sees knowledge/.
|
|
253
|
+
...(liveAsset === undefined ? { neighbours: promotionNeighbours(opts.stashDir, proposal) } : {}),
|
|
222
254
|
});
|
|
223
255
|
const dispatch = await dispatchJudgment(opts.judgment, prompt, seams);
|
|
224
256
|
notices.add(dispatch.notices);
|
|
225
257
|
if (dispatch.error)
|
|
226
258
|
warn(`[triage] judgment dispatch failed for ${item.id}: ${dispatch.error}`);
|
|
227
259
|
const verdict = dispatch.error ? null : dispatch.verdict;
|
|
228
|
-
if (!verdict
|
|
260
|
+
if (!verdict) {
|
|
261
|
+
deferNotes.set(item.id, { reason: dispatch.error ? "judgment-error" : "judgment-parse-failure" });
|
|
262
|
+
stillDeferred.push(item);
|
|
263
|
+
continue;
|
|
264
|
+
}
|
|
265
|
+
if (verdict.decision === "defer") {
|
|
266
|
+
deferNotes.set(item.id, {
|
|
267
|
+
reason: "judgment-deferred",
|
|
268
|
+
...(verdict.reason ? { judgeReason: verdict.reason } : {}),
|
|
269
|
+
});
|
|
229
270
|
stillDeferred.push(item);
|
|
230
271
|
continue;
|
|
231
272
|
}
|
|
232
273
|
if (verdict.decision === "reject") {
|
|
233
|
-
const failure = await rejectProposal(opts, item.id, verdict.reason || "judgment: reject", "judgment-reject", rejectFn);
|
|
274
|
+
const failure = await rejectProposal(opts, item.id, verdict.reason || "judgment: reject", "judgment-reject", rejectFn, verdict.reason);
|
|
234
275
|
if (failure === undefined) {
|
|
235
276
|
result.rejected.push(item.id);
|
|
236
277
|
}
|
|
@@ -252,6 +293,7 @@ async function runJudgmentTier(opts, result, pending, acceptBudget, promoteFn, r
|
|
|
252
293
|
reason: "judgment-accept",
|
|
253
294
|
contentHash: proposalContentHash(proposal),
|
|
254
295
|
gate: DRAIN_GATE,
|
|
296
|
+
...(verdict.reason ? { judgeReason: verdict.reason } : {}),
|
|
255
297
|
});
|
|
256
298
|
result.staged.push(item.id);
|
|
257
299
|
}
|
|
@@ -265,7 +307,7 @@ async function runJudgmentTier(opts, result, pending, acceptBudget, promoteFn, r
|
|
|
265
307
|
result.skippedByCap.push(item.id);
|
|
266
308
|
continue;
|
|
267
309
|
}
|
|
268
|
-
const outcome = await acceptProposal(opts, proposal, item.id, "judgment-accept", promoteFn, rejectFn);
|
|
310
|
+
const outcome = await acceptProposal(opts, proposal, item.id, "judgment-accept", promoteFn, rejectFn, verdict.reason);
|
|
269
311
|
if (outcome === "promoted") {
|
|
270
312
|
result.promoted.push(item.id);
|
|
271
313
|
acceptBudget -= 1;
|
|
@@ -286,6 +328,21 @@ async function runJudgmentTier(opts, result, pending, acceptBudget, promoteFn, r
|
|
|
286
328
|
result.notices = notices.list();
|
|
287
329
|
result.deferred = stillDeferred;
|
|
288
330
|
}
|
|
331
|
+
/** The knowledge notes nearest to a consolidate promotion's source memory; none for any other proposal. */
|
|
332
|
+
function promotionNeighbours(stashDir, proposal) {
|
|
333
|
+
if (proposal.source !== "consolidate" || proposal.promotionSource === undefined)
|
|
334
|
+
return [];
|
|
335
|
+
try {
|
|
336
|
+
const parsed = parseRefInput(proposal.promotionSource);
|
|
337
|
+
const typeDir = stashDirFor(parsed.type);
|
|
338
|
+
if (!typeDir)
|
|
339
|
+
return [];
|
|
340
|
+
return nearestKnowledgeNotes(assetPathForName(parsed.type, path.join(stashDir, typeDir), parsed.name));
|
|
341
|
+
}
|
|
342
|
+
catch {
|
|
343
|
+
return [];
|
|
344
|
+
}
|
|
345
|
+
}
|
|
289
346
|
/** The live asset a proposal would overwrite, if any. */
|
|
290
347
|
function readLiveAssetContent(stashDir, ref) {
|
|
291
348
|
try {
|
|
@@ -314,8 +371,8 @@ export async function drainProposals(opts, promoteFn = akmProposalAccept, reject
|
|
|
314
371
|
const empties = [];
|
|
315
372
|
for (const proposal of pending) {
|
|
316
373
|
// A consolidate pair-pass `retire` proposal is auto-accepted only when the
|
|
317
|
-
// pair
|
|
318
|
-
//
|
|
374
|
+
// pair pass staged it: nothing unique on either side, confirmed by a
|
|
375
|
+
// second look, no continuity risk; every other one waits for a direct
|
|
319
376
|
// `akm proposal accept` (spec §25.6). Checked before isEmptyDiff, which
|
|
320
377
|
// has nothing meaningful to read on a delete-primary change.
|
|
321
378
|
if (isRetireProposal(proposal)) {
|
|
@@ -324,7 +381,7 @@ export async function drainProposals(opts, promoteFn = akmProposalAccept, reject
|
|
|
324
381
|
staged.gate === PAIR_PASS_GATE &&
|
|
325
382
|
staged.contentHash === proposalContentHash(proposal) &&
|
|
326
383
|
!proposal.retirement?.continuityRisk) {
|
|
327
|
-
accepts.push({ id: proposal.id, reason:
|
|
384
|
+
accepts.push({ id: proposal.id, reason: staged.reason });
|
|
328
385
|
}
|
|
329
386
|
continue;
|
|
330
387
|
}
|
|
@@ -394,15 +451,22 @@ export async function drainProposals(opts, promoteFn = akmProposalAccept, reject
|
|
|
394
451
|
}
|
|
395
452
|
}
|
|
396
453
|
}
|
|
454
|
+
const deferNotes = new Map();
|
|
397
455
|
if (opts.judgment && result.deferred.length > 0) {
|
|
398
|
-
await runJudgmentTier({ ...opts, judgment: opts.judgment }, result, pending, cap - promotedHere, promoteFn, rejectFn, judgmentSeams);
|
|
456
|
+
await runJudgmentTier({ ...opts, judgment: opts.judgment }, result, pending, cap - promotedHere, promoteFn, rejectFn, judgmentSeams, deferNotes);
|
|
399
457
|
}
|
|
400
458
|
// #577: whatever stays undecided is left for review (`review_needed` in the ledger).
|
|
401
459
|
if (!opts.dryRun) {
|
|
402
|
-
const reviewReason = opts.judgment ? "judgment-deferred" : "no-judge-configured";
|
|
403
460
|
for (const item of result.deferred) {
|
|
461
|
+
const note = deferNotes.get(item.id);
|
|
462
|
+
const reviewReason = note?.reason ?? (opts.judgment ? "judgment-deferred" : "no-judge-configured");
|
|
404
463
|
try {
|
|
405
|
-
recordGateDecision(opts.stashDir, item.id, {
|
|
464
|
+
recordGateDecision(opts.stashDir, item.id, {
|
|
465
|
+
outcome: "deferred",
|
|
466
|
+
reason: reviewReason,
|
|
467
|
+
gate: DRAIN_GATE,
|
|
468
|
+
...(note?.judgeReason ? { judgeReason: note.judgeReason } : {}),
|
|
469
|
+
});
|
|
406
470
|
}
|
|
407
471
|
catch (err) {
|
|
408
472
|
warn(`[triage] failed to record gate decision for ${item.id}: ${errMessage(err)}`);
|
|
@@ -53,7 +53,7 @@ export function isRetireProposal(proposal) {
|
|
|
53
53
|
}
|
|
54
54
|
/** A promote refused because the target changed after mint (STALE, R20) — not a merit judgement. */
|
|
55
55
|
export const STALE_TARGET_GATE_REASON = "stale-target";
|
|
56
|
-
/** The gate on a retire proposal the triage drain may accept unattended: a pair-judged
|
|
56
|
+
/** The gate on a retire proposal the triage drain may accept unattended: a staged pair-judged retirement. */
|
|
57
57
|
export const PAIR_PASS_GATE = "consolidate-pair";
|
|
58
58
|
export const EXPIRED_GATE_REASON = "expired";
|
|
59
59
|
export const ASSET_MISSING_GATE_REASON = "asset-missing";
|
|
@@ -78,9 +78,9 @@ const qualityGateField = z
|
|
|
78
78
|
.optional();
|
|
79
79
|
/**
|
|
80
80
|
* WS-3b: CLS (Complementary Learning System) interleaving (step 9).
|
|
81
|
-
* distill
|
|
82
|
-
*
|
|
83
|
-
* Default
|
|
81
|
+
* The distill prompt includes the lessons, knowledge notes and skills the library already holds near the
|
|
82
|
+
* memory, so the writer answers NONE for a rule one of them states and does not overwrite a prior
|
|
83
|
+
* generalization. Default ON; `enabled: false` turns it off. Only meaningful on the `distill` process.
|
|
84
84
|
*/
|
|
85
85
|
const clsField = z
|
|
86
86
|
.object({
|
package/dist/core/paths.js
CHANGED
|
@@ -332,15 +332,6 @@ export function getStashStateKey(stashDir) {
|
|
|
332
332
|
function stashScopedDir(base, stashDir) {
|
|
333
333
|
return path.join(base, getStashStateKey(stashDir));
|
|
334
334
|
}
|
|
335
|
-
/**
|
|
336
|
-
* `$STATE/improve/measurement/verdicts/<stash>/` — `akm-eval-proactive-verdict`
|
|
337
|
-
* reports. Moved out of `$STASH/.akm/measurement/verdicts/` (itlackey/akm#890);
|
|
338
|
-
* the pilot treatment file at `$STASH/.akm/measurement/` is manually-authored
|
|
339
|
-
* measurement input and stays put.
|
|
340
|
-
*/
|
|
341
|
-
export function getMeasurementVerdictsDir(stashDir) {
|
|
342
|
-
return stashScopedDir(path.join(getStateDir(), "improve", "measurement", "verdicts"), stashDir);
|
|
343
|
-
}
|
|
344
335
|
/**
|
|
345
336
|
* `$CACHE/index/unresolved-sources/<stash>/` — synthetic placeholder path for
|
|
346
337
|
* a configured source whose content root did not resolve this run. Never
|
package/dist/indexer/indexer.js
CHANGED
|
@@ -5,7 +5,7 @@ import fs from "node:fs";
|
|
|
5
5
|
import path from "node:path";
|
|
6
6
|
import { detectAdapterId } from "../core/adapter/detect-adapter.js";
|
|
7
7
|
import { adapterForId } from "../core/adapter/registry.js";
|
|
8
|
-
import { isHttpUrl } from "../core/common.js";
|
|
8
|
+
import { compareCodePoints, isHttpUrl } from "../core/common.js";
|
|
9
9
|
import { classifyPathAccess, describeInaccessiblePath } from "../core/path-access.js";
|
|
10
10
|
import { getDbPath } from "../core/paths.js";
|
|
11
11
|
import { SCRIPT_EXTENSIONS } from "../core/recognition-util.js";
|
|
@@ -14,6 +14,7 @@ import { isVerbose, warn, warnOnce, warnVerbose } from "../core/warn.js";
|
|
|
14
14
|
import { resolveSourcesForOrigin } from "../registry/origin-resolve.js";
|
|
15
15
|
import { closeDatabase, openExistingDatabase, openIndexDatabase, openReadonlyExistingDatabase, } from "../storage/repositories/index-connection.js";
|
|
16
16
|
import { deleteEntriesByBundle, deleteEntriesByDirAndBundle, deleteEntriesByDirExceptRefs, deleteEntriesByIds, deleteUsageEventsByEntryIds, findEntryIdByRef, getAllEntries, getEmbeddableEntryCount, getEntryCount, getIndexedBundleIdsByDir, getIndexedDirPathsByBundleId, relinkUsageEvents, upsertEntry, } from "../storage/repositories/index-entries-repository.js";
|
|
17
|
+
import { rebuildFtsIfTotalsStale } from "../storage/repositories/index-fts-repository.js";
|
|
17
18
|
import { clearStaleCacheEntries } from "../storage/repositories/index-llm-cache-repository.js";
|
|
18
19
|
import { deleteIndexDirState, deleteMeta, getIndexDirState, getMeta, setMeta, upsertIndexDirState, } from "../storage/repositories/index-meta-repository.js";
|
|
19
20
|
import { VACUUM_PENDING_META } from "../storage/repositories/index-schema.js";
|
|
@@ -125,7 +126,14 @@ export async function runEmbeddingPass(params) {
|
|
|
125
126
|
*/
|
|
126
127
|
function finalizeIndex(args) {
|
|
127
128
|
const { db, sources, sourceDirs, stashDir, signal, onProgress } = args;
|
|
128
|
-
|
|
129
|
+
// Rows replaced or removed since its BM25 totals were taken leave them off; recompute them (#1048).
|
|
130
|
+
const rebuilt = rebuildFtsIfTotalsStale(db);
|
|
131
|
+
onProgress({
|
|
132
|
+
phase: "fts",
|
|
133
|
+
message: rebuilt
|
|
134
|
+
? "Rebuilt the full-text search index to recompute its totals."
|
|
135
|
+
: "Full-text search index is current.",
|
|
136
|
+
});
|
|
129
137
|
const tFtsEnd = Date.now();
|
|
130
138
|
// Re-link state.db usage events to the regenerated index and recompute the
|
|
131
139
|
// derived utility cache. Stored refs already use the current item-ref grammar,
|
|
@@ -907,11 +915,6 @@ function requiresWorkflowSourcePreflight(ctxs) {
|
|
|
907
915
|
}
|
|
908
916
|
/** Phase 2 (sync): write all pre-generated scan records inside a single transaction. */
|
|
909
917
|
function persistDirRecords(db, dirRecords, warnings, bundleByRoot) {
|
|
910
|
-
// Per-source dedup: the same logical asset can appear more than once within
|
|
911
|
-
// one owning source, where source order still makes the first occurrence win.
|
|
912
|
-
// The owner is part of the key so identical concepts in different bundles
|
|
913
|
-
// remain distinct indexed rows.
|
|
914
|
-
const indexedAssetIdentities = new Set();
|
|
915
918
|
const deletedUsageEntryIds = new Set();
|
|
916
919
|
const findPersisted = db.prepare("SELECT id, content_hash, file_path, adapter_id FROM entries WHERE item_ref = ?");
|
|
917
920
|
const insertTransaction = db.transaction(() => {
|
|
@@ -961,7 +964,7 @@ function persistDirRecords(db, dirRecords, warnings, bundleByRoot) {
|
|
|
961
964
|
let persistedRows = 0;
|
|
962
965
|
let dedupedRows = 0;
|
|
963
966
|
if (stash) {
|
|
964
|
-
const
|
|
967
|
+
const sourceRoot = `${path.resolve(currentStashDir)}${path.sep}`;
|
|
965
968
|
for (const entry of stash.entries) {
|
|
966
969
|
const entryPath = entry.filename ? path.join(dirPath, entry.filename) : null;
|
|
967
970
|
if (!entryPath) {
|
|
@@ -973,23 +976,36 @@ function persistDirRecords(db, dirRecords, warnings, bundleByRoot) {
|
|
|
973
976
|
warn(`Skipping entry without adapter-owned concept identity: ${entryPath}`);
|
|
974
977
|
continue;
|
|
975
978
|
}
|
|
976
|
-
// Adapter-owned concept identity is path-based and cannot be replaced
|
|
977
|
-
// by presentation fields such as type/title.
|
|
978
|
-
const identityKey = `${ownerIdentity}\0${adapterConceptId}`;
|
|
979
|
-
if (indexedAssetIdentities.has(identityKey)) {
|
|
980
|
-
dedupedRows++;
|
|
981
|
-
continue;
|
|
982
|
-
}
|
|
983
|
-
indexedAssetIdentities.add(identityKey);
|
|
984
979
|
// content_hash = doc.hash from the drain, keyed by the recognized
|
|
985
980
|
// file's path. A missing hash preserves the existing value on upsert.
|
|
986
981
|
const contentHash = hashByFile?.get(entryPath);
|
|
982
|
+
// Adapter-owned concept identity is path-based and cannot be replaced
|
|
983
|
+
// by presentation fields such as type/title.
|
|
987
984
|
const provenance = deriveEntryProvenance(bundle, entry.type, entry.name, adapterConceptId);
|
|
985
|
+
// Two files of one source can claim one ref (a skill's references/a.md and
|
|
986
|
+
// knowledge/skills/x/references/a.md are both knowledge/skills/x/references/a). The file with
|
|
987
|
+
// the smaller path holds it, whichever directories this run drains and in whatever order the
|
|
988
|
+
// walk met them (#1050): an entry yields to a row held by a file with a smaller path and
|
|
989
|
+
// otherwise takes the ref over. The same concept in another bundle is a different row, and a
|
|
990
|
+
// holder whose file is gone is no claim.
|
|
991
|
+
const holder = findPersisted.get(provenance.itemRef) ?? undefined;
|
|
992
|
+
if (holder &&
|
|
993
|
+
holder.file_path !== entryPath &&
|
|
994
|
+
holder.file_path.startsWith(sourceRoot) &&
|
|
995
|
+
fs.existsSync(holder.file_path)) {
|
|
996
|
+
const yields = compareCodePoints(holder.file_path, entryPath) < 0;
|
|
997
|
+
const [indexed, skipped] = yields ? [holder.file_path, entryPath] : [entryPath, holder.file_path];
|
|
998
|
+
warnings.push(`Two files claim ${provenance.itemRef}: indexed ${indexed}, skipped ${skipped}.`);
|
|
999
|
+
if (yields) {
|
|
1000
|
+
dedupedRows++;
|
|
1001
|
+
continue;
|
|
1002
|
+
}
|
|
1003
|
+
// The directory the ref comes from now has a row fewer than its files: drain it again.
|
|
1004
|
+
deleteIndexDirState(db, path.dirname(holder.file_path));
|
|
1005
|
+
}
|
|
988
1006
|
keptItemRefs.add(provenance.itemRef);
|
|
989
1007
|
persistedRows++;
|
|
990
|
-
const previous = sameVariant
|
|
991
|
-
? (findPersisted.get(provenance.itemRef) ?? undefined)
|
|
992
|
-
: undefined;
|
|
1008
|
+
const previous = sameVariant ? holder : undefined;
|
|
993
1009
|
const unchanged = previous !== undefined &&
|
|
994
1010
|
contentHash !== undefined &&
|
|
995
1011
|
previous.content_hash === contentHash &&
|
|
@@ -19,12 +19,13 @@
|
|
|
19
19
|
* `./result-extractor.ts`. Always emitted, mirroring how the Claude builder
|
|
20
20
|
* always emits `--print`: dispatch is the captured, non-interactive path.
|
|
21
21
|
* - Codex is the NATIVE-SCHEMA tier (plan §"Structured-output
|
|
22
|
-
* normalization"): `req.schema` is written to a
|
|
23
|
-
* `--output-schema <file>`. The file is
|
|
24
|
-
*
|
|
25
|
-
* post-run hook, and the
|
|
26
|
-
*
|
|
27
|
-
*
|
|
22
|
+
* normalization"): `req.schema` is written to a file and passed via
|
|
23
|
+
* `--output-schema <file>`. The file is named by the hash of the schema
|
|
24
|
+
* and lives in akm's cache dir, so a schema is written once however many
|
|
25
|
+
* units dispatch with it: `BuiltCommand` has no post-run hook, and the
|
|
26
|
+
* spawned process reads the file after `build()` returns, so a file per
|
|
27
|
+
* dispatch could only leak (#1051). The engine still validates the output
|
|
28
|
+
* defensively (the constrained output is trusted but verified).
|
|
28
29
|
* - `codex exec` has no system-prompt flag; `req.systemPrompt` is folded
|
|
29
30
|
* into the prompt payload (system text first, blank line, then the task),
|
|
30
31
|
* after the `--` end-of-options separator so it can never be parsed as a
|
|
@@ -52,22 +53,29 @@
|
|
|
52
53
|
* that registry, so this builder is reachable under the `"codex"` platform
|
|
53
54
|
* name without any further wiring.
|
|
54
55
|
*/
|
|
55
|
-
import {
|
|
56
|
-
import { tmpdir } from "node:os";
|
|
56
|
+
import { mkdirSync } from "node:fs";
|
|
57
57
|
import { join } from "node:path";
|
|
58
|
+
import { writeFileAtomic } from "../../../core/common.js";
|
|
59
|
+
import { getCacheDir } from "../../../core/paths.js";
|
|
60
|
+
import { sha256Hex } from "../../../runtime.js";
|
|
58
61
|
import { resolveDispatchModel } from "../../agent/builder-shared.js";
|
|
59
62
|
import { createAgentRequestLowerer } from "../../agent/request-lowering.js";
|
|
60
63
|
/**
|
|
61
|
-
* Write a node's JSON Schema to a
|
|
64
|
+
* Write a node's JSON Schema to a file for `--output-schema`, named by the hash
|
|
65
|
+
* of its text, in akm's cache dir. Returns the absolute file path (the value
|
|
66
|
+
* handed to the flag).
|
|
62
67
|
*
|
|
63
|
-
*
|
|
64
|
-
*
|
|
65
|
-
*
|
|
68
|
+
* The same schema is the same file, so concurrent fan-out units share it, and
|
|
69
|
+
* the atomic rename makes a write of identical bytes safe under a reader: no
|
|
70
|
+
* unit can see a half-written schema, and none removes it from another. Nothing
|
|
71
|
+
* is cleaned up, because nothing accumulates but one file per distinct schema.
|
|
66
72
|
*/
|
|
67
73
|
export function writeCodexOutputSchemaFile(schema) {
|
|
68
|
-
const
|
|
69
|
-
const
|
|
70
|
-
|
|
74
|
+
const text = `${JSON.stringify(schema, null, 2)}\n`;
|
|
75
|
+
const dir = join(getCacheDir(), "codex-output-schemas");
|
|
76
|
+
mkdirSync(dir, { recursive: true });
|
|
77
|
+
const file = join(dir, `${sha256Hex(text)}.json`);
|
|
78
|
+
writeFileAtomic(file, text);
|
|
71
79
|
return file;
|
|
72
80
|
}
|
|
73
81
|
/**
|