akm-cli 0.9.25-alpha.2 → 0.9.25-alpha.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +102 -0
- package/dist/assets/prompts/reflect-feedback-framing.md +1 -1
- package/dist/assets/prompts/reflect-llm-framed-contract.md +2 -9
- package/dist/assets/prompts/reflect-llm-schema-contract.md +1 -3
- package/dist/assets/prompts/reflect-output-repair.md +1 -1
- package/dist/commands/improve/extract-cli.js +3 -2
- package/dist/commands/improve/extract.js +1 -1
- package/dist/commands/improve/reflect.js +140 -320
- package/dist/commands/improve/session-asset.js +6 -0
- package/dist/commands/improve/stage.js +15 -5
- package/dist/commands/proposal/validators/proposal-quality-validators.js +7 -3
- package/dist/commands/proposal/validators/proposal-validators.js +4 -5
- package/dist/core/asset/asset-serialize.js +1 -1
- package/dist/core/content-safety.js +0 -24
- package/dist/integrations/agent/prompts.js +51 -91
- package/dist/integrations/harnesses/codex/index.js +6 -11
- package/dist/integrations/harnesses/codex/session-log.js +211 -0
- package/dist/integrations/harnesses/types.js +3 -3
- package/dist/scripts/akm-migrate-node.js +258 -92
- package/dist/scripts/akm-migrate.js +258 -92
- package/docs/reference/cli.md +17 -9
- package/docs/reference/configuration.md +1 -3
- package/package.json +1 -1
|
@@ -99,6 +99,12 @@ export function buildSessionAccessInstructions(harness, logPath, sessionId) {
|
|
|
99
99
|
`Parse messages: jq -r 'select(.type=="message") | .message.content[]? | select(.type=="text") | .text' ${logPath}`,
|
|
100
100
|
].join("\n");
|
|
101
101
|
}
|
|
102
|
+
if (harness === "codex") {
|
|
103
|
+
return [
|
|
104
|
+
`Read with: cat ${logPath}`,
|
|
105
|
+
`Parse messages: jq -r 'select(.type=="response_item" and .payload.type=="message") | .payload.content[]? | .text // empty' ${logPath}`,
|
|
106
|
+
].join("\n");
|
|
107
|
+
}
|
|
102
108
|
if (harness === "opencode") {
|
|
103
109
|
return [
|
|
104
110
|
`Open the SQLite database at ${JSON.stringify(logPath)} in read-only mode.`,
|
|
@@ -269,17 +269,27 @@ function buildChangedRegion(sourceContent, candidateContent) {
|
|
|
269
269
|
*/
|
|
270
270
|
function reflectJudgeToolRules(ref) {
|
|
271
271
|
const asset = ref ? `The asset is \`${ref}\`: read it with akm_show, ` : "Read an asset with akm_show ";
|
|
272
|
-
return `Tools: ${asset}or an asset the changed region names, only to verify a fact the revision adds or alters; the text above already shows every change. Do not search, do not read anything else, and do not use a tool to judge structure or wording. One or two reads at most. Before scoring, check
|
|
272
|
+
return `Tools: ${asset}or an asset the changed region names, only to verify a fact the revision adds or alters; the text above already shows every change. Do not search, do not read anything else, and do not use a tool to judge structure or wording. One or two reads at most. Before scoring, check two lists: (1) every statement the revision adds: find each in the asset, or as a fact the feedback states about the subject, and score QUALITY 1-2 if any is in neither; a statement is found only when the asset or the feedback says it, in any words: a new step, cause, consequence or detail that merely seems to follow is not found; feedback says what to fix and is not content, so an added statement about how the asset was used, found or verified is unsupported; (2) every fact, caveat and field of the source: find each in the revision, and score PRESERVATION 1-3 if any is missing. A read that finds nothing wrong raises no score above what these lists support. Then reply with the JSON.`;
|
|
273
273
|
}
|
|
274
|
+
/**
|
|
275
|
+
* The reflect judge's rubric, for the frontmatter-only revisions reflect makes, and the paragraph that says what
|
|
276
|
+
* drove the revision: negative feedback (any `[negative]` line) or maintenance. Tuned on the production model
|
|
277
|
+
* against 113 reviewed proposals (the stash's eval/judge-gate/tuning/reflect/judge-fm, rubric f07).
|
|
278
|
+
*/
|
|
279
|
+
const REFLECT_JUDGE_INTRO = "You are evaluating a proposed revision of an existing akm asset's frontmatter. The revision may change only the `description`, the `when_to_use` and the title (a level-1 heading added when the body has none); the body is unchanged.";
|
|
280
|
+
const REFLECT_JUDGE_NEGATIVE = "This revision answers negative feedback. It cannot change the body, so it need not resolve the feedback: judge only the fields it changes, and never fault it for a field it leaves as it was. Most negative feedback says the asset did not help with a task it was retrieved for: a retrieval miss, not a defect. It justifies changing a field only when the field claims more than the body covers (narrow it to what the body covers), or when the feedback calls the asset stale, outdated, superseded or historical (a new `when_to_use` must then name the version or date the body records). Feedback about a task the asset never claims to cover justifies no change. Repairing a missing or broken field is always needed, whatever the feedback says.";
|
|
281
|
+
const REFLECT_JUDGE_MAINTENANCE = "This revision is maintenance: there is no negative feedback. Only a missing or broken field needs a change; rewording a sound `description` or `when_to_use` is churn, however accurate.";
|
|
274
282
|
/** Judge prompt for an in-place revision. `tools` is set when the judge runs on an agent engine. */
|
|
275
283
|
export function buildReflectJudgePrompt(candidateContent, sourceContent, feedback, tools) {
|
|
276
284
|
return [
|
|
277
|
-
|
|
285
|
+
REFLECT_JUDGE_INTRO,
|
|
286
|
+
"",
|
|
287
|
+
feedback.some((line) => line.startsWith("[negative]")) ? REFLECT_JUDGE_NEGATIVE : REFLECT_JUDGE_MAINTENANCE,
|
|
278
288
|
"",
|
|
279
289
|
"Score this revision on each criterion from 1 (poor) to 5 (excellent):",
|
|
280
|
-
"1. NEED: Does
|
|
281
|
-
"2. PRESERVATION: Does
|
|
282
|
-
"3. QUALITY: Is
|
|
290
|
+
"1. NEED: Does every changed field fix a real problem? Real problems: a missing `description`, `when_to_use` or title; a broken description (a sentence split by a stray period at a line wrap, an escaped or unbalanced quote, a truncated ending, a heading fragment); and, for a revision answering negative feedback, a field that claims more than the body covers, or a stale note's `when_to_use` that does not name the version or date the body records. Replacing a stray period that splits a sentence with a comma, a word or nothing repairs a broken description, however small the change looks. Score 4-5 when every changed field fixes one. Score 1-2 when any changed field rewrites a sound field. Score NEED on the changed fields alone: leaving negative feedback about the body unresolved never lowers it.",
|
|
291
|
+
"2. PRESERVATION: Does the new description keep every fact the old one carried: names, identifiers, numbers, versions, paths, qualifiers and status words such as 'Proposal' or 'draft'? Are all other frontmatter fields unchanged? Score 1-3 when anything is dropped or changed.",
|
|
292
|
+
"3. QUALITY: Is every new value supported by the body, without inventing, over-claiming or misdescribing? Check each new value as a claim against the body; restating the body in other words is supported. Score only values the revision adds or changes: a field it leaves as it was, however stale, is never this revision's fault. Score 1-2 when a new value says something the body does not support or the opposite of what it says, keeps a truncated or garbled fragment, or offers a dated or historical note for current work: when the feedback calls the note stale, outdated, superseded or historical, or the body records the version or date it was true for, a new `when_to_use` that does not name that version or date, or a new value that calls a dated snapshot 'current', scores 1-2.",
|
|
283
293
|
"",
|
|
284
294
|
"Feedback:",
|
|
285
295
|
"```",
|
|
@@ -333,7 +333,11 @@ const redactedContentValidator = {
|
|
|
333
333
|
];
|
|
334
334
|
},
|
|
335
335
|
};
|
|
336
|
-
/**
|
|
336
|
+
/**
|
|
337
|
+
* The run-only "Avoid These Patterns" section (#963) in a reflect proposal: one
|
|
338
|
+
* made before reflect kept the body, or a body that already had it. Reflect
|
|
339
|
+
* keeps a body as it is, so a person accepting has no way to remove it.
|
|
340
|
+
*/
|
|
337
341
|
const reflectPromptScaffoldingValidator = {
|
|
338
342
|
name: "reflect-prompt-scaffolding",
|
|
339
343
|
appliesTo(proposal) {
|
|
@@ -361,15 +365,15 @@ function advisory(validator) {
|
|
|
361
365
|
validate: (proposal, ctx) => validator.validate(proposal, ctx).map((finding) => ({ ...finding, severity: "warn" })),
|
|
362
366
|
};
|
|
363
367
|
}
|
|
364
|
-
/** The quality validators `validateProposal` runs; the last
|
|
368
|
+
/** The quality validators `validateProposal` runs; the last two protect durable content and block. */
|
|
365
369
|
export const defaultProposalQualityValidators = [
|
|
366
370
|
...[
|
|
367
371
|
descriptionQualityValidator,
|
|
368
372
|
lessonContentQualityValidator,
|
|
369
373
|
sourceNotSupersededValidator,
|
|
370
374
|
reflectSizeGuardValidator,
|
|
375
|
+
reflectPromptScaffoldingValidator,
|
|
371
376
|
].map(advisory),
|
|
372
377
|
reflectTruncationMarkerValidator,
|
|
373
378
|
redactedContentValidator,
|
|
374
|
-
reflectPromptScaffoldingValidator,
|
|
375
379
|
];
|
|
@@ -131,14 +131,13 @@ export const defaultProposalValidators = [
|
|
|
131
131
|
* and was previously safe to run in full there because every quality
|
|
132
132
|
* validator was advisory (`advisory()` downgrades findings to `severity:
|
|
133
133
|
* "warn"`, which {@link runProposalValidators}'s `ok` never treats as
|
|
134
|
-
* failing). Blocking durable-content validators (the #952 truncation marker
|
|
135
|
-
* #962 redaction marker
|
|
134
|
+
* failing). Blocking durable-content validators (the #952 truncation marker
|
|
135
|
+
* and #962 redaction marker) deliberately
|
|
136
136
|
* remain outside this subset, so running the full
|
|
137
137
|
* {@link defaultProposalValidators} list at mint time could throw
|
|
138
138
|
* `invalid_canonical_structure` for any lesson/task/workflow reflect
|
|
139
|
-
* proposal whose body
|
|
140
|
-
*
|
|
141
|
-
* `reflect-truncation-leak`, per the #952 design. Quality validators (prose
|
|
139
|
+
* proposal whose body carries the truncation marker — instead of minting it
|
|
140
|
+
* for a person to review. Quality validators (prose
|
|
142
141
|
* shape, reflect size ratio, and durable-content guards) belong at
|
|
143
142
|
* `proposal accept` / drain-promotion time, which already calls
|
|
144
143
|
* {@link validateProposal} (the full list) via `preflightProposalPromotion`
|
|
@@ -95,7 +95,7 @@ export function assembleAsset(frontmatter, body) {
|
|
|
95
95
|
* - exactly one `\n` terminates the file
|
|
96
96
|
*
|
|
97
97
|
* This helper is the single point of truth for the fence-and-body template.
|
|
98
|
-
*
|
|
98
|
+
* Two command surfaces (`distill`, `consolidate`) call it
|
|
99
99
|
* directly because their inputs are pre-validated LLM payloads where the
|
|
100
100
|
* full `yamlStringify` may emit shapes (`|`-block scalars, anchors) that
|
|
101
101
|
* the project's hand-rolled `parseFrontmatter` subset parser cannot read.
|
|
@@ -6,36 +6,12 @@ export const REDACTED_CONTENT_MARKER = "[REDACTED]";
|
|
|
6
6
|
/** Reflect prompt section that contains run diagnostics, not proposed asset content. */
|
|
7
7
|
export const REFLECT_AVOID_PATTERNS_HEADING = "Avoid These Patterns";
|
|
8
8
|
const REFLECT_AVOID_PATTERNS_RE = /^##[ \t]+Avoid These Patterns[ \t]*$/i;
|
|
9
|
-
const SECTION_BOUNDARY_RE = /^#{1,2}(?:[ \t]+|$)/;
|
|
10
9
|
export function containsRedactedContent(content) {
|
|
11
10
|
return content.includes(REDACTED_CONTENT_MARKER);
|
|
12
11
|
}
|
|
13
12
|
export function containsReflectPromptScaffolding(content) {
|
|
14
13
|
return content.split(/\r?\n/).some((line) => REFLECT_AVOID_PATTERNS_RE.test(line));
|
|
15
14
|
}
|
|
16
|
-
/**
|
|
17
|
-
* Remove every echoed run-only "Avoid These Patterns" section while preserving
|
|
18
|
-
* the next peer/top-level section. Reflect alone calls this sanitizer; authored
|
|
19
|
-
* source assets are never rewritten by this helper.
|
|
20
|
-
*/
|
|
21
|
-
export function stripReflectPromptScaffolding(content) {
|
|
22
|
-
const newline = content.includes("\r\n") ? "\r\n" : "\n";
|
|
23
|
-
const lines = content.split(/\r?\n/);
|
|
24
|
-
const kept = [];
|
|
25
|
-
let stripped = false;
|
|
26
|
-
for (let index = 0; index < lines.length;) {
|
|
27
|
-
if (!REFLECT_AVOID_PATTERNS_RE.test(lines[index] ?? "")) {
|
|
28
|
-
kept.push(lines[index] ?? "");
|
|
29
|
-
index += 1;
|
|
30
|
-
continue;
|
|
31
|
-
}
|
|
32
|
-
stripped = true;
|
|
33
|
-
index += 1;
|
|
34
|
-
while (index < lines.length && !SECTION_BOUNDARY_RE.test(lines[index] ?? ""))
|
|
35
|
-
index += 1;
|
|
36
|
-
}
|
|
37
|
-
return { content: kept.join(newline).replace(/(?:\r?\n){3,}/g, `${newline}${newline}`), stripped };
|
|
38
|
-
}
|
|
39
15
|
/**
|
|
40
16
|
* Return a secret-free rejection reason when generated content is unsafe to
|
|
41
17
|
* persist. `redactedContent` is the same body after applying the dispatch
|
|
@@ -70,10 +70,8 @@ function knownTypeList() {
|
|
|
70
70
|
export const REFLECT_CONTENT_CAP = 12_000;
|
|
71
71
|
/**
|
|
72
72
|
* Marker appended to truncated asset content when it exceeds the active
|
|
73
|
-
* content budget (#952). Exported so
|
|
74
|
-
*
|
|
75
|
-
* real content, and so the output contracts can reference the exact string
|
|
76
|
-
* to forbid.
|
|
73
|
+
* content budget (#952). Exported so the proposal validators can refuse a body
|
|
74
|
+
* that still carries it.
|
|
77
75
|
*/
|
|
78
76
|
export const REFLECT_TRUNCATION_MARKER = "... [truncated — focus on the visible portion]";
|
|
79
77
|
/**
|
|
@@ -119,16 +117,12 @@ export function reflectResponseContract(mode, targetScoped) {
|
|
|
119
117
|
if (mode === "json_schema") {
|
|
120
118
|
return reflectLlmSchemaContract
|
|
121
119
|
.replace("{{FIELD_RULE}}", targetScoped
|
|
122
|
-
? "The response has exactly the required fields `
|
|
123
|
-
: "The response has exactly the required fields `ref`, `
|
|
124
|
-
.replaceAll("{{TRUNCATION_MARKER}}", REFLECT_TRUNCATION_MARKER)
|
|
120
|
+
? "The response has exactly the required fields `confidence` and `frontmatterPatch`; do not echo `ref` or arbitrary `frontmatter`."
|
|
121
|
+
: "The response has exactly the required fields `ref`, `confidence`, and `frontmatterPatch`; `ref` must identify the selected asset.")
|
|
125
122
|
.trim();
|
|
126
123
|
}
|
|
127
124
|
const refLine = targetScoped ? "" : "AKM_REFLECT_REF: <selected asset ref>\n";
|
|
128
|
-
return reflectLlmFramedContract
|
|
129
|
-
.replace("{{REF_LINE}}", refLine)
|
|
130
|
-
.replaceAll("{{TRUNCATION_MARKER}}", REFLECT_TRUNCATION_MARKER)
|
|
131
|
-
.trim();
|
|
125
|
+
return reflectLlmFramedContract.replace("{{REF_LINE}}", refLine).trim();
|
|
132
126
|
}
|
|
133
127
|
export function buildReflectOutputRepairPrompt(mode, targetScoped) {
|
|
134
128
|
return reflectOutputRepair.replace("{{OUTPUT_CONTRACT}}", reflectResponseContract(mode, targetScoped)).trim();
|
|
@@ -158,23 +152,47 @@ function sourceHasNonEmptyDescription(assetContent) {
|
|
|
158
152
|
return value.length > 0;
|
|
159
153
|
}
|
|
160
154
|
/**
|
|
161
|
-
*
|
|
162
|
-
*
|
|
163
|
-
* an
|
|
164
|
-
*
|
|
155
|
+
* The frontmatter problems akm can see for itself, named in the prompt so the
|
|
156
|
+
* model fixes them instead of having to spot them: a description split by a
|
|
157
|
+
* stray period or carrying an escaped quote, no `when_to_use`, no title. A
|
|
158
|
+
* missing description has its own instruction (#636).
|
|
159
|
+
*/
|
|
160
|
+
function frontmatterProblems(assetContent) {
|
|
161
|
+
const fm = assetContent?.match(/^---\r?\n([\s\S]*?)\r?\n---\r?\n?/);
|
|
162
|
+
if (!assetContent || !fm)
|
|
163
|
+
return [];
|
|
164
|
+
const block = fm[1] ?? "";
|
|
165
|
+
const problems = [];
|
|
166
|
+
const raw = block
|
|
167
|
+
.match(/^description\s*:(.*(?:\r?\n[ \t]+.*)*)/m)?.[1]
|
|
168
|
+
?.replace(/\s+/g, " ")
|
|
169
|
+
.trim() ?? "";
|
|
170
|
+
const split = [...raw.matchAll(/[\w`)\]]\. [a-z]/g)].find((m) => !/\b(?:e\.g|i\.e|vs|etc|cf)$/i.test(raw.slice(0, (m.index ?? 0) + 1)));
|
|
171
|
+
if (split) {
|
|
172
|
+
problems.push(`the \`description\` is broken: a stray period splits a sentence ("${raw}"); repair only the break, keeping its wording and every name, number, path and status word in it`);
|
|
173
|
+
}
|
|
174
|
+
else if (raw.includes('\\"')) {
|
|
175
|
+
problems.push("the `description` is broken by an escaped quote; repair only the quoting, keeping the rest of its wording");
|
|
176
|
+
}
|
|
177
|
+
if (!/^when_to_use\s*:\s*(?:\S|\r?\n[ \t]+\S)/m.test(block)) {
|
|
178
|
+
problems.push("there is no `when_to_use`: write one, a single sentence the body supports, saying when to reach for this asset");
|
|
179
|
+
}
|
|
180
|
+
if (!/^title\s*:\s*\S/m.test(block) && !/^#[ \t]+\S/m.test(assetContent.slice(fm[0].length))) {
|
|
181
|
+
problems.push("the body has no level-1 title: give one in `title`");
|
|
182
|
+
}
|
|
183
|
+
return problems;
|
|
184
|
+
}
|
|
185
|
+
/**
|
|
186
|
+
* Build the prompt for `akm reflect [ref]`. Asks the agent to check an
|
|
187
|
+
* existing asset's `description`, `when_to_use` and title against its body
|
|
188
|
+
* (plus any negative feedback / lint findings) and return the fields that
|
|
189
|
+
* need a fix. Returns a {@link ReflectPromptResult} containing the prompt
|
|
190
|
+
* string.
|
|
165
191
|
*/
|
|
166
192
|
export function buildReflectPrompt(input) {
|
|
167
193
|
const sections = [];
|
|
168
194
|
if (input.ref && input.type && input.name) {
|
|
169
|
-
|
|
170
|
-
const isLesson = input.type === "lesson";
|
|
171
|
-
const isSkill = input.type === "skill";
|
|
172
|
-
const goalSentence = isLesson
|
|
173
|
-
? `Your task is to distill what usage signals reveal about this ${input.type} asset — when to reach for it, what goes wrong without it, and what real use has revealed that the asset itself does not say. Do not reproduce the source content; your proposal must add information the source does not contain.`
|
|
174
|
-
: isSkill
|
|
175
|
-
? "Your task is to review this skill asset, identify what the feedback and related distilled lessons show is broken, missing, unclear, or durable enough to promote into long-term documentation, and produce a single improved proposal. If the strongest evidence points to companion reference material rather than the main SKILL.md, you may instead propose a skill-adjacent knowledge doc such as `knowledge/skills/<skill>/references/<topic>`."
|
|
176
|
-
: `Your task is to review this ${input.type} asset, identify what the feedback signals as broken, missing, or unclear, and produce an improved version. Do not reproduce the source content unchanged; your proposal must correct or add something the source lacks.`;
|
|
177
|
-
sections.push(goalSentence);
|
|
195
|
+
sections.push(`Your task is to check this ${input.type} asset's \`description\`, \`when_to_use\` and title against its body and the feedback below, and fix any that is missing, broken, or claims something the body does not cover. AKM keeps the body exactly as it is: you change only these fields, and when none needs a change you return null for each.`);
|
|
178
196
|
sections.push(`Target ref: ${input.ref}`);
|
|
179
197
|
sections.push(`Asset-type guidance: ${hintForType(input.type)}`);
|
|
180
198
|
}
|
|
@@ -200,12 +218,9 @@ export function buildReflectPrompt(input) {
|
|
|
200
218
|
sections.push("Recent feedback / signals:");
|
|
201
219
|
sections.push("- (no feedback events recorded)");
|
|
202
220
|
}
|
|
203
|
-
else if (input.type === "skill" && input.relatedLessons && input.relatedLessons.length > 0) {
|
|
204
|
-
sections.push("No direct feedback events were recorded. Limit substantive changes to what is justified by the related distilled lessons below; do not speculate beyond that evidence.");
|
|
205
|
-
}
|
|
206
221
|
else {
|
|
207
|
-
// ref is set but no feedback — explicitly constrain scope to
|
|
208
|
-
sections.push("No usage feedback recorded.
|
|
222
|
+
// ref is set but no feedback — explicitly constrain scope to broken or missing fields
|
|
223
|
+
sections.push("No usage feedback recorded. Fix only a missing or broken `description`, `when_to_use` or title; otherwise return null for each.");
|
|
209
224
|
}
|
|
210
225
|
if (input.standardsContext?.trim()) {
|
|
211
226
|
sections.push("Standards to follow (the rulebook for this target):");
|
|
@@ -253,7 +268,7 @@ export function buildReflectPrompt(input) {
|
|
|
253
268
|
sections.push("```");
|
|
254
269
|
}
|
|
255
270
|
else if (input.ref) {
|
|
256
|
-
sections.push("(No existing content
|
|
271
|
+
sections.push("(No existing content.)");
|
|
257
272
|
}
|
|
258
273
|
else {
|
|
259
274
|
sections.push("(No existing asset content was supplied.)");
|
|
@@ -263,23 +278,10 @@ export function buildReflectPrompt(input) {
|
|
|
263
278
|
for (const line of input.schemaHints)
|
|
264
279
|
sections.push(`- ${line}`);
|
|
265
280
|
}
|
|
266
|
-
if (input.relatedLessons && input.relatedLessons.length > 0) {
|
|
267
|
-
sections.push("Related distilled lessons to evaluate for consolidation:");
|
|
268
|
-
for (const lesson of input.relatedLessons) {
|
|
269
|
-
sections.push(`Lesson ref: ${lesson.ref}`);
|
|
270
|
-
sections.push("```");
|
|
271
|
-
sections.push(lesson.content.trimEnd());
|
|
272
|
-
sections.push("```");
|
|
273
|
-
}
|
|
274
|
-
sections.push("Evaluate whether these lessons contain strong evidence of factual, repeatable guidance that should be promoted into long-term skill documentation.");
|
|
275
|
-
sections.push("Promote only guidance that is durable, generally applicable, and supported by repeated evidence. Do not copy anecdotal details, one-off incidents, or duplicate wording verbatim.");
|
|
276
|
-
sections.push("If the guidance belongs in the main skill instructions, update the skill proposal. If it belongs in a companion reference document, return a `knowledge/skills/<skill>/references/<topic>` proposal instead.");
|
|
277
|
-
}
|
|
278
281
|
if (input.rejectedProposals && input.rejectedProposals.length > 0) {
|
|
279
282
|
const lines = ["## Previously Rejected Proposals"];
|
|
280
283
|
lines.push("The following proposals for this ref were already reviewed and rejected. " +
|
|
281
|
-
"Do
|
|
282
|
-
"Your new proposal must meaningfully differ from each of these in its approach, framing, or evidence used.");
|
|
284
|
+
"Do not propose the same change again; if no other change is justified, return null for each field.");
|
|
283
285
|
for (const rp of input.rejectedProposals) {
|
|
284
286
|
lines.push(`\nRef: ${rp.ref}`);
|
|
285
287
|
lines.push(`Rejection reason: ${rp.reason}`);
|
|
@@ -303,58 +305,16 @@ export function buildReflectPrompt(input) {
|
|
|
303
305
|
"The following is your previous draft proposal. " +
|
|
304
306
|
"Identify specific weaknesses: missing evidence, vague wording, incomplete frontmatter, " +
|
|
305
307
|
"or claims that duplicate existing content without adding new signal. " +
|
|
306
|
-
"Then produce an improved version that addresses those weaknesses
|
|
307
|
-
"The revised proposal must be meaningfully better than the draft below — " +
|
|
308
|
-
"do not return the same content unchanged.\n\n" +
|
|
308
|
+
"Then produce an improved version that addresses those weaknesses.\n\n" +
|
|
309
309
|
"Previous draft:\n```\n" +
|
|
310
310
|
input.priorDraft.trimEnd() +
|
|
311
311
|
"\n```");
|
|
312
312
|
}
|
|
313
|
-
sections.push("Produce a single proposal that addresses the feedback and respects the asset-type contract. If the proposal's frontmatter is missing `when_to_use`, you MUST generate one — a one-line trigger sentence describing exactly when a user should reach for this asset.");
|
|
314
|
-
// Content-preservation safety rails (#reflect-pipeline-fixes).
|
|
315
|
-
// These rules counter the observed failure modes where reflect rewrites
|
|
316
|
-
// asset content into shorter prose, drops concrete structure, or strips
|
|
317
|
-
// load-bearing frontmatter. Loud and explicit so small models follow.
|
|
318
|
-
//
|
|
319
|
-
// Guard-audit finding 15: this used to also hand back a maxOutputChars
|
|
320
|
-
// value so an LLM-path caller could convert it into a hard `max_tokens`
|
|
321
|
-
// cap on the API request. llm/client.ts's own doc comment (and
|
|
322
|
-
// commands/improve/reflect.ts's recorded history of responses actually
|
|
323
|
-
// getting cut off) is explicit that a character-derived max_tokens causes
|
|
324
|
-
// silent truncation — a real model's output is measured in tokens, not
|
|
325
|
-
// characters, and the ratio between the two varies enough that any fixed
|
|
326
|
-
// conversion either truncates legitimate output or provides no real cap at
|
|
327
|
-
// all. The size policy below is already enforced twice more (the prompt
|
|
328
|
-
// rules the model reads, and the post-processor's own size check), so nothing
|
|
329
|
-
// is lost by not adding a THIRD, byte-derived enforcement point that can
|
|
330
|
-
// only ever cut a response off early, never usefully re-check it.
|
|
331
313
|
if (input.ref && input.assetContent?.trim()) {
|
|
332
|
-
|
|
333
|
-
|
|
334
|
-
|
|
335
|
-
|
|
336
|
-
const sourceBodyLen = (fmBodyMatch ? fmBodyMatch[1] : rawContent).trim().length;
|
|
337
|
-
// Compute concrete char bounds matching checkReflectSize constants:
|
|
338
|
-
// REFLECT_SIZE_GUARD_MIN_BYTES=200, REFLECT_SHRINK_RATIO_MIN=0.5,
|
|
339
|
-
// REFLECT_ABSOLUTE_FLOOR_BYTES=150, REFLECT_EXPAND_RATIO_MAX=2.5,
|
|
340
|
-
// REFLECT_ABSOLUTE_CEILING_BYTES=2500, REFLECT_ABSOLUTE_MAX_BYTES=25000.
|
|
341
|
-
// Embed concrete counts only when the gate will actually fire (source >= 200 chars).
|
|
342
|
-
const showCharBounds = sourceBodyLen >= 200;
|
|
343
|
-
const minChars = Math.max(Math.round(0.5 * sourceBodyLen), 150);
|
|
344
|
-
// A source already past the 25000 cap may not grow, and need not shrink to the cap.
|
|
345
|
-
const maxChars = Math.max(Math.min(Math.max(Math.round(2.5 * sourceBodyLen), 2500), 25000), sourceBodyLen);
|
|
346
|
-
sections.push([
|
|
347
|
-
"## Content preservation rules (MUST follow)",
|
|
348
|
-
"1. PRESERVE ALL concrete content: code blocks, fenced snippets, CLI commands, numbered/bulleted checklists, tables, YAML/JSON examples, file paths, configuration keys, environment variable names, and CSS/HTML selectors. These are load-bearing — do NOT replace them with prose summaries.",
|
|
349
|
-
"2. PRESERVE the source asset's frontmatter. The post-processor reassembles the final asset from the original frontmatter plus your body. Do NOT emit `---` frontmatter delimiters at the top of `content` — start `content` with the markdown body (e.g. `# Heading` or the first paragraph). If you include frontmatter anyway, identity fields (`name`, `ref`, `id`, `slug`, `type`) will be reset to the original values.",
|
|
350
|
-
showCharBounds
|
|
351
|
-
? `3. DO NOT shrink the asset. Your body must be at least ${minChars} characters (source body is ${sourceBodyLen} chars; floor is 50%). If you genuinely need to remove a major section, explain why in a comment line at the top of the body (e.g. \`<!-- removed obsolete section X because ... -->\`).`
|
|
352
|
-
: "3. DO NOT shrink the asset dramatically. The improved body must be at least 50% of the source body length. If you genuinely need to remove a major section, explain why in a comment line at the top of the body (e.g. `<!-- removed obsolete section X because ... -->`).",
|
|
353
|
-
showCharBounds
|
|
354
|
-
? `4. DO NOT pad the asset with speculative material. Your body must be at most ${maxChars} characters (source body is ${sourceBodyLen} chars; ceiling is ${maxChars === sourceBodyLen ? "100%" : "250%"}). Do not add invented sections, hypothetical examples, or padding prose.`
|
|
355
|
-
: "4. DO NOT pad the asset with speculative material. The improved body must be at most 250% of the source body length unless the feedback explicitly requests added sections.",
|
|
356
|
-
"5. Improve clarity of surrounding prose, fix structural issues, add missing required frontmatter fields. Do NOT rewrite a runbook into an essay.",
|
|
357
|
-
].join("\n"));
|
|
314
|
+
const problems = frontmatterProblems(input.assetContent);
|
|
315
|
+
sections.push(problems.length > 0
|
|
316
|
+
? `akm found these problems in the asset; fix each one:\n${problems.map((p) => `- ${p}`).join("\n")}`
|
|
317
|
+
: "akm found no missing or broken field.");
|
|
358
318
|
}
|
|
359
319
|
sections.push(reflectResponseContract(input.outputMode ?? "json_schema", input.ref !== undefined));
|
|
360
320
|
return { prompt: sections.join("\n\n") };
|
|
@@ -1,23 +1,14 @@
|
|
|
1
1
|
// This Source Code Form is subject to the terms of the Mozilla Public
|
|
2
2
|
// License, v. 2.0. If a copy of the MPL was not distributed with this
|
|
3
3
|
// file, You can obtain one at https://mozilla.org/MPL/2.0/.
|
|
4
|
-
/**
|
|
5
|
-
* OpenAI Codex CLI harness (P2 integration, plan §"The adapter contract").
|
|
6
|
-
*
|
|
7
|
-
* Per-harness barrel gathering the Codex integration surfaces:
|
|
8
|
-
* - agent command builder → ./agent-builder.ts (codexBuilder)
|
|
9
|
-
* - result extractor → ./result-extractor.ts (codexResultExtractor)
|
|
10
|
-
*
|
|
11
|
-
* It also defines {@link CodexHarness}, the {@link AkmHarness} descriptor that
|
|
12
|
-
* `HARNESS_REGISTRY` registers. Dispatch-only: no native session-log reader or
|
|
13
|
-
* config importer yet.
|
|
14
|
-
*/
|
|
15
4
|
import { caps } from "../shared.js";
|
|
16
5
|
import { BaseHarness } from "../types.js";
|
|
17
6
|
import { codexBuilder } from "./agent-builder.js";
|
|
18
7
|
import { codexResultExtractor } from "./result-extractor.js";
|
|
8
|
+
import { CodexProvider } from "./session-log.js";
|
|
19
9
|
export { codexBuilder, codexResumeArgs, writeCodexOutputSchemaFile } from "./agent-builder.js";
|
|
20
10
|
export { codexResultExtractor } from "./result-extractor.js";
|
|
11
|
+
export { CodexProvider } from "./session-log.js";
|
|
21
12
|
/**
|
|
22
13
|
* OpenAI Codex CLI.
|
|
23
14
|
*
|
|
@@ -26,6 +17,8 @@ export { codexResultExtractor } from "./result-extractor.js";
|
|
|
26
17
|
export class CodexHarness extends BaseHarness {
|
|
27
18
|
id = "codex";
|
|
28
19
|
displayName = "OpenAI Codex CLI";
|
|
20
|
+
// No `setupDetectionDir`: `~/.codex` holds credentials and session rollouts,
|
|
21
|
+
// not assets, so `akm setup` must not offer it as a stash source.
|
|
29
22
|
agentBuilder = codexBuilder;
|
|
30
23
|
resultExtractor = codexResultExtractor;
|
|
31
24
|
// ── Workflow-engine descriptor (plan §"Capability matrix", P2) ────────────
|
|
@@ -40,7 +33,9 @@ export class CodexHarness extends BaseHarness {
|
|
|
40
33
|
// user config-dir var commonly exported in shell profiles, so it would stamp
|
|
41
34
|
// identity onto manual runs (see `AkmHarness.presenceEnv`).
|
|
42
35
|
presenceEnv = ["CODEX_SANDBOX"];
|
|
36
|
+
sessionLogProvider = () => new CodexProvider();
|
|
43
37
|
capabilities = caps({
|
|
38
|
+
sessionLogs: true,
|
|
44
39
|
agentDispatch: true,
|
|
45
40
|
detection: true,
|
|
46
41
|
});
|
|
@@ -0,0 +1,211 @@
|
|
|
1
|
+
// This Source Code Form is subject to the terms of the Mozilla Public
|
|
2
|
+
// License, v. 2.0. If a copy of the MPL was not distributed with this
|
|
3
|
+
// file, You can obtain one at https://mozilla.org/MPL/2.0/.
|
|
4
|
+
import fs from "node:fs";
|
|
5
|
+
import os from "node:os";
|
|
6
|
+
import path from "node:path";
|
|
7
|
+
import { extractInlineRefMentions } from "../../session-logs/inline-refs.js";
|
|
8
|
+
import { AbstractSessionLogProvider } from "../../session-logs/provider-base.js";
|
|
9
|
+
/**
|
|
10
|
+
* Root directory holding Codex's rollout files, one per session, as
|
|
11
|
+
* `YYYY/MM/DD/rollout-<timestamp>-<thread-id>.jsonl`.
|
|
12
|
+
*
|
|
13
|
+
* Resolved per call (not memoized at module load) so `CODEX_HOME` — Codex's own
|
|
14
|
+
* override of `~/.codex` — can be set after import; tests point it at a fixture
|
|
15
|
+
* directory instead of the real history.
|
|
16
|
+
*/
|
|
17
|
+
function codexSessionsDir() {
|
|
18
|
+
return path.join(process.env.CODEX_HOME || path.join(os.homedir(), ".codex"), "sessions");
|
|
19
|
+
}
|
|
20
|
+
/** Bytes read to get a rollout's first record, `session_meta`: its embedded base instructions make it 20-50 KB. */
|
|
21
|
+
const META_PEEK_BYTES = 128 * 1024;
|
|
22
|
+
/**
|
|
23
|
+
* User-role blocks Codex injects itself, which are not the person's words: the
|
|
24
|
+
* AGENTS.md instructions and the context tags it is known to wrap (environment,
|
|
25
|
+
* plugin recommendations, an invoked skill's text, ...). Any other block is kept:
|
|
26
|
+
* a stray bit of context costs less than a dropped reply.
|
|
27
|
+
*/
|
|
28
|
+
const INJECTED_CONTEXT_RE = /^\s*(?:# AGENTS\.md instructions[\s\S]*<\/INSTRUCTIONS>|<(environment_context|user_instructions|recommended_plugins|skill|turn_aborted)>[\s\S]*<\/\1>)\s*$/i;
|
|
29
|
+
/** The `session_meta` record, always a rollout's first; `undefined` for any other record. */
|
|
30
|
+
function parseSessionMeta(entry) {
|
|
31
|
+
const e = entry;
|
|
32
|
+
if (e?.type !== "session_meta" || !e.payload)
|
|
33
|
+
return undefined;
|
|
34
|
+
const { timestamp, cwd, source } = e.payload;
|
|
35
|
+
const startedAt = typeof timestamp === "string" ? Date.parse(timestamp) : Number.NaN;
|
|
36
|
+
return {
|
|
37
|
+
...(Number.isNaN(startedAt) ? {} : { startedAt }),
|
|
38
|
+
...(typeof cwd === "string" ? { cwd } : {}),
|
|
39
|
+
isAgent: typeof source === "object" && source !== null && ("subagent" in source || "internal" in source),
|
|
40
|
+
};
|
|
41
|
+
}
|
|
42
|
+
/** The text of a message's or tool output's content: a plain string, or blocks whose images and audio carry none. */
|
|
43
|
+
function blockText(blocks, keep = () => true) {
|
|
44
|
+
if (typeof blocks === "string")
|
|
45
|
+
return blocks;
|
|
46
|
+
if (!Array.isArray(blocks))
|
|
47
|
+
return "";
|
|
48
|
+
const parts = [];
|
|
49
|
+
for (const block of blocks) {
|
|
50
|
+
const text = block?.text;
|
|
51
|
+
if (typeof text === "string" && keep(text))
|
|
52
|
+
parts.push(text);
|
|
53
|
+
}
|
|
54
|
+
return parts.join("\n");
|
|
55
|
+
}
|
|
56
|
+
/**
|
|
57
|
+
* A tool call's input as text. A shell call surfaces its command line (`cmd`, or
|
|
58
|
+
* `command` as a string or an argv array) so the inline-ref scanner can match
|
|
59
|
+
* `akm remember "..."` without JSON-quote escaping mangling the regex; any
|
|
60
|
+
* other input stays as it was written.
|
|
61
|
+
*/
|
|
62
|
+
function toolInputText(input) {
|
|
63
|
+
let args = input;
|
|
64
|
+
if (typeof input === "string") {
|
|
65
|
+
try {
|
|
66
|
+
args = JSON.parse(input);
|
|
67
|
+
}
|
|
68
|
+
catch {
|
|
69
|
+
return input;
|
|
70
|
+
}
|
|
71
|
+
}
|
|
72
|
+
const fields = args && typeof args === "object" ? args : {};
|
|
73
|
+
const cmd = fields.cmd ?? fields.command;
|
|
74
|
+
if (typeof cmd === "string")
|
|
75
|
+
return cmd;
|
|
76
|
+
if (Array.isArray(cmd))
|
|
77
|
+
return cmd.join(" ");
|
|
78
|
+
return typeof input === "string" ? input : (JSON.stringify(input) ?? "");
|
|
79
|
+
}
|
|
80
|
+
/**
|
|
81
|
+
* Parse one rollout record into a normalized {@link SessionEvent}. The
|
|
82
|
+
* conversation lives in `response_item` records; the rest (`event_msg` UI events
|
|
83
|
+
* that repeat it, `turn_context`, `world_state`, token counts, compaction
|
|
84
|
+
* markers) and items with nothing to read (`reasoning` is encrypted) give
|
|
85
|
+
* `undefined`. Codex's `developer` messages are its own instructions, so they
|
|
86
|
+
* are skipped too.
|
|
87
|
+
*
|
|
88
|
+
* Tool calls become `[tool:<name>] <input>` and their results `[tool_result]
|
|
89
|
+
* <output>`, the shapes the Claude reader writes. A result is a `tool` event:
|
|
90
|
+
* Codex has no enclosing user message to give it.
|
|
91
|
+
*/
|
|
92
|
+
function parseCodexRecord(entry, sessionId, filePath, fallbackTsMs) {
|
|
93
|
+
const e = entry;
|
|
94
|
+
const item = e?.type === "response_item" ? e.payload : undefined;
|
|
95
|
+
if (!item)
|
|
96
|
+
return undefined;
|
|
97
|
+
const toolName = typeof item.name === "string" ? item.name : "tool";
|
|
98
|
+
let role = "assistant";
|
|
99
|
+
let text;
|
|
100
|
+
switch (item.type) {
|
|
101
|
+
case "message":
|
|
102
|
+
if (item.role !== "user" && item.role !== "assistant")
|
|
103
|
+
return undefined;
|
|
104
|
+
role = item.role;
|
|
105
|
+
text = blockText(item.content, item.role === "user" ? (t) => !INJECTED_CONTEXT_RE.test(t) : undefined);
|
|
106
|
+
break;
|
|
107
|
+
case "function_call":
|
|
108
|
+
text = `[tool:${toolName}] ${toolInputText(item.arguments)}`;
|
|
109
|
+
break;
|
|
110
|
+
case "custom_tool_call":
|
|
111
|
+
text = `[tool:${toolName}] ${typeof item.input === "string" ? item.input : ""}`;
|
|
112
|
+
break;
|
|
113
|
+
case "local_shell_call":
|
|
114
|
+
text = `[tool:shell] ${toolInputText(item.action)}`;
|
|
115
|
+
break;
|
|
116
|
+
case "function_call_output":
|
|
117
|
+
case "custom_tool_call_output": {
|
|
118
|
+
role = "tool";
|
|
119
|
+
const output = blockText(item.output);
|
|
120
|
+
text = output ? `[tool_result] ${output}` : "";
|
|
121
|
+
break;
|
|
122
|
+
}
|
|
123
|
+
default:
|
|
124
|
+
return undefined;
|
|
125
|
+
}
|
|
126
|
+
if (!text.trim())
|
|
127
|
+
return undefined;
|
|
128
|
+
const ts = typeof e?.timestamp === "string" ? Date.parse(e.timestamp) || fallbackTsMs : fallbackTsMs;
|
|
129
|
+
return { harness: "codex", text, ts, sessionId, role, filePath };
|
|
130
|
+
}
|
|
131
|
+
/**
|
|
132
|
+
* Codex native session-log reader.
|
|
133
|
+
*
|
|
134
|
+
* Events, refs, extraction keys, and the harness registry all use `codex`. The
|
|
135
|
+
* session id is the thread id in the rollout's file name (the id `codex resume`
|
|
136
|
+
* takes) and `projectHint` is the session's working directory.
|
|
137
|
+
*/
|
|
138
|
+
export class CodexProvider extends AbstractSessionLogProvider {
|
|
139
|
+
name = "codex";
|
|
140
|
+
availabilityRoot() {
|
|
141
|
+
return codexSessionsDir();
|
|
142
|
+
}
|
|
143
|
+
listSessions(input = {}) {
|
|
144
|
+
return this.listSessionsFromFiles({
|
|
145
|
+
sinceMs: input.sinceMs ?? 0,
|
|
146
|
+
enumerate: () => this.walkFiles(input.location ?? codexSessionsDir(), (name) => /^rollout-.+\.jsonl$/.test(name)),
|
|
147
|
+
summarize: (rolloutPath, stat) => {
|
|
148
|
+
const meta = this.#peekMeta(rolloutPath);
|
|
149
|
+
if (meta?.isAgent)
|
|
150
|
+
return undefined;
|
|
151
|
+
return this.sessionRef({
|
|
152
|
+
// The thread id is the 36-character UUID that ends the file name.
|
|
153
|
+
sessionId: path.basename(rolloutPath, ".jsonl").slice(-36),
|
|
154
|
+
filePath: rolloutPath,
|
|
155
|
+
startedAt: meta?.startedAt ?? stat.ctimeMs,
|
|
156
|
+
endedAt: stat.mtimeMs,
|
|
157
|
+
projectHint: meta?.cwd,
|
|
158
|
+
});
|
|
159
|
+
},
|
|
160
|
+
});
|
|
161
|
+
}
|
|
162
|
+
readSession(ref) {
|
|
163
|
+
const stat = fs.statSync(ref.filePath);
|
|
164
|
+
const events = [];
|
|
165
|
+
const inlineRefs = [];
|
|
166
|
+
let meta;
|
|
167
|
+
for (const line of fs.readFileSync(ref.filePath, "utf8").split("\n").filter(Boolean)) {
|
|
168
|
+
let entry;
|
|
169
|
+
try {
|
|
170
|
+
entry = JSON.parse(line);
|
|
171
|
+
}
|
|
172
|
+
catch {
|
|
173
|
+
continue; // a partial line (the session is still being written) or garbage
|
|
174
|
+
}
|
|
175
|
+
meta ??= parseSessionMeta(entry);
|
|
176
|
+
const parsed = parseCodexRecord(entry, ref.sessionId, ref.filePath, stat.mtimeMs);
|
|
177
|
+
if (!parsed)
|
|
178
|
+
continue;
|
|
179
|
+
events.push(parsed);
|
|
180
|
+
inlineRefs.push(...extractInlineRefMentions(parsed.text, parsed.ts));
|
|
181
|
+
}
|
|
182
|
+
return {
|
|
183
|
+
ref: this.sessionRef({
|
|
184
|
+
sessionId: ref.sessionId,
|
|
185
|
+
filePath: ref.filePath,
|
|
186
|
+
startedAt: meta?.startedAt ?? events[0]?.ts ?? stat.ctimeMs,
|
|
187
|
+
endedAt: events[events.length - 1]?.ts ?? stat.mtimeMs,
|
|
188
|
+
projectHint: meta?.cwd,
|
|
189
|
+
}),
|
|
190
|
+
events,
|
|
191
|
+
inlineRefs,
|
|
192
|
+
};
|
|
193
|
+
}
|
|
194
|
+
/** The `session_meta` record on a rollout's first line, read without loading the file. */
|
|
195
|
+
#peekMeta(filePath) {
|
|
196
|
+
try {
|
|
197
|
+
const fd = fs.openSync(filePath, "r");
|
|
198
|
+
try {
|
|
199
|
+
const head = Buffer.alloc(META_PEEK_BYTES);
|
|
200
|
+
const firstLine = head.toString("utf8", 0, fs.readSync(fd, head, 0, head.length, 0)).split("\n", 1)[0] ?? "";
|
|
201
|
+
return parseSessionMeta(JSON.parse(firstLine));
|
|
202
|
+
}
|
|
203
|
+
finally {
|
|
204
|
+
fs.closeSync(fd);
|
|
205
|
+
}
|
|
206
|
+
}
|
|
207
|
+
catch {
|
|
208
|
+
return undefined; // unreadable / vanished file, or a first line that is not one whole record
|
|
209
|
+
}
|
|
210
|
+
}
|
|
211
|
+
}
|