akm-cli 0.9.26 → 0.9.27-alpha.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +125 -0
- package/dist/assets/prompts/consolidate-system.md +2 -2
- package/dist/assets/prompts/extract-session.md +2 -2
- package/dist/commands/improve/consolidate/pair-pass.js +1 -1
- package/dist/commands/improve/consolidate.js +14 -4
- package/dist/commands/improve/distill.js +9 -5
- package/dist/commands/improve/extract-prompt.js +61 -39
- package/dist/commands/improve/extract.js +2 -1
- package/dist/commands/improve/reflect.js +36 -4
- package/dist/commands/improve/retrieval-gate.js +1 -1
- package/dist/commands/improve/session-asset.js +3 -2
- package/dist/commands/improve/stage.js +1 -1
- package/dist/indexer/indexer.js +35 -19
- package/dist/integrations/harnesses/codex/agent-builder.js +23 -15
- package/dist/llm/client.js +44 -14
- package/dist/llm/memory-infer.js +1 -1
- package/dist/scripts/akm-migrate-node.js +25 -20
- package/dist/scripts/akm-migrate.js +25 -20
- package/dist/storage/repositories/index-entry-schema.js +20 -4
- package/dist/storage/repositories/index-fts-repository.js +44 -3
- package/dist/storage/repositories/index-schema.js +14 -8
- package/docs/reference/cli.md +1 -0
- package/docs/reference/configuration.md +5 -2
- package/package.json +1 -1
package/CHANGELOG.md
CHANGED
|
@@ -6,6 +6,131 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/).
|
|
|
6
6
|
|
|
7
7
|
## [Unreleased]
|
|
8
8
|
|
|
9
|
+
## [0.9.27-alpha.1] - 2026-10-06
|
|
10
|
+
|
|
11
|
+
### Fixed
|
|
12
|
+
|
|
13
|
+
- **A codex dispatch with an output schema no longer leaves a temp folder
|
|
14
|
+
behind.** Every build of the codex command for a request with a schema made a
|
|
15
|
+
new `akm-codex-schema-*` folder in the OS temp dir for `--output-schema` and
|
|
16
|
+
nothing ever removed it (the builder has no post-run hook, and the file is read
|
|
17
|
+
after it returns), so a machine whose `/tmp` is tmpfs held a folder in RAM per
|
|
18
|
+
dispatch until reboot. The schema is now written once to akm's cache dir, in a
|
|
19
|
+
file named by its hash: concurrent units dispatching the same schema share it,
|
|
20
|
+
a rewrite is an atomic rename of identical bytes, and the only residue is one
|
|
21
|
+
small file per distinct schema.
|
|
22
|
+
- **An index that has been updated ranks like a fresh index of the same files,
|
|
23
|
+
and `akm index --full` no longer doubles the full-text totals.** `entries_fts`
|
|
24
|
+
is contentless, and FTS5 cannot take a deleted row out of a contentless
|
|
25
|
+
table's BM25 totals (its row count, and the token counts the average document
|
|
26
|
+
length comes from). Every replaced or removed row left them one row too high:
|
|
27
|
+
40 notes read 40, then 41 after one edit, then 81 after `--full`, and a delete
|
|
28
|
+
never lowered them, so scores drifted away from what a fresh index gives.
|
|
29
|
+
SQLite has no command that recomputes them (`delete` and `rebuild` are refused
|
|
30
|
+
on a contentless table), so a delete that removes a row now stamps
|
|
31
|
+
`index_meta.ftsTotalsStale` in its own transaction, whichever process made it,
|
|
32
|
+
and the next `akm index` rebuilds the table from `entries` before it finishes:
|
|
33
|
+
about a second at 25,000 entries, and only when rows have left the table. A
|
|
34
|
+
row the write-path index replaced after an accepted proposal is settled the
|
|
35
|
+
same way, and an index that has already drifted is corrected by the first run
|
|
36
|
+
that replaces or removes a row.
|
|
37
|
+
- **`akm improve` runs against an API that rejects `chat_template_kwargs`,
|
|
38
|
+
OpenAI's among them.** Improve's reflect, consolidate and judge calls always
|
|
39
|
+
ask for thinking off, and the client sends that as
|
|
40
|
+
`chat_template_kwargs.enable_thinking` and a top-level `enable_thinking`. A
|
|
41
|
+
strict API answers 400 `Unknown parameter: 'chat_template_kwargs'`, the retry
|
|
42
|
+
without the response schema sent both fields again, and `akm improve judge`
|
|
43
|
+
reported `judge timeout/error — routed to review`. No engine setting could
|
|
44
|
+
stop it: the call sites override the engine's `enableThinking`, and
|
|
45
|
+
`extraParams` can only add fields. A 4xx that names either field is now
|
|
46
|
+
answered by one retry without both (which may in turn fall back without the
|
|
47
|
+
schema), and akm stops sending them to that endpoint and model for the rest
|
|
48
|
+
of the process, as it already does for `response_format`.
|
|
49
|
+
- **A lesson the model wrote without a `when_to_use` is no longer thrown away
|
|
50
|
+
silently by `akm proposal extract`.** The extract schema left `when_to_use`
|
|
51
|
+
optional while the parser dropped a lesson without one (or with one under 15
|
|
52
|
+
characters) and said nothing, so a model that followed the schema could
|
|
53
|
+
write a sound lesson that akm discarded and the session reported no
|
|
54
|
+
candidates: on akm-eval's extract eval qwen3.8-27b kept 3 of 11 expected
|
|
55
|
+
insights against 9 for gpt-oss-120b. Every property of the schema is now
|
|
56
|
+
required, `when_to_use` and `rationale_if_empty` included, with an empty
|
|
57
|
+
string for none (a memory or knowledge candidate needs no trigger, a
|
|
58
|
+
non-empty answer no rationale), which is also what a strict structured-output
|
|
59
|
+
provider needs; the prompt's output contract says the same. Any candidate the
|
|
60
|
+
contract still refuses, for this reason or another, is named in its session's
|
|
61
|
+
`warnings` as `<type>:<name> dropped: <reason>`.
|
|
62
|
+
- **Distill's response schemas are valid for a strict structured-output
|
|
63
|
+
provider.** The client sends a response schema `strict: true`, and OpenAI
|
|
64
|
+
refuses one whose objects leave a property out of `required`: `400 Invalid
|
|
65
|
+
schema for response_format 'akm_response': ... Missing 'tags'`. The lesson
|
|
66
|
+
schema left out `tags`, and the knowledge schema `tags` and `sources`, so
|
|
67
|
+
the first distill request on such a provider always failed. After a 4xx the
|
|
68
|
+
client retries once without the schema, but a gateway that answers the same
|
|
69
|
+
rejection with a 502 is not retried, and every distill call through it
|
|
70
|
+
failed; the workaround, `supportsJsonSchema: false`, loses the guidance that
|
|
71
|
+
keeps a model from leaving out `when_to_use`. Every property is now
|
|
72
|
+
required, and an empty array stands for none (distill already dropped an
|
|
73
|
+
empty `tags` or `sources`).
|
|
74
|
+
- **`akm improve` no longer reflects on an asset whose file a proposal would not
|
|
75
|
+
write, so accepting a reflect proposal no longer adds a second file.** A
|
|
76
|
+
skill's `references/a.md` is indexed as `knowledge/skills/<name>/references/a`,
|
|
77
|
+
but a proposal writes the path derived from that ref,
|
|
78
|
+
`knowledge/skills/<name>/references/a.md`. Nothing is there, so the proposal
|
|
79
|
+
was a `create`, and accepting it wrote a copy beside the skill's own file;
|
|
80
|
+
later proposals then revised the copy while the skill's file drifted. Reflect
|
|
81
|
+
now refuses before it calls the model, naming the file and the path a
|
|
82
|
+
proposal would write, and the loop records it as a skip (`unsupported_type`,
|
|
83
|
+
`file_outside_layout` in the `reflect_completed` event). This is the rule
|
|
84
|
+
0.9.26 added to `akm feedback --replace`. An asset that another bundle owns is
|
|
85
|
+
still refused by `createProposal` (#1000).
|
|
86
|
+
- **An index that 0.9.1 wrote no longer crashes akm.** Every `akm index` on it
|
|
87
|
+
exited 70 with `null is not an object (evaluating 'doc.xrefs')`, `akm migrate
|
|
88
|
+
apply` and `akm index --full` did not help, and `akm search` failed with
|
|
89
|
+
`null is not an object (evaluating 'item.entry.quality')`; the only way out
|
|
90
|
+
was moving `index.db` aside. That layout (20) keeps the transitional
|
|
91
|
+
`entry_key`, `dir_path`, `stash_dir`, `entry_json` and `entry_type` columns
|
|
92
|
+
that layout 21 removed, each NOT NULL, beside the current columns, and leaves
|
|
93
|
+
`document_json` NULL on every row. The table had every column akm checks for,
|
|
94
|
+
so it was taken for a current one: the links migration read the NULL
|
|
95
|
+
documents, and no insert could ever have succeeded (`NOT NULL constraint
|
|
96
|
+
failed: entries.entry_key`). akm now treats a table that still has a retired
|
|
97
|
+
column as older than layout 21, as the compat notes already said: the
|
|
98
|
+
writable opener recreates its entries-keyed tables (the LLM enrichment cache
|
|
99
|
+
is kept) and the next `akm index` re-walks every source, and a read rebuilds
|
|
100
|
+
inline. An index that a failed open already half-migrated (layout stamp still
|
|
101
|
+
20, `search_text` dropped, `asset_links` created) recovers the same way.
|
|
102
|
+
- **Two files that claim one ref no longer trade places in the index, and `akm
|
|
103
|
+
index` says so.** A skill's `references/a.md` and a note at
|
|
104
|
+
`knowledge/skills/x/references/a.md` are both the ref
|
|
105
|
+
`knowledge/skills/x/references/a`, and the index holds one row for it. The
|
|
106
|
+
first file a run persisted held it, so a full build followed the filesystem's
|
|
107
|
+
listing order (one bundle indexed on tmpfs and on ext4 held different files),
|
|
108
|
+
and the first incremental run after a full build handed the row to the other
|
|
109
|
+
file, because it drains only the directory that lost: with no file touched,
|
|
110
|
+
the row, its search entry and the text its vector is embedded from changed,
|
|
111
|
+
and the vector was dropped and recomputed. When a smaller-path file was added
|
|
112
|
+
later and then deleted, the ref also left the index until `--full`, although
|
|
113
|
+
its other file was still on disk. The file with the smaller path (code-point
|
|
114
|
+
order, as `akm show`'s refusal lists them) now holds the ref however the
|
|
115
|
+
directories are drained and the walk is ordered, a directory that gives a ref
|
|
116
|
+
up is drained again so the ref passes back when its holder goes, and each
|
|
117
|
+
pair is reported in the `warnings` of `akm index`, naming the file indexed
|
|
118
|
+
and the one skipped.
|
|
119
|
+
- **Consolidate's plan schema and the session summary schema are valid for a
|
|
120
|
+
strict structured-output provider, and a new response schema can no longer
|
|
121
|
+
skip the rule.** The client sends a response schema `strict: true`, and
|
|
122
|
+
OpenAI refuses one whose objects leave a property out of `required`. The
|
|
123
|
+
consolidate plan left out `description` and `confidence`, and the session
|
|
124
|
+
summary `tags`, so the first request of every plan and every summary to such
|
|
125
|
+
a provider was refused. The client then retries without the schema and
|
|
126
|
+
remembers that per connection (endpoint and model), not per schema, so one
|
|
127
|
+
invalid schema also switched the response schema off for every valid one
|
|
128
|
+
that followed on that connection. Every property of both is now required: an
|
|
129
|
+
empty `description` keeps the memory's own, a null `confidence` records none
|
|
130
|
+
and an empty `tags` array is no tags, which is how all three were already
|
|
131
|
+
read. A contract test runs every schema akm sends through the rule, so a
|
|
132
|
+
property added later without being required fails CI.
|
|
133
|
+
|
|
9
134
|
## [0.9.26] - 2026-10-05
|
|
10
135
|
|
|
11
136
|
The stable release of the 0.9.26 line: 0.9.26-alpha.1 and alpha.2, and the
|
|
@@ -7,10 +7,10 @@ Rules:
|
|
|
7
7
|
Return ONLY JSON (no prose, no code fences):
|
|
8
8
|
{
|
|
9
9
|
"operations": [
|
|
10
|
-
{ "op": "promote", "ref": "memories/<name>", "knowledgeRef": "knowledge/<suggested-slug>", "reason": "<brief reason>", "description": "<one sentence describing the new knowledge asset>", "confidence": 0.92 }
|
|
10
|
+
{ "op": "promote", "ref": "memories/<name>", "knowledgeRef": "knowledge/<suggested-slug>", "reason": "<brief reason>", "description": "<one sentence describing the new knowledge asset, or an empty string to keep the memory's own>", "confidence": 0.92 }
|
|
11
11
|
]
|
|
12
12
|
}
|
|
13
13
|
|
|
14
|
-
For every operation, emit a `confidence` field in [0, 1] expressing your certainty that the operation is correct and safe. Use 0.95+ only when evidence is unambiguous.
|
|
14
|
+
For every operation, emit a `confidence` field in [0, 1] expressing your certainty that the operation is correct and safe. Use 0.95+ only when evidence is unambiguous. Use `null` rather than guessing if you are uncertain.
|
|
15
15
|
|
|
16
16
|
When the merged content includes an `updated` frontmatter field, the value MUST be a real ISO date string (e.g. `updated: 2026-05-20`). NEVER emit `updated: today`, `updated: {today}`, `updated: {today: null}`, `updated: now`, or any other literal placeholder/template-variable. If you do not have a real source-of-truth date, OMIT the `updated` field entirely — the post-processor will not invent one for you.
|
|
@@ -48,13 +48,13 @@ Respond with EXACTLY one JSON object matching this shape:
|
|
|
48
48
|
"type": "memory" | "lesson" | "knowledge",
|
|
49
49
|
"name": "<kebab-case name, e.g. jwt-token; optionally under one kebab-case scope, e.g. auth/jwt-token>",
|
|
50
50
|
"description": "<one sentence 20-400 chars>",
|
|
51
|
-
"when_to_use": "<one sentence 15-400 chars;
|
|
51
|
+
"when_to_use": "<one sentence 15-400 chars for a lesson; an empty string for a memory or knowledge candidate>",
|
|
52
52
|
"body": "<markdown body, 200-3000 chars typical>",
|
|
53
53
|
"confidence": <number 0.0-1.0>,
|
|
54
54
|
"evidence": "<one-line pointer to the moment in the session>"
|
|
55
55
|
}
|
|
56
56
|
],
|
|
57
|
-
"rationale_if_empty": "<one sentence
|
|
57
|
+
"rationale_if_empty": "<one sentence when candidates is empty; an empty string otherwise>"
|
|
58
58
|
}
|
|
59
59
|
```
|
|
60
60
|
|
|
@@ -93,7 +93,9 @@ export function isHotCapturedMemory(filePath) {
|
|
|
93
93
|
/**
|
|
94
94
|
* Structured-output schema for a plan. Promote-only: merge/delete/contradict
|
|
95
95
|
* were removed in 0.9.17-alpha.1 (`e82eec811`) after running in production —
|
|
96
|
-
* they cost thousands of completion tokens.
|
|
96
|
+
* they cost thousands of completion tokens. Every property is required, as a
|
|
97
|
+
* strict structured-output provider needs: an empty `description` keeps the
|
|
98
|
+
* memory's own, a null `confidence` is none.
|
|
97
99
|
*/
|
|
98
100
|
export const CONSOLIDATE_PLAN_JSON_SCHEMA = {
|
|
99
101
|
type: "object",
|
|
@@ -105,15 +107,23 @@ export const CONSOLIDATE_PLAN_JSON_SCHEMA = {
|
|
|
105
107
|
description: "Ordered list of promote operations the planner proposes.",
|
|
106
108
|
items: {
|
|
107
109
|
type: "object",
|
|
108
|
-
required: ["op", "ref", "knowledgeRef", "reason"],
|
|
110
|
+
required: ["op", "ref", "knowledgeRef", "reason", "description", "confidence"],
|
|
109
111
|
additionalProperties: false,
|
|
110
112
|
properties: {
|
|
111
113
|
op: { type: "string", enum: ["promote"] },
|
|
112
114
|
ref: { type: "string", minLength: 1 },
|
|
113
115
|
knowledgeRef: { type: "string", minLength: 1 },
|
|
114
116
|
reason: { type: "string", minLength: 1, maxLength: 200 },
|
|
115
|
-
description: {
|
|
116
|
-
|
|
117
|
+
description: {
|
|
118
|
+
type: "string",
|
|
119
|
+
description: "One sentence describing the new knowledge asset; an empty string keeps the memory's own.",
|
|
120
|
+
},
|
|
121
|
+
confidence: {
|
|
122
|
+
type: ["number", "null"],
|
|
123
|
+
minimum: 0,
|
|
124
|
+
maximum: 1,
|
|
125
|
+
description: "Certainty in [0, 1] that the operation is correct and safe; null when unsure.",
|
|
126
|
+
},
|
|
117
127
|
},
|
|
118
128
|
},
|
|
119
129
|
},
|
|
@@ -72,9 +72,13 @@ export function deriveLessonRef(inputRef) {
|
|
|
72
72
|
return `lessons/${safeScope ? `${safeScope}/` : ""}${clean(`${parsed.type}-${parts.join("-")}`)}-lesson`;
|
|
73
73
|
}
|
|
74
74
|
// ── Output contract ──────────────────────────────────────────────────────────
|
|
75
|
+
//
|
|
76
|
+
// The client sends a response schema `strict: true`, and a strict provider
|
|
77
|
+
// (OpenAI's) rejects an object whose `required` leaves out any of its
|
|
78
|
+
// properties, so every property is required and "none" is an empty array (#1046).
|
|
75
79
|
export const DISTILL_LESSON_JSON_SCHEMA = {
|
|
76
80
|
type: "object",
|
|
77
|
-
required: ["description", "when_to_use", "body"],
|
|
81
|
+
required: ["description", "when_to_use", "body", "tags"],
|
|
78
82
|
additionalProperties: false,
|
|
79
83
|
properties: {
|
|
80
84
|
description: {
|
|
@@ -95,13 +99,13 @@ export const DISTILL_LESSON_JSON_SCHEMA = {
|
|
|
95
99
|
tags: {
|
|
96
100
|
type: "array",
|
|
97
101
|
items: { type: "string" },
|
|
98
|
-
description: "
|
|
102
|
+
description: "Tag list. Use an empty array for none; the post-processor drops it if empty.",
|
|
99
103
|
},
|
|
100
104
|
},
|
|
101
105
|
};
|
|
102
106
|
export const DISTILL_KNOWLEDGE_JSON_SCHEMA = {
|
|
103
107
|
type: "object",
|
|
104
|
-
required: ["description", "body"],
|
|
108
|
+
required: ["description", "body", "tags", "sources"],
|
|
105
109
|
additionalProperties: false,
|
|
106
110
|
properties: {
|
|
107
111
|
description: { type: "string", minLength: 1, description: "One-line summary of the knowledge asset." },
|
|
@@ -113,12 +117,12 @@ export const DISTILL_KNOWLEDGE_JSON_SCHEMA = {
|
|
|
113
117
|
tags: {
|
|
114
118
|
type: "array",
|
|
115
119
|
items: { type: "string" },
|
|
116
|
-
description: "
|
|
120
|
+
description: "Tag list. Use an empty array for none; the post-processor drops it if empty.",
|
|
117
121
|
},
|
|
118
122
|
sources: {
|
|
119
123
|
type: "array",
|
|
120
124
|
items: { type: "string" },
|
|
121
|
-
description: "
|
|
125
|
+
description: "Source refs the knowledge was distilled from. Use an empty array for none.",
|
|
122
126
|
},
|
|
123
127
|
},
|
|
124
128
|
};
|
|
@@ -24,16 +24,21 @@ const EXTRACT_CANDIDATE_NAME_RE = new RegExp(EXTRACT_CANDIDATE_NAME_PATTERN);
|
|
|
24
24
|
*
|
|
25
25
|
* Shape:
|
|
26
26
|
* {
|
|
27
|
-
* "candidates": [{type, name, description, when_to_use
|
|
28
|
-
* "rationale_if_empty"
|
|
27
|
+
* "candidates": [{type, name, description, when_to_use, body, confidence, evidence}, ...],
|
|
28
|
+
* "rationale_if_empty": string
|
|
29
29
|
* }
|
|
30
30
|
*
|
|
31
31
|
* `additionalProperties: false` at each level so any hallucinated keys are
|
|
32
|
-
* dropped before parsing.
|
|
32
|
+
* dropped before parsing. Every property is required, as a strict structured-
|
|
33
|
+
* output provider needs (#1046), and an empty string stands for "none": a
|
|
34
|
+
* `when_to_use` only a lesson needs, a `rationale_if_empty` only an empty
|
|
35
|
+
* answer needs. Before, `when_to_use` was optional, so a model that followed
|
|
36
|
+
* the schema could leave it out of a lesson and the parser dropped the lesson
|
|
37
|
+
* (#1047).
|
|
33
38
|
*/
|
|
34
39
|
export const EXTRACT_JSON_SCHEMA = {
|
|
35
40
|
type: "object",
|
|
36
|
-
required: ["candidates"],
|
|
41
|
+
required: ["candidates", "rationale_if_empty"],
|
|
37
42
|
additionalProperties: false,
|
|
38
43
|
properties: {
|
|
39
44
|
candidates: {
|
|
@@ -41,7 +46,7 @@ export const EXTRACT_JSON_SCHEMA = {
|
|
|
41
46
|
description: "Zero or more durable-insight candidates extracted from the session.",
|
|
42
47
|
items: {
|
|
43
48
|
type: "object",
|
|
44
|
-
required: ["type", "name", "description", "body", "confidence", "evidence"],
|
|
49
|
+
required: ["type", "name", "description", "when_to_use", "body", "confidence", "evidence"],
|
|
45
50
|
additionalProperties: false,
|
|
46
51
|
properties: {
|
|
47
52
|
type: {
|
|
@@ -62,9 +67,8 @@ export const EXTRACT_JSON_SCHEMA = {
|
|
|
62
67
|
},
|
|
63
68
|
when_to_use: {
|
|
64
69
|
type: "string",
|
|
65
|
-
minLength: 15,
|
|
66
70
|
maxLength: 400,
|
|
67
|
-
description: "Trigger sentence
|
|
71
|
+
description: "Trigger sentence of at least 15 characters; REQUIRED when type=lesson (a lesson without one is dropped). An empty string for a memory or knowledge candidate.",
|
|
68
72
|
},
|
|
69
73
|
body: {
|
|
70
74
|
type: "string",
|
|
@@ -87,8 +91,7 @@ export const EXTRACT_JSON_SCHEMA = {
|
|
|
87
91
|
},
|
|
88
92
|
rationale_if_empty: {
|
|
89
93
|
type: "string",
|
|
90
|
-
|
|
91
|
-
description: "Required when `candidates` is empty — explains why nothing rose to durable-insight level.",
|
|
94
|
+
description: "When `candidates` is empty, one sentence on why nothing rose to durable-insight level; an empty string otherwise.",
|
|
92
95
|
},
|
|
93
96
|
},
|
|
94
97
|
};
|
|
@@ -214,10 +217,47 @@ function parseFirstJsonObject(stdout) {
|
|
|
214
217
|
}
|
|
215
218
|
return { objectFound: true };
|
|
216
219
|
}
|
|
220
|
+
/** The candidate the model wrote when the contract keeps it, else the first rule it breaks. */
|
|
221
|
+
function readCandidate(c) {
|
|
222
|
+
const { type, name, description, body, confidence, evidence } = c;
|
|
223
|
+
if (type !== "memory" && type !== "lesson" && type !== "knowledge") {
|
|
224
|
+
return { problem: "type is not memory, lesson or knowledge" };
|
|
225
|
+
}
|
|
226
|
+
if (typeof name !== "string" || !EXTRACT_CANDIDATE_NAME_RE.test(name)) {
|
|
227
|
+
return { problem: "name is not a kebab-case slug" };
|
|
228
|
+
}
|
|
229
|
+
if (typeof description !== "string" || description.trim().length < 20) {
|
|
230
|
+
return { problem: "description is shorter than 20 characters" };
|
|
231
|
+
}
|
|
232
|
+
if (typeof body !== "string" || body.trim().length < 50)
|
|
233
|
+
return { problem: "body is shorter than 50 characters" };
|
|
234
|
+
if (typeof confidence !== "number" || !Number.isFinite(confidence))
|
|
235
|
+
return { problem: "confidence is not a number" };
|
|
236
|
+
if (typeof evidence !== "string" || evidence.trim().length < 5) {
|
|
237
|
+
return { problem: "evidence is shorter than 5 characters" };
|
|
238
|
+
}
|
|
239
|
+
// An empty string is how a strict-schema reply says "none".
|
|
240
|
+
const whenToUse = typeof c.when_to_use === "string" ? c.when_to_use.trim() : "";
|
|
241
|
+
if (type === "lesson" && whenToUse.length < 15) {
|
|
242
|
+
return { problem: "a lesson needs a when_to_use of at least 15 characters" };
|
|
243
|
+
}
|
|
244
|
+
return {
|
|
245
|
+
candidate: {
|
|
246
|
+
type,
|
|
247
|
+
name,
|
|
248
|
+
description: description.trim(),
|
|
249
|
+
...(whenToUse ? { when_to_use: whenToUse } : {}),
|
|
250
|
+
body,
|
|
251
|
+
confidence: Math.max(0, Math.min(1, confidence)),
|
|
252
|
+
evidence: evidence.trim(),
|
|
253
|
+
},
|
|
254
|
+
};
|
|
255
|
+
}
|
|
217
256
|
/**
|
|
218
257
|
* Parse the LLM's JSON response into a structured {@link ExtractPayload}.
|
|
219
258
|
* Defensive — drops candidates that violate the shape rather than failing
|
|
220
|
-
* the whole call
|
|
259
|
+
* the whole call, and names each one in `dropped` so the run can say so.
|
|
260
|
+
* Returns the empty-candidates payload when nothing parses.
|
|
221
261
|
*/
|
|
222
262
|
export function parseExtractPayload(stdout) {
|
|
223
263
|
if (!stdout || stdout.trim().length === 0) {
|
|
@@ -238,42 +278,24 @@ export function parseExtractPayload(stdout) {
|
|
|
238
278
|
}
|
|
239
279
|
const rawCandidates = obj.candidates;
|
|
240
280
|
const candidates = [];
|
|
281
|
+
const dropped = [];
|
|
241
282
|
for (const raw of rawCandidates) {
|
|
242
|
-
if (!raw || typeof raw !== "object")
|
|
283
|
+
if (!raw || typeof raw !== "object") {
|
|
284
|
+
dropped.push("candidate dropped: not an object");
|
|
243
285
|
continue;
|
|
286
|
+
}
|
|
244
287
|
const c = raw;
|
|
245
|
-
const
|
|
246
|
-
if (
|
|
247
|
-
|
|
248
|
-
if (typeof c.name !== "string" || !EXTRACT_CANDIDATE_NAME_RE.test(c.name))
|
|
249
|
-
continue;
|
|
250
|
-
if (typeof c.description !== "string" || c.description.trim().length < 20)
|
|
251
|
-
continue;
|
|
252
|
-
if (typeof c.body !== "string" || c.body.trim().length < 50)
|
|
288
|
+
const read = readCandidate(c);
|
|
289
|
+
if ("problem" in read) {
|
|
290
|
+
dropped.push(`${typeof c.type === "string" ? c.type : "candidate"}:${typeof c.name === "string" ? c.name : "(unnamed)"} dropped: ${read.problem}`);
|
|
253
291
|
continue;
|
|
254
|
-
if (typeof c.confidence !== "number" || !Number.isFinite(c.confidence))
|
|
255
|
-
continue;
|
|
256
|
-
if (typeof c.evidence !== "string" || c.evidence.trim().length < 5)
|
|
257
|
-
continue;
|
|
258
|
-
if (type === "lesson") {
|
|
259
|
-
if (typeof c.when_to_use !== "string" || c.when_to_use.trim().length < 15)
|
|
260
|
-
continue;
|
|
261
292
|
}
|
|
262
|
-
|
|
263
|
-
const candidate = {
|
|
264
|
-
type,
|
|
265
|
-
name: c.name,
|
|
266
|
-
description: c.description.trim(),
|
|
267
|
-
body: c.body,
|
|
268
|
-
confidence,
|
|
269
|
-
evidence: c.evidence.trim(),
|
|
270
|
-
};
|
|
271
|
-
if (typeof c.when_to_use === "string")
|
|
272
|
-
candidate.when_to_use = c.when_to_use.trim();
|
|
273
|
-
candidates.push(candidate);
|
|
293
|
+
candidates.push(read.candidate);
|
|
274
294
|
}
|
|
275
295
|
const result = { candidates };
|
|
276
|
-
if (
|
|
296
|
+
if (dropped.length > 0)
|
|
297
|
+
result.dropped = dropped;
|
|
298
|
+
if (typeof obj.rationale_if_empty === "string" && obj.rationale_if_empty.trim()) {
|
|
277
299
|
result.rationale_if_empty = obj.rationale_if_empty.trim();
|
|
278
300
|
}
|
|
279
301
|
return result;
|
|
@@ -468,7 +468,8 @@ async function processSession(run, sessionRef, gate) {
|
|
|
468
468
|
});
|
|
469
469
|
}
|
|
470
470
|
const { payload } = extraction;
|
|
471
|
-
|
|
471
|
+
// A candidate the contract refused is reported, not silently lost (#1047).
|
|
472
|
+
const warnings = [...(payload.dropped ?? [])];
|
|
472
473
|
// Provenance xrefs are added only after the cited session asset exists.
|
|
473
474
|
const { warning, ...sessionAsset } = await maybeWriteSessionAsset(run, data);
|
|
474
475
|
if (warning)
|
|
@@ -13,11 +13,13 @@
|
|
|
13
13
|
* pre-dispatch refusals still emit both).
|
|
14
14
|
*/
|
|
15
15
|
import fs from "node:fs";
|
|
16
|
+
import path from "node:path";
|
|
17
|
+
import { assetPathForName, stashDirFor } from "../../core/asset/asset-placement.js";
|
|
16
18
|
import { serializeFrontmatter } from "../../core/asset/asset-serialize.js";
|
|
17
19
|
import { parseFrontmatter } from "../../core/asset/frontmatter.js";
|
|
18
20
|
import { parseRefInput } from "../../core/asset/resolve-ref.js";
|
|
19
21
|
import { DESCRIPTION_MAX_CHARS, requiresDescription } from "../../core/authoring-rules.js";
|
|
20
|
-
import { resolveStashDir } from "../../core/common.js";
|
|
22
|
+
import { isWithin, resolveStashDir, safeRealpath } from "../../core/common.js";
|
|
21
23
|
import { loadConfig } from "../../core/config/config.js";
|
|
22
24
|
import { generatedContentRejection } from "../../core/content-safety.js";
|
|
23
25
|
import { ConfigError, UsageError } from "../../core/errors.js";
|
|
@@ -248,7 +250,7 @@ export const REFLECT_JSON_SCHEMA = {
|
|
|
248
250
|
frontmatterPatch: REFLECT_FRONTMATTER_PATCH_JSON_SCHEMA,
|
|
249
251
|
},
|
|
250
252
|
};
|
|
251
|
-
const REFLECT_UNSCOPED_JSON_SCHEMA = {
|
|
253
|
+
export const REFLECT_UNSCOPED_JSON_SCHEMA = {
|
|
252
254
|
type: "object",
|
|
253
255
|
required: ["ref", "confidence", "frontmatterPatch"],
|
|
254
256
|
additionalProperties: false,
|
|
@@ -539,6 +541,24 @@ function unsupportedTypeFailure(ref, type, detail, emitFailed) {
|
|
|
539
541
|
},
|
|
540
542
|
};
|
|
541
543
|
}
|
|
544
|
+
/**
|
|
545
|
+
* A proposal writes the file derived from the ref's type and name under the bundle's root. An asset indexed
|
|
546
|
+
* anywhere else (a skill's `references/a.md` is `knowledge/skills/<name>/references/a`) has nothing there, so the
|
|
547
|
+
* proposal would be a `create` and accepting it would add a second file for the ref (#1052).
|
|
548
|
+
*/
|
|
549
|
+
function fileOutsideLayoutFailure(ref, file, writes, emitFailed) {
|
|
550
|
+
emitFailed("unsupported_type", "file_outside_layout", ref);
|
|
551
|
+
return {
|
|
552
|
+
failure: {
|
|
553
|
+
schemaVersion: 2,
|
|
554
|
+
ok: false,
|
|
555
|
+
reason: "unsupported_type",
|
|
556
|
+
error: `Reflect refused: the file for ${ref} is ${file}, but a proposal would write ${writes}. Edit the file directly.`,
|
|
557
|
+
ref,
|
|
558
|
+
exitCode: null,
|
|
559
|
+
},
|
|
560
|
+
};
|
|
561
|
+
}
|
|
542
562
|
/** The target's parsed ref and current content, or a refusal for a type reflect cannot patch. */
|
|
543
563
|
async function resolveReflectSource(options, stash, emitFailed) {
|
|
544
564
|
if (!options.ref)
|
|
@@ -549,18 +569,21 @@ async function resolveReflectSource(options, stash, emitFailed) {
|
|
|
549
569
|
return unsupportedTypeFailure(options.ref, parsedRef.type, "secret material is never read or sent to an LLM", emitFailed);
|
|
550
570
|
}
|
|
551
571
|
let assetContent = options.assetContent;
|
|
572
|
+
let assetFile;
|
|
552
573
|
if (assetContent === undefined) {
|
|
553
574
|
try {
|
|
554
575
|
const qualifiedRef = options.itemRef ?? options.ref;
|
|
555
576
|
const localFilePath = await findAssetFilePath(qualifiedRef, stash);
|
|
556
577
|
if (localFilePath && fs.existsSync(localFilePath)) {
|
|
557
|
-
|
|
578
|
+
assetFile = localFilePath;
|
|
558
579
|
}
|
|
559
580
|
else {
|
|
560
581
|
const entry = await lookup(parseRefInput(qualifiedRef));
|
|
561
582
|
if (entry?.filePath && fs.existsSync(entry.filePath))
|
|
562
|
-
|
|
583
|
+
assetFile = entry.filePath;
|
|
563
584
|
}
|
|
585
|
+
if (assetFile !== undefined)
|
|
586
|
+
assetContent = fs.readFileSync(assetFile, "utf8");
|
|
564
587
|
}
|
|
565
588
|
catch {
|
|
566
589
|
// An index miss is not fatal: reflect then has no content to patch.
|
|
@@ -570,6 +593,15 @@ async function resolveReflectSource(options, stash, emitFailed) {
|
|
|
570
593
|
(assetContent === undefined || parseFrontmatter(assetContent).frontmatter === null)) {
|
|
571
594
|
return unsupportedTypeFailure(options.ref, parsedRef.type, "its content is not frontmatter + markdown", emitFailed);
|
|
572
595
|
}
|
|
596
|
+
// A file in another bundle is `createProposal`'s to refuse (#1000); this one is in the proposal's own bundle.
|
|
597
|
+
const root = path.resolve(options.target?.root ?? stash);
|
|
598
|
+
const typeDir = stashDirFor(parsedRef.type);
|
|
599
|
+
if (assetFile !== undefined && typeDir !== undefined && isWithin(assetFile, root)) {
|
|
600
|
+
const writes = assetPathForName(parsedRef.type, path.join(root, typeDir), parsedRef.name);
|
|
601
|
+
if (safeRealpath(writes) !== safeRealpath(assetFile)) {
|
|
602
|
+
return fileOutsideLayoutFailure(options.ref, assetFile, writes, emitFailed);
|
|
603
|
+
}
|
|
604
|
+
}
|
|
573
605
|
return { assetContent, parsedRef };
|
|
574
606
|
}
|
|
575
607
|
/**
|
|
@@ -26,7 +26,7 @@ const MAX_QUERIES = 5;
|
|
|
26
26
|
const MAX_QUERY_CHARS = 2000;
|
|
27
27
|
/** The judge sees this much of the body, as in the retrieval eval. */
|
|
28
28
|
const MAX_DOC_CHARS = 1500;
|
|
29
|
-
const GRADE_SCHEMA = {
|
|
29
|
+
export const GRADE_SCHEMA = {
|
|
30
30
|
type: "object",
|
|
31
31
|
required: ["grade", "reason"],
|
|
32
32
|
additionalProperties: false,
|
|
@@ -14,9 +14,10 @@ import { assembleAsset } from "../../core/asset/asset-serialize.js";
|
|
|
14
14
|
import { conceptIdFromTypeName } from "../../core/asset/resolve-ref.js";
|
|
15
15
|
import { parseEmbeddedJsonResponse } from "../../core/parse.js";
|
|
16
16
|
import { recordWrittenPath } from "../../core/write-provenance.js";
|
|
17
|
+
// Every property is required, as a strict structured-output provider needs; an empty `tags` array is none.
|
|
17
18
|
export const SESSION_SUMMARY_JSON_SCHEMA = {
|
|
18
19
|
type: "object",
|
|
19
|
-
required: ["summary", "key_topics"],
|
|
20
|
+
required: ["summary", "key_topics", "tags"],
|
|
20
21
|
additionalProperties: false,
|
|
21
22
|
properties: {
|
|
22
23
|
summary: { type: "string" },
|
|
@@ -61,7 +62,7 @@ export function buildSessionSummaryPrompt(data) {
|
|
|
61
62
|
"Transcript:",
|
|
62
63
|
renderTranscriptForSummary(data.events),
|
|
63
64
|
"",
|
|
64
|
-
'Respond as JSON: {"summary": string, "key_topics": string[], "tags"
|
|
65
|
+
'Respond as JSON: {"summary": string, "key_topics": string[], "tags": string[]} (an empty array for no tags).',
|
|
65
66
|
].join("\n");
|
|
66
67
|
}
|
|
67
68
|
/** The summary JSON, tolerating prose around it; `undefined` when nothing usable parses. */
|
|
@@ -370,7 +370,7 @@ function parseJudgeResponse(raw, keys) {
|
|
|
370
370
|
}
|
|
371
371
|
return inRange(parsed.score) ? { score: parsed.score, lowest: parsed.score, reason } : undefined;
|
|
372
372
|
}
|
|
373
|
-
function judgeResponseSchema(keys) {
|
|
373
|
+
export function judgeResponseSchema(keys) {
|
|
374
374
|
return {
|
|
375
375
|
type: "object",
|
|
376
376
|
required: ["scores", "reason"],
|