akm-cli 0.9.20 → 0.9.22
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +113 -1
- package/dist/assets/hints/cli-hints-full.md +9 -4
- package/dist/assets/hints/cli-hints-short.md +7 -4
- package/dist/assets/improve-strategies/default.json +2 -2
- package/dist/assets/improve-strategies/proactive-maintenance.json +1 -1
- package/dist/assets/improve-strategies/thorough.json +1 -1
- package/dist/assets/stash-skeleton/README.md +5 -1
- package/dist/commands/feedback-cli.js +14 -6
- package/dist/commands/health/improve-metrics.js +1 -1
- package/dist/commands/improve/distill.js +3 -1
- package/dist/commands/improve/improve-cli.js +1 -1
- package/dist/commands/improve/improve.js +3 -27
- package/dist/commands/improve/preparation.js +35 -29
- package/dist/commands/improve/proactive-maintenance.js +2 -9
- package/dist/commands/improve/reflect.js +3 -2
- package/dist/commands/improve/stage.js +36 -31
- package/dist/commands/proposal/repository.js +14 -11
- package/dist/commands/proposal/validators/proposals.js +8 -2
- package/dist/commands/remember.js +7 -1
- package/dist/core/improve-result.js +4 -1
- package/dist/core/write-source.js +28 -5
- package/dist/scripts/akm-migrate-node.js +4 -4
- package/dist/scripts/akm-migrate.js +4 -4
- package/dist/sources/providers/git-stash.js +16 -9
- package/dist/storage/repositories/improve-runs-repository.js +1 -3
- package/docs/reference/cli.md +41 -15
- package/package.json +1 -1
package/CHANGELOG.md
CHANGED
|
@@ -4,7 +4,119 @@ All notable changes to this project will be documented in this file.
|
|
|
4
4
|
|
|
5
5
|
The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/).
|
|
6
6
|
|
|
7
|
-
## [
|
|
7
|
+
## [0.9.22] - 2026-10-01
|
|
8
|
+
|
|
9
|
+
### Fixed
|
|
10
|
+
|
|
11
|
+
- **Accepting a proposal no longer rewrites a long description at its first line
|
|
12
|
+
break.** `yaml.stringify`, which reflect and the other writers serialize
|
|
13
|
+
frontmatter with, wraps a description past about 80 columns over indented
|
|
14
|
+
lines. The truncation repair that `akm proposal accept` runs before it
|
|
15
|
+
promotes read only the first of those lines. When that line ended in a
|
|
16
|
+
connector word (`the`, `and`) or a comma it took the wrap for a truncation,
|
|
17
|
+
dropped the tail words, added a period and left the continuation lines
|
|
18
|
+
behind: `...the cooldowns that drive the` followed by `rest of the pipeline`
|
|
19
|
+
was written as `...the cooldowns that drive.` followed by `rest of the
|
|
20
|
+
pipeline`. Plain, single-quoted and double-quoted descriptions were all
|
|
21
|
+
damaged this way, and the repair has behaved so since it shipped in
|
|
22
|
+
0.9.0-beta.36 (#645). It now runs only on a single-line description and
|
|
23
|
+
leaves a wrapped one exactly as proposed. Assets already promoted with a
|
|
24
|
+
damaged description are not repaired by this change (the dropped words and the
|
|
25
|
+
stray period stay in them), so they need a separate repair.
|
|
26
|
+
- **The same repair no longer rewrites a body line or another key.** It looked
|
|
27
|
+
for `description:` in the whole file, and its `\s*` ran past the end of the
|
|
28
|
+
line. A file whose frontmatter had no description had a body line starting
|
|
29
|
+
with `description:` completed instead, and an empty `description:` had the
|
|
30
|
+
next key's line rewritten (`title: Notes about how we configure the` became
|
|
31
|
+
`title: Notes about how we configure.`). It now reads only the `description:`
|
|
32
|
+
line of the frontmatter block. Content already promoted that way is not
|
|
33
|
+
repaired by this change.
|
|
34
|
+
|
|
35
|
+
## [0.9.21] - 2026-10-01
|
|
36
|
+
|
|
37
|
+
### Changed
|
|
38
|
+
|
|
39
|
+
- **`akm improve` rewrites an asset only from negative feedback.** Reflect's
|
|
40
|
+
signal delta reads negative feedback only, so an asset whose recent feedback
|
|
41
|
+
is positive or a note is no longer planned for a rewrite; any signal planned
|
|
42
|
+
one before. `akm feedback <ref> --negative --reason "<what is wrong and what
|
|
43
|
+
should change>"` flags the asset for review, and the next improve run
|
|
44
|
+
proposes a fix based on the reason. `--positive` records that the asset
|
|
45
|
+
helped (it raises its ranking) and never triggers a rewrite. An explicit ref
|
|
46
|
+
(`akm improve skills/x`) still plans one, and distill still reads any recent
|
|
47
|
+
signal on a memory.
|
|
48
|
+
- **The fallback lanes score assets and no longer plan them.** High salience,
|
|
49
|
+
and proactive maintenance in a strategy that enables it (the shipped
|
|
50
|
+
`proactive-maintenance` strategy), still pick assets and score them
|
|
51
|
+
(salience and outcome), but a pick no longer enters the loop, so nothing is
|
|
52
|
+
reflected or distilled for it and improve no longer rewrites assets on a
|
|
53
|
+
proactive cadence. `eligibilitySource` is now `signal-delta` or `scope`;
|
|
54
|
+
`proactive` and `high-salience` stay valid on the rows an older release wrote.
|
|
55
|
+
`--require-feedback-signal` still turns the lanes off.
|
|
56
|
+
- **`distill.requirePlannedRefs` is `false` in the `default` and `thorough`
|
|
57
|
+
strategies** (the other presets inherit it from `default`). With reflect
|
|
58
|
+
planned only from negative feedback, `true` would have skipped distill on
|
|
59
|
+
every run where no ref has negative feedback; distill now keeps working
|
|
60
|
+
through memories with fresh feedback on each run. A strategy that sets it
|
|
61
|
+
`true` still skips distill when no ref is planned for reflect.
|
|
62
|
+
- **`akm feedback`'s help and errors, both CLI hints, the stash README and the
|
|
63
|
+
docs say what each signal does:** `--negative --reason` flags the asset for
|
|
64
|
+
review and improve proposes a fix from the reason, so the reason should say
|
|
65
|
+
what is wrong and what should change; `--positive` raises the ranking and
|
|
66
|
+
does not trigger a rewrite. They also say that improve no longer rewrites on
|
|
67
|
+
a proactive cadence or from positive signals. The `default` and
|
|
68
|
+
`proactive-maintenance` strategy descriptions match.
|
|
69
|
+
- **The reflect judge asks whether a rewrite is needed, and both judges pass
|
|
70
|
+
only when every criterion scores 4 or more.** Judging on the mean (3.5 or
|
|
71
|
+
more) let rewrites that only reworded a correct asset through. The reflect
|
|
72
|
+
judge's criteria are now **need** (does the revision fix a concrete problem
|
|
73
|
+
in the source: something the feedback reports, a factual error, or broken,
|
|
74
|
+
garbled, truncated or missing text, frontmatter fields included; rewording,
|
|
75
|
+
restating, reformatting or adding headings scores 1 or 2), **preservation**
|
|
76
|
+
and **quality**, replacing feedback alignment, preservation and quality, and
|
|
77
|
+
the "overlap with the source is expected" line is gone. Its JSON keys are
|
|
78
|
+
`need`, `preservation` and `quality`. The lesson judge passes only when
|
|
79
|
+
novelty and non-redundancy both score 4 or more. A verdict that does not
|
|
80
|
+
pass keeps its routing (a mean of 2.5 or more is a review, below that a
|
|
81
|
+
rejection) and the grounding rules are unchanged, grounding staying outside
|
|
82
|
+
the mean and the pass rule.
|
|
83
|
+
- **A quality-judge pass keeps its evidence.** The `staged` decision the
|
|
84
|
+
quality gate stamps on a proposal now carries the per-criterion `scores` and
|
|
85
|
+
the judge's `judgeReason` (`gateDecision.scores`, `gateDecision.judgeReason`
|
|
86
|
+
in `akm proposal show --format json`), and they stay on the proposal when the
|
|
87
|
+
drain accepts it, so a later audit can read why a rewrite passed.
|
|
88
|
+
- **Each accepted proposal is committed as it happens when its bundle is a git
|
|
89
|
+
repository.** Manual `akm proposal accept`, `akm proposal drain`, the improve
|
|
90
|
+
triage pre-pass and judgment accepts all commit exactly the paths the accept
|
|
91
|
+
wrote or removed, a retirement's archived copy and tombstone and the source
|
|
92
|
+
memory a consolidate promotion retires included, as one local commit
|
|
93
|
+
`akm accept: <generator> <proposal-id-8> <ref>`. Before, only a `kind: "git"`
|
|
94
|
+
bundle committed at accept; an accept into a filesystem bundle with a `.git`
|
|
95
|
+
directory (the working bundle `akm init` creates is one) stayed uncommitted
|
|
96
|
+
until a sync, and a standalone `akm proposal accept` never committed it. A
|
|
97
|
+
commit that fails warns and the accept stands. A non-git bundle is
|
|
98
|
+
unchanged.
|
|
99
|
+
- **The end-of-run sync and `akm sync` commit even when they cannot push.** A
|
|
100
|
+
branch with no upstream, or behind or diverged from it, used to fail before
|
|
101
|
+
committing, leaving the run's changes in the working tree. They are now
|
|
102
|
+
committed and only the push is skipped, with `not pushed: ...` as the result's
|
|
103
|
+
`reason`. A branch ahead of its upstream (an accept commits locally) is pushed
|
|
104
|
+
along with the sync commit, where it used to be refused.
|
|
105
|
+
|
|
106
|
+
### Fixed
|
|
107
|
+
|
|
108
|
+
- **A dotted token no longer splits a synthesized description.** `akm
|
|
109
|
+
remember` ended a sentence at every `.`, so `192.168.0.203` became `192. 168.
|
|
110
|
+
0. 203` and `0.9.12` became `0. 9. 12`. A `.`, `!` or `?` now ends a sentence
|
|
111
|
+
only when whitespace or the end of the text follows it.
|
|
112
|
+
- **The nightly commit message's `{accepted}` and the run metrics count the
|
|
113
|
+
proposals triage promoted.** `{accepted}`, `autoAcceptedCount` in
|
|
114
|
+
`improve_runs.metrics_json` and `improve.autoAccept.promoted` in `akm health`
|
|
115
|
+
read `gateAutoAcceptedCount`, which nothing has set since the confidence gate
|
|
116
|
+
was deleted, so they were always 0. They read the triage pre-pass's promoted
|
|
117
|
+
count, and the dead field is gone from the result type. A result row an older
|
|
118
|
+
akm wrote with the field still decodes in `akm health`: the decoder accepts
|
|
119
|
+
the key and ignores its value.
|
|
8
120
|
|
|
9
121
|
## [0.9.20] - 2026-09-30
|
|
10
122
|
|
|
@@ -94,15 +94,20 @@ akm import ./doc.md --target my-other-bundle # Route import to a named writab
|
|
|
94
94
|
akm workflow create ship-release # Create a workflow asset in the bundle
|
|
95
95
|
akm lint --type workflows # Parse and compile every .md/.yml workflow source; list every error
|
|
96
96
|
akm workflow run workflows/ship-release # Start or resume and execute the workflow
|
|
97
|
-
akm feedback skills/code-review --positive # Record that an asset helped
|
|
98
|
-
akm feedback agents/reviewer --negative --reason "wrong framework" #
|
|
97
|
+
akm feedback skills/code-review --positive # Record that an asset helped (ranks it higher; no rewrite)
|
|
98
|
+
akm feedback agents/reviewer --negative --reason "wrong framework" # Flag it for review: improve proposes a fix from the reason
|
|
99
99
|
akm feedback memories/deployment-notes --positive # Works for memories too
|
|
100
100
|
akm feedback env/prod --positive # Records env feedback without surfacing values
|
|
101
101
|
```
|
|
102
102
|
|
|
103
103
|
Use `akm feedback` whenever an asset's content materially helps, or proves wrong,
|
|
104
|
-
stale or unhelpful, so future search ranking can learn from actual usage.
|
|
105
|
-
|
|
104
|
+
stale or unhelpful, so future search ranking can learn from actual usage.
|
|
105
|
+
`akm feedback <ref> --negative --reason "<what is wrong and what should change>"`
|
|
106
|
+
flags the asset for review: the next improve run proposes a fix based on your
|
|
107
|
+
reason, so be specific. `--positive` records that an asset helped (it raises its
|
|
108
|
+
ranking) and does not trigger a rewrite; improve no longer rewrites assets from
|
|
109
|
+
positive signals or on a proactive cadence. An akm command that fails says
|
|
110
|
+
nothing about the asset; don't record it as feedback.
|
|
106
111
|
|
|
107
112
|
## LLM Wiki bundles
|
|
108
113
|
|
|
@@ -8,7 +8,7 @@ For any task, follow this loop:
|
|
|
8
8
|
1. `akm curate "<task>"` — find the best matching asset
|
|
9
9
|
2. `akm show <ref>` — read the schema (field names and structure)
|
|
10
10
|
3. Edit the workspace file using schema field names + task-specific values from your README
|
|
11
|
-
4. `akm feedback <ref> --positive` — record that the asset helped
|
|
11
|
+
4. `akm feedback <ref> --positive` — record that the asset helped (it raises its ranking and does not trigger a rewrite); when its content was wrong, stale or unhelpful, `akm feedback <ref> --negative --reason "<what is wrong and what should change>"` flags it for review: the next improve run proposes a fix based on your reason, so be specific. A failed akm command (e.g. `akm show` erroring) is not feedback on the asset — don't record it.
|
|
12
12
|
|
|
13
13
|
For workflow tasks:
|
|
14
14
|
1. `akm show workflows/<name>` — inspect the procedure before executing it
|
|
@@ -39,7 +39,8 @@ akm import ./doc.md --target my-bundle # Route import to a named writabl
|
|
|
39
39
|
akm proposal diff skills/akm-dream # Diff proposal by ref, UUID, or 8-char prefix
|
|
40
40
|
akm proposal accept 7c115132 # Accept by UUID prefix
|
|
41
41
|
akm proposal reject skills/my-skill --reason "..." # Reject by ref
|
|
42
|
-
akm feedback <ref> --positive
|
|
42
|
+
akm feedback <ref> --positive # Record that an asset helped (ranks it higher; no rewrite)
|
|
43
|
+
akm feedback <ref> --negative --reason "..." # Flag it for review: the next improve run proposes a fix from your reason
|
|
43
44
|
akm bundle add <ref> # Add a source (npm, GitHub, git, local dir)
|
|
44
45
|
akm clone <ref> # Copy an asset to the working bundle (optional --dest arg to clone to specific location)
|
|
45
46
|
akm sync # Commit (and push if writable remote) changes in the primary bundle (--no-push to commit only)
|
|
@@ -65,8 +66,10 @@ akm search "<query>" --from registry # Search all registries (registry
|
|
|
65
66
|
|
|
66
67
|
When an asset's content meaningfully helps, or proves wrong, stale or unhelpful,
|
|
67
68
|
record that with `akm feedback` so future search ranking can learn from real
|
|
68
|
-
usage.
|
|
69
|
-
|
|
69
|
+
usage. Only negative feedback with a specific reason gets the asset reviewed and
|
|
70
|
+
fixed: improve no longer rewrites assets from positive signals or on a
|
|
71
|
+
proactive cadence. An akm command that fails says nothing about the asset;
|
|
72
|
+
don't record it as feedback.
|
|
70
73
|
|
|
71
74
|
## Error Shapes and Exit Codes
|
|
72
75
|
|
|
@@ -1,12 +1,12 @@
|
|
|
1
1
|
{
|
|
2
|
-
"description": "Standard improve pass — reflect, distill, consolidation (promotion plus the reviewed pair-pass retire/supersede proposals), and validation. Memory inference is listed below but only runs when experimental.improveAutonomy is set; improve-stage extract and proactive maintenance off.",
|
|
2
|
+
"description": "Standard improve pass — reflect (rewrites only from negative feedback), distill, consolidation (promotion plus the reviewed pair-pass retire/supersede proposals), and validation. Memory inference is listed below but only runs when experimental.improveAutonomy is set; improve-stage extract and proactive maintenance off.",
|
|
3
3
|
"processes": {
|
|
4
4
|
"reflect": {
|
|
5
5
|
"enabled": true,
|
|
6
6
|
"limit": 25,
|
|
7
7
|
"allowedTypes": ["agent", "command", "knowledge", "lesson", "memory", "skill", "workflow"]
|
|
8
8
|
},
|
|
9
|
-
"distill": { "enabled": true, "allowedTypes": ["memory"], "requirePlannedRefs":
|
|
9
|
+
"distill": { "enabled": true, "allowedTypes": ["memory"], "requirePlannedRefs": false },
|
|
10
10
|
"consolidate": { "enabled": true, "allowedTypes": ["memory"] },
|
|
11
11
|
"memoryInference": { "enabled": true },
|
|
12
12
|
"extract": { "enabled": false, "triage": { "enabled": true, "minScore": 2 } },
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
{
|
|
2
|
-
"description": "Opt-in proactive-maintenance pass — reflect, distill, proposal triage (promote, high budget), and the proactive-maintenance lane (maxPerRun 100); consolidate/memoryInference/extract off. Sync disabled: an interrupted run would otherwise leave an uncommitted backlog.",
|
|
2
|
+
"description": "Opt-in proactive-maintenance pass — reflect (on negative feedback), distill, proposal triage (promote, high budget), and the proactive-maintenance lane (maxPerRun 100), which picks due assets for scoring but plans no rewrite; consolidate/memoryInference/extract off. Sync disabled: an interrupted run would otherwise leave an uncommitted backlog.",
|
|
3
3
|
"processes": {
|
|
4
4
|
"reflect": {
|
|
5
5
|
"enabled": true,
|
|
@@ -77,9 +77,13 @@ akm search "<query>" --type skill
|
|
|
77
77
|
### Recording feedback and new knowledge
|
|
78
78
|
|
|
79
79
|
```sh
|
|
80
|
-
# Mark an asset as helpful (
|
|
80
|
+
# Mark an asset as helpful (raises its ranking; does not trigger a rewrite)
|
|
81
81
|
akm feedback <ref> --positive
|
|
82
82
|
|
|
83
|
+
# Flag an asset for review: the next improve run proposes a fix based on your
|
|
84
|
+
# reason, so be specific about what is wrong and what should change
|
|
85
|
+
akm feedback <ref> --negative --reason "<what is wrong and what should change>"
|
|
86
|
+
|
|
83
87
|
# Capture a durable lesson or memory from the current session
|
|
84
88
|
akm remember "<fact or lesson>"
|
|
85
89
|
```
|
|
@@ -190,6 +190,10 @@ export const feedbackCommand = defineJsonCommand({
|
|
|
190
190
|
meta: {
|
|
191
191
|
name: "feedback",
|
|
192
192
|
description: "Record positive or negative feedback for any indexed bundle asset.\n\n" +
|
|
193
|
+
'`akm feedback <ref> --negative --reason "<what is wrong and what should change>"` flags\n' +
|
|
194
|
+
"the asset for review: the next improve run proposes a fix based on your reason, so be\n" +
|
|
195
|
+
"specific. `--positive` records that an asset helped (it raises its ranking) and does not\n" +
|
|
196
|
+
"trigger a rewrite.\n\n" +
|
|
193
197
|
"Both signals adjust the asset's usefulness score right away, in the same\n" +
|
|
194
198
|
"process: positive feedback raises it, negative lowers it, and recent\n" +
|
|
195
199
|
"feedback counts for more than old feedback. No reindex is needed — the new\n" +
|
|
@@ -200,15 +204,19 @@ export const feedbackCommand = defineJsonCommand({
|
|
|
200
204
|
// and throw a structured UsageError below so exit code is 2 (USAGE) rather
|
|
201
205
|
// than citty's default 0 (help banner).
|
|
202
206
|
ref: { type: "positional", description: "Asset ref ([bundle//]conceptId, e.g. lessons/deploy)", required: false },
|
|
203
|
-
positive: {
|
|
207
|
+
positive: {
|
|
208
|
+
type: "boolean",
|
|
209
|
+
description: "Record that the asset helped (raises its ranking immediately; does not trigger a rewrite)",
|
|
210
|
+
default: false,
|
|
211
|
+
},
|
|
204
212
|
negative: {
|
|
205
213
|
type: "boolean",
|
|
206
|
-
description: "
|
|
214
|
+
description: "Flag the asset for review: the next improve run proposes a fix from --reason (also lowers its ranking immediately, no reindex needed).",
|
|
207
215
|
default: false,
|
|
208
216
|
},
|
|
209
217
|
reason: {
|
|
210
218
|
type: "string",
|
|
211
|
-
description: "What
|
|
219
|
+
description: "What is wrong with the asset's content and what should change; the next improve run proposes a fix from it, so be specific (required for negative feedback by default). Not for akm command errors.",
|
|
212
220
|
},
|
|
213
221
|
"failure-mode": {
|
|
214
222
|
type: "string",
|
|
@@ -260,12 +268,12 @@ export const feedbackCommand = defineJsonCommand({
|
|
|
260
268
|
const cfg = loadConfig();
|
|
261
269
|
const requireReason = cfg.feedback?.requireReason ?? true; // Default: true (F-3 / #384)
|
|
262
270
|
if (requireReason) {
|
|
263
|
-
throw new UsageError("Negative feedback requires --reason
|
|
271
|
+
throw new UsageError("Negative feedback requires --reason: the next improve run proposes a fix from it, so say what is wrong and what should change. " +
|
|
264
272
|
"Use --failure-mode for a curated taxonomy or --reason for free text. " +
|
|
265
|
-
"Set feedback.requireReason: false in akm.json to downgrade to a warning.", "MISSING_REQUIRED_ARGUMENT", `Hint: akm feedback ${ref} --negative --reason "
|
|
273
|
+
"Set feedback.requireReason: false in akm.json to downgrade to a warning.", "MISSING_REQUIRED_ARGUMENT", `Hint: akm feedback ${ref} --negative --reason "<what is wrong and what should change>" [--failure-mode incorrect|outdated|dangerous|incomplete|redundant]`);
|
|
266
274
|
}
|
|
267
275
|
else {
|
|
268
|
-
warn("Warning: negative feedback without --reason
|
|
276
|
+
warn("Warning: negative feedback without --reason gives the next improve run nothing to base a fix on.");
|
|
269
277
|
}
|
|
270
278
|
}
|
|
271
279
|
const rawTags = parseAllFlagValues("--tag");
|
|
@@ -190,7 +190,7 @@ function projectRunMetrics(result) {
|
|
|
190
190
|
(metrics.actions.distill.skippedByReason[reason] ?? 0) + toFiniteNumber(count);
|
|
191
191
|
}
|
|
192
192
|
}
|
|
193
|
-
metrics.autoAccept.promoted += toFiniteNumber(result.
|
|
193
|
+
metrics.autoAccept.promoted += toFiniteNumber(result.triage?.promoted);
|
|
194
194
|
metrics.autoAccept.validationFailed += toFiniteNumber(result.gateAutoAcceptFailedCount);
|
|
195
195
|
const memorySummary = result.memorySummary;
|
|
196
196
|
if (memorySummary) {
|
|
@@ -434,6 +434,7 @@ function qualityGateEnabled(run) {
|
|
|
434
434
|
async function judgeAndQueue(run, out) {
|
|
435
435
|
let content = out.content;
|
|
436
436
|
let confidence;
|
|
437
|
+
let judged;
|
|
437
438
|
if (qualityGateEnabled(run)) {
|
|
438
439
|
const similarLessons = await run.similar(content.slice(0, 500), 3);
|
|
439
440
|
// The judge reads what the generator read: the source body, without its frontmatter (buildDistillPrompt).
|
|
@@ -452,6 +453,7 @@ async function judgeAndQueue(run, out) {
|
|
|
452
453
|
}
|
|
453
454
|
if (verdict.score > 0)
|
|
454
455
|
confidence = verdict.score / 5;
|
|
456
|
+
judged = verdict;
|
|
455
457
|
}
|
|
456
458
|
let frontmatter;
|
|
457
459
|
if (out.promotion) {
|
|
@@ -489,7 +491,7 @@ async function judgeAndQueue(run, out) {
|
|
|
489
491
|
...(run.options.eligibilitySource ? { eligibilitySource: run.options.eligibilitySource } : {}),
|
|
490
492
|
// The ledger keys the attempt by the input, not the output.
|
|
491
493
|
attemptedRefs: [run.ledgerRef],
|
|
492
|
-
}, { judged
|
|
494
|
+
}, { judged });
|
|
493
495
|
persistOutputEncodingSalience(run, out.ref, content);
|
|
494
496
|
const swapped = out.descriptionSwapped ? { descriptionSwapped: out.descriptionSwapped } : {};
|
|
495
497
|
emitDistill(run, {
|
|
@@ -220,7 +220,7 @@ export const improveCommand = defineCommand({
|
|
|
220
220
|
},
|
|
221
221
|
"require-feedback-signal": {
|
|
222
222
|
type: "boolean",
|
|
223
|
-
description: "
|
|
223
|
+
description: "Turn the proactive/high-salience fallback lanes off (they only select and score assets; a rewrite needs negative feedback)",
|
|
224
224
|
default: false,
|
|
225
225
|
},
|
|
226
226
|
"json-to-stdout": {
|
|
@@ -9,7 +9,7 @@
|
|
|
9
9
|
import fs from "node:fs";
|
|
10
10
|
import path from "node:path";
|
|
11
11
|
import { parseRefInput } from "../../core/asset/resolve-ref.js";
|
|
12
|
-
import { bundlesToSourceEntries, loadConfig
|
|
12
|
+
import { bundlesToSourceEntries, loadConfig } from "../../core/config/config.js";
|
|
13
13
|
import { ConfigError, rethrowIfTestIsolationError, UsageError } from "../../core/errors.js";
|
|
14
14
|
import { appendEvent, readEvents } from "../../core/events.js";
|
|
15
15
|
import { classifyImproveAction, foldDistillSkipped } from "../../core/improve-types.js";
|
|
@@ -39,13 +39,11 @@ import { akmDistill } from "./distill.js";
|
|
|
39
39
|
import { collectEligibleRefs, collectEligibleRefsReadOnly, memoryCleanupParentRef, resolveImproveScope, shouldAnalyzeMemoryCleanup, } from "./eligibility.js";
|
|
40
40
|
import { eligibleRefCount, projectResolvedProcessRouting, resolveImprovePlan, resolveImproveStrategy, } from "./improve-strategies.js";
|
|
41
41
|
import { buildImproveUsageReport } from "./improve-usage-report.js";
|
|
42
|
-
import { lastAttemptByRef, loadLedgerSnapshot } from "./ledger.js";
|
|
43
42
|
import { improveLockPath, releaseImproveLock, tryAcquireImproveLock } from "./locks.js";
|
|
44
43
|
import { runImproveLoopStage, runImprovePostLoopStage } from "./loop-stages.js";
|
|
45
44
|
import { analyzeMemoryCleanup, purgeGracedArchive, RETIRE_GRACE_DAYS, } from "./memory/memory-improve.js";
|
|
46
45
|
import { buildImproveExecutionPlan } from "./planner.js";
|
|
47
46
|
import { CONSOLIDATION_CONFIG_KEYS, pickDefined, recordImproveSkip, runImprovePreparationStage } from "./preparation.js";
|
|
48
|
-
import { DEFAULT_DUE_DAYS, filterProactiveDue } from "./proactive-maintenance.js";
|
|
49
47
|
import { akmReflect } from "./reflect.js";
|
|
50
48
|
import { errMessage, noticeSet } from "./stage.js";
|
|
51
49
|
export { runImproveMaintenancePasses } from "./loop-stages.js";
|
|
@@ -57,7 +55,7 @@ export function renderSyncCommitMessage(template, result, nowMs) {
|
|
|
57
55
|
time: iso.slice(11, 19),
|
|
58
56
|
scope: result.scope.value ?? result.scope.mode,
|
|
59
57
|
refs: String(result.plannedRefs.length),
|
|
60
|
-
accepted: String(result.
|
|
58
|
+
accepted: String(result.triage?.promoted ?? 0),
|
|
61
59
|
triage_promoted: String(result.triage?.promoted ?? 0),
|
|
62
60
|
triage_rejected: String(result.triage?.rejected ?? 0),
|
|
63
61
|
runId: result.runId ?? "",
|
|
@@ -758,25 +756,6 @@ function makeCommitStashBatch(deps) {
|
|
|
758
756
|
}
|
|
759
757
|
};
|
|
760
758
|
}
|
|
761
|
-
/**
|
|
762
|
-
* Re-read the improve ledger under the lock and drop proactive refs another run
|
|
763
|
-
* attempted after this one planned.
|
|
764
|
-
*/
|
|
765
|
-
export function refilterProactiveLoopRefs(loopRefs, improveProfile, ledgerAccess) {
|
|
766
|
-
const proactiveLoopRefs = loopRefs.filter((r) => r.eligibilitySource === "proactive");
|
|
767
|
-
if (proactiveLoopRefs.length === 0 || !ledgerAccess.stashDir)
|
|
768
|
-
return loopRefs;
|
|
769
|
-
const ledger = loadLedgerSnapshot({ eventsCtx: ledgerAccess.eventsCtx }, ledgerAccess.stashDir, [
|
|
770
|
-
"reflect",
|
|
771
|
-
"distill",
|
|
772
|
-
]);
|
|
773
|
-
const stillDue = new Set(filterProactiveDue(proactiveLoopRefs, lastAttemptByRef(ledger, "reflect", proactiveLoopRefs), lastAttemptByRef(ledger, "distill", proactiveLoopRefs), improveProfile.processes?.proactiveMaintenance?.dueDays ?? DEFAULT_DUE_DAYS, Date.now()).map((r) => r.ref));
|
|
774
|
-
const dropped = proactiveLoopRefs.filter((r) => !stillDue.has(r.ref));
|
|
775
|
-
if (dropped.length === 0)
|
|
776
|
-
return loopRefs;
|
|
777
|
-
info(`[improve] post-lock cooldown re-filter: dropped ${dropped.length} proactive ref(s) claimed by concurrent run (${dropped.map((r) => r.ref).join(", ")})`);
|
|
778
|
-
return loopRefs.filter((r) => r.eligibilitySource !== "proactive" || stillDue.has(r.ref));
|
|
779
|
-
}
|
|
780
759
|
/**
|
|
781
760
|
* The audit events for refs and lanes this run will not touch, then
|
|
782
761
|
* preparation → loop → post-loop. No post-loop work starts past the budget; the
|
|
@@ -824,10 +803,7 @@ async function runImproveStageSequence(run, collected, preEnsureCleanupWarnings,
|
|
|
824
803
|
options,
|
|
825
804
|
reflectFn: run.reflectFn,
|
|
826
805
|
distillFn: run.distillFn,
|
|
827
|
-
loopRefs:
|
|
828
|
-
stashDir: primaryStashDir ?? options.stashDir,
|
|
829
|
-
eventsCtx,
|
|
830
|
-
}),
|
|
806
|
+
loopRefs: preparation.loopRefs,
|
|
831
807
|
actions: preparation.actions,
|
|
832
808
|
signalBearingSet: preparation.signalBearingSet,
|
|
833
809
|
distillCooledRefs: preparation.distillCooledRefs,
|
|
@@ -8,10 +8,11 @@
|
|
|
8
8
|
*
|
|
9
9
|
* Candidate selection reads the improve ledger plus one set of signals: a ref
|
|
10
10
|
* is eligible for a source when feedback newer than its last attempt landed and
|
|
11
|
-
* no ledger window holds it
|
|
12
|
-
* by the fallback lanes (proactive
|
|
13
|
-
*
|
|
14
|
-
*
|
|
11
|
+
* no ledger window holds it, and reflect reads only negative feedback. Refs
|
|
12
|
+
* without recent feedback can still be picked by the fallback lanes (proactive
|
|
13
|
+
* maintenance, high salience), which only score them; the survivors are ranked
|
|
14
|
+
* by salience, checked on disk and capped. A plan-only run evaluates the same
|
|
15
|
+
* selectors against read snapshots and writes nothing.
|
|
15
16
|
*/
|
|
16
17
|
import fs from "node:fs";
|
|
17
18
|
import path from "node:path";
|
|
@@ -549,6 +550,7 @@ export function buildSnapshotManifest(args) {
|
|
|
549
550
|
const candidates = args.postCleanupRefs.filter((r) => !args.validationFailureRefs.has(r.ref));
|
|
550
551
|
const refByKey = new Map(candidates.map((r) => [keyOf(r), r.ref]));
|
|
551
552
|
const latestFeedbackTs = new Map();
|
|
553
|
+
const latestNegativeTs = new Map();
|
|
552
554
|
const feedback = new Map(candidates.map((r) => [r.ref, { hasSignal: false, positive: 0, negative: 0 }]));
|
|
553
555
|
if (candidates.length > 0) {
|
|
554
556
|
for (const e of readEvents({ type: "feedback" }, eventsCtx).events) {
|
|
@@ -557,12 +559,14 @@ export function buildSnapshotManifest(args) {
|
|
|
557
559
|
if (!ref || !entry)
|
|
558
560
|
continue;
|
|
559
561
|
const ts = e.ts ?? "";
|
|
562
|
+
const signal = e.metadata?.signal;
|
|
560
563
|
if (ts >= feedbackSinceCutoff && isSignalEvent(e.metadata)) {
|
|
561
564
|
entry.hasSignal = true;
|
|
562
565
|
if (ts > (latestFeedbackTs.get(ref) ?? ""))
|
|
563
566
|
latestFeedbackTs.set(ref, ts);
|
|
567
|
+
if (signal === "negative" && ts > (latestNegativeTs.get(ref) ?? ""))
|
|
568
|
+
latestNegativeTs.set(ref, ts);
|
|
564
569
|
}
|
|
565
|
-
const signal = e.metadata?.signal;
|
|
566
570
|
if (signal === "positive")
|
|
567
571
|
entry.positive++;
|
|
568
572
|
else if (signal === "negative")
|
|
@@ -576,6 +580,7 @@ export function buildSnapshotManifest(args) {
|
|
|
576
580
|
feedbackSinceCutoff,
|
|
577
581
|
nowIso: new Date().toISOString(),
|
|
578
582
|
latestFeedbackTs,
|
|
583
|
+
latestNegativeTs,
|
|
579
584
|
ledger,
|
|
580
585
|
lastReflectAttemptAt: lastAttemptByRef(ledger, "reflect", candidates),
|
|
581
586
|
lastDistillAttemptAt: lastAttemptByRef(ledger, "distill", candidates),
|
|
@@ -584,19 +589,21 @@ export function buildSnapshotManifest(args) {
|
|
|
584
589
|
}
|
|
585
590
|
/**
|
|
586
591
|
* Partition the post-cleanup refs against the ledger:
|
|
587
|
-
* - eligibleRefs: reflect's signal delta passes
|
|
588
|
-
*
|
|
592
|
+
* - eligibleRefs: reflect's signal delta passes, which only fresh negative
|
|
593
|
+
* feedback can do (distill may still be cooled);
|
|
594
|
+
* - distillOnlyRefs: only distill's passes (any signal, a positive or a note
|
|
595
|
+
* included), on a distill candidate;
|
|
589
596
|
* - noFeedbackPool: no recent feedback and no reflect window, left to the
|
|
590
|
-
* fallback lanes;
|
|
597
|
+
* fallback lanes, which only score them;
|
|
591
598
|
* - fullySkippedCount: feedback on record but nothing new, or a live window.
|
|
592
599
|
* An explicit `--scope <ref>` bypasses every gate.
|
|
593
600
|
*/
|
|
594
601
|
export function partitionBySignalDelta(args) {
|
|
595
602
|
const { postCleanupRefs, validationFailureRefs } = args;
|
|
596
|
-
const { latestFeedbackTs, ledger, nowIso } = args.snapshot;
|
|
597
|
-
// Newer feedback lifts a revisit window, never a rejection.
|
|
603
|
+
const { latestFeedbackTs, latestNegativeTs, ledger, nowIso } = args.snapshot;
|
|
604
|
+
// Newer feedback lifts a revisit window, never a rejection. Reflect reads negative feedback only.
|
|
598
605
|
const deltaPasses = (candidate, source) => {
|
|
599
|
-
const feedbackAt = latestFeedbackTs.get(candidate.ref);
|
|
606
|
+
const feedbackAt = (source === "reflect" ? latestNegativeTs : latestFeedbackTs).get(candidate.ref);
|
|
600
607
|
if (!feedbackAt)
|
|
601
608
|
return false;
|
|
602
609
|
const row = ledgerRowFor(ledger, source, candidate.ref, candidate.itemRef);
|
|
@@ -639,8 +646,8 @@ export function partitionBySignalDelta(args) {
|
|
|
639
646
|
}
|
|
640
647
|
/**
|
|
641
648
|
* Pick the loop's refs: signal delta, the fallback lanes (unless
|
|
642
|
-
* `--require-feedback-signal
|
|
643
|
-
* no-op-dampened ranking, the disk check and the limit.
|
|
649
|
+
* `--require-feedback-signal`; they score, never plan), lane attribution,
|
|
650
|
+
* salience, the no-op-dampened ranking, the disk check and the limit.
|
|
644
651
|
*/
|
|
645
652
|
async function selectLoopCandidates(args, postCleanupRefs, validationFailureRefs, actions, persist) {
|
|
646
653
|
const { scope, options, primaryStashDir, eventsCtx, improveProfile } = args;
|
|
@@ -678,16 +685,13 @@ async function selectLoopCandidates(args, postCleanupRefs, validationFailureRefs
|
|
|
678
685
|
const highSalienceRefs = allowFallbacks
|
|
679
686
|
? selectHighSalienceLane(options, improveProfile, eventsCtx, noFeedbackCandidates.filter((r) => !proactive.proactiveRefs.some((p) => p.ref === r.ref)), snapshot.lastReflectAttemptAt, persist)
|
|
680
687
|
: [];
|
|
681
|
-
// An explicit ref scope always acts on its ref; otherwise
|
|
688
|
+
// An explicit ref scope always acts on its ref; otherwise only feedback plans the loop. The fallback lanes'
|
|
689
|
+
// picks are scored with it, never planned: a rewrite needs negative feedback.
|
|
682
690
|
const signalAndRetrievalRefs = dedupeRefs([...signalFiltered, ...proactive.proactiveRefs, ...highSalienceRefs]);
|
|
683
|
-
const mergedRefs = scope.mode === "ref" ? processableRefs :
|
|
684
|
-
|
|
685
|
-
//
|
|
691
|
+
const mergedRefs = scope.mode === "ref" ? processableRefs : signalFiltered;
|
|
692
|
+
const scoredRefs = scope.mode === "ref" ? processableRefs : signalAndRetrievalRefs;
|
|
693
|
+
// Lane attribution: signal-delta, and an explicit ref scope over it.
|
|
686
694
|
const sourceByRef = new Map();
|
|
687
|
-
for (const r of highSalienceRefs)
|
|
688
|
-
sourceByRef.set(r.ref, "high-salience");
|
|
689
|
-
for (const r of proactive.proactiveRefs)
|
|
690
|
-
sourceByRef.set(r.ref, "proactive");
|
|
691
695
|
for (const r of signalFiltered)
|
|
692
696
|
sourceByRef.set(r.ref, "signal-delta");
|
|
693
697
|
if (scope.mode === "ref")
|
|
@@ -695,7 +699,7 @@ async function selectLoopCandidates(args, postCleanupRefs, validationFailureRefs
|
|
|
695
699
|
sourceByRef.set(r.ref, "scope");
|
|
696
700
|
for (const r of mergedRefs)
|
|
697
701
|
r.eligibilitySource = sourceByRef.get(r.ref) ?? "unknown";
|
|
698
|
-
const salienceMap = scoreSalience(args,
|
|
702
|
+
const salienceMap = scoreSalience(args, scoredRefs, snapshot.feedback, retrieval.retrievalCounts, persist);
|
|
699
703
|
// Rank by salience; a ref skipped as a no-op repeatedly sorts lower (its stored rank is untouched).
|
|
700
704
|
const noOps = new Map();
|
|
701
705
|
withRunState(eventsCtx, persist, (db) => {
|
|
@@ -724,8 +728,7 @@ async function selectLoopCandidates(args, postCleanupRefs, validationFailureRefs
|
|
|
724
728
|
const deferred = actionableRefs.length - selection.loopRefs.length;
|
|
725
729
|
info(`[improve] ${actionableRefs.length} actionable; ${selection.loopRefs.length} will be processed` +
|
|
726
730
|
(options.limit && deferred > 0 ? ` (--limit ${options.limit} applied; ${deferred} deferred)` : ""));
|
|
727
|
-
// Skip observability
|
|
728
|
-
// survivors, so a rescued ref is never also reported skipped.
|
|
731
|
+
// Skip observability: every candidate that did not survive into the loop pool is reported skipped.
|
|
729
732
|
const survivors = new Set(sorted.map((c) => c.ref));
|
|
730
733
|
const skipped = fallbackEligible.filter((c) => !survivors.has(c.ref));
|
|
731
734
|
const retrievalSkipped = skipped.filter((c) => outOfScope.has(c.ref));
|
|
@@ -770,7 +773,7 @@ async function selectLoopCandidates(args, postCleanupRefs, validationFailureRefs
|
|
|
770
773
|
{
|
|
771
774
|
name: "signal",
|
|
772
775
|
removed: signalSkipped.length,
|
|
773
|
-
reason: "no fresh signal since the last attempt
|
|
776
|
+
reason: "no fresh negative feedback (or, for distill, any signal) since the last attempt, or a ledger window holds it",
|
|
774
777
|
},
|
|
775
778
|
{ name: "disk", removed: missing.length, reason: "backing asset is absent on disk" },
|
|
776
779
|
{ name: "limit", removed: selection.limitRemoved, reason: "deferred by the effective run limit" },
|
|
@@ -803,9 +806,11 @@ function fetchRetrievalSignals(options, signalFiltered, noFeedbackCandidates, ev
|
|
|
803
806
|
return out;
|
|
804
807
|
}
|
|
805
808
|
/**
|
|
806
|
-
* Proactive maintenance (default off, whole-stash/type runs):
|
|
807
|
-
* assets
|
|
808
|
-
*
|
|
809
|
+
* Proactive maintenance (default off, whole-stash/type runs): pick stable
|
|
810
|
+
* assets due for a revisit. The picks are scored and never planned: improve
|
|
811
|
+
* does not rewrite on a proactive cadence. The due gate doubles as the
|
|
812
|
+
* rotation cooldown: a freshly reflected asset waits `dueDays` before it is
|
|
813
|
+
* picked again.
|
|
809
814
|
*/
|
|
810
815
|
function selectProactiveMaintenanceLane(args, candidates, snapshot, retrieval, persist) {
|
|
811
816
|
if (args.scope.mode === "ref" || !args.resolvedPlan.processes.proactiveMaintenance.enabled) {
|
|
@@ -856,7 +861,8 @@ function selectProactiveMaintenanceLane(args, candidates, snapshot, retrieval, p
|
|
|
856
861
|
/**
|
|
857
862
|
* High salience: zero-feedback refs whose content-derived encoding score (not
|
|
858
863
|
* a per-type stub) reaches `salienceThreshold` and that were never reflected,
|
|
859
|
-
* top-N by score, capped at 10% of the effective limit.
|
|
864
|
+
* top-N by score, capped at 10% of the effective limit. The picks are scored,
|
|
865
|
+
* never planned.
|
|
860
866
|
*/
|
|
861
867
|
function selectHighSalienceLane(options, improveProfile, eventsCtx, candidates, lastReflectAttemptAt, persist) {
|
|
862
868
|
const threshold = (options.config ?? loadConfig()).improve?.salience?.salienceThreshold ?? 0.75;
|
|
@@ -11,8 +11,8 @@ export const DEFAULT_MAX_PER_RUN = 25;
|
|
|
11
11
|
/** Size floor for the cost term, so tiny files don't divide by ~0. */
|
|
12
12
|
const SIZE_FLOOR_BYTES = 200;
|
|
13
13
|
/**
|
|
14
|
-
* The due gate,
|
|
15
|
-
*
|
|
14
|
+
* The due gate: never touched, or last touched more than `dueDays` ago. It
|
|
15
|
+
* doubles as the rotation cooldown.
|
|
16
16
|
*/
|
|
17
17
|
function staleness(ref, lastReflectTs, lastDistillTs, dueDays, now) {
|
|
18
18
|
const lastTouchMs = Math.max(0, Date.parse(lastReflectTs.get(ref) ?? "") || 0, Date.parse(lastDistillTs.get(ref) ?? "") || 0);
|
|
@@ -64,10 +64,3 @@ export function selectProactiveMaintenanceRefs(params) {
|
|
|
64
64
|
scored,
|
|
65
65
|
};
|
|
66
66
|
}
|
|
67
|
-
/**
|
|
68
|
-
* Re-apply the due gate under the run lock with fresh timestamps, dropping refs
|
|
69
|
-
* another run attempted after this one planned.
|
|
70
|
-
*/
|
|
71
|
-
export function filterProactiveDue(selected, lastReflectTs, lastDistillTs, dueDays, now) {
|
|
72
|
-
return selected.filter((c) => staleness(c.ref, lastReflectTs, lastDistillTs, dueDays, now).due);
|
|
73
|
-
}
|
|
@@ -979,8 +979,9 @@ async function finalizeReflectProposal(args) {
|
|
|
979
979
|
}, options.eventsCtx);
|
|
980
980
|
return reflectFailure(run, result, "quality_rejected", message, false);
|
|
981
981
|
};
|
|
982
|
+
let verdict;
|
|
982
983
|
if (judged) {
|
|
983
|
-
|
|
984
|
+
verdict = await runReflectQualityJudge(run.config, payload.content, assetContent ?? "", feedback, options.chat, {
|
|
984
985
|
runnerSelectionFrozen: true,
|
|
985
986
|
...(judge.runner ? { llmRunner: judge.runner } : {}),
|
|
986
987
|
...(Object.hasOwn(options, "timeoutMs") ? { timeoutMs: options.timeoutMs } : {}),
|
|
@@ -1045,7 +1046,7 @@ async function finalizeReflectProposal(args) {
|
|
|
1045
1046
|
...(sanitized.sizeGuardRatio ? { measured: Math.round(sanitized.sizeGuardRatio.ratio * 100) } : {}),
|
|
1046
1047
|
},
|
|
1047
1048
|
}
|
|
1048
|
-
: { judged });
|
|
1049
|
+
: { judged: verdict });
|
|
1049
1050
|
appendEvent({
|
|
1050
1051
|
eventType: "reflect_completed",
|
|
1051
1052
|
ref: proposal.ref,
|