akm-cli 0.9.27-alpha.1 → 0.9.27-alpha.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +131 -0
- package/LICENSE +3 -4
- package/dist/assets/prompts/distill-lesson-system.md +29 -7
- package/dist/commands/improve/consolidate/coverage.js +71 -17
- package/dist/commands/improve/consolidate/pair-pass.js +11 -8
- package/dist/commands/improve/consolidate.js +63 -1
- package/dist/commands/improve/distill-guards.js +9 -10
- package/dist/commands/improve/distill.js +62 -18
- package/dist/commands/improve/preparation.js +14 -1
- package/dist/commands/improve/stage.js +34 -52
- package/dist/commands/proposal/drain.js +90 -22
- package/dist/commands/proposal/proposal-types.js +1 -1
- package/dist/core/config/schema/improve-processes.js +3 -3
- package/dist/core/paths.js +0 -9
- package/dist/storage/repositories/improve-ledger-repository.js +4 -1
- package/docs/README.md +1 -2
- package/docs/integration/bundling-akm.md +1 -1
- package/docs/migration/v0.8-to-v0.9.md +3 -1
- package/docs/reference/README.md +1 -1
- package/docs/reference/cli.md +7 -4
- package/docs/reference/data-and-telemetry.md +1 -1
- package/package.json +1 -1
package/CHANGELOG.md
CHANGED
|
@@ -6,6 +6,137 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/).
|
|
|
6
6
|
|
|
7
7
|
## [Unreleased]
|
|
8
8
|
|
|
9
|
+
## [0.9.27-alpha.3] - 2026-10-07
|
|
10
|
+
|
|
11
|
+
### Fixed
|
|
12
|
+
|
|
13
|
+
- **The drain's judge rejects a promotion that is a status snapshot, a plan or
|
|
14
|
+
something retired, not only a duplicate.** Its rubric for a promotion (a memory
|
|
15
|
+
proposed as a new knowledge note) said to reject a note that is wrong, a
|
|
16
|
+
duplicate or contradicts the live asset, and to accept a correct, valuable one,
|
|
17
|
+
so it judged only whether the note was new: the reasons it gave for accepting a
|
|
18
|
+
note pinned to commits and versions, a rollout status or a retired host were
|
|
19
|
+
"distinct from the existing notes". The reject line now also names a note that
|
|
20
|
+
reports the state of something that changes (a status, rollout, branch, commit,
|
|
21
|
+
version, test count, "as of <date>"), a plan not yet carried out, or something
|
|
22
|
+
already retired, replaced or superseded, and says a lesson drawn from an
|
|
23
|
+
incident is durable. On the owner's 65 labelled promotions (chat/qwen3.8-27b,
|
|
24
|
+
two runs each) the judge accepted 7 and 8 of the 14 stale ones and 4 and 5 of
|
|
25
|
+
the 7 ephemeral ones before, and 1 and 3, 0 and 1 after. Other proposals keep
|
|
26
|
+
the old rubric. No new settings.
|
|
27
|
+
|
|
28
|
+
## [0.9.27-alpha.2] - 2026-10-07
|
|
29
|
+
|
|
30
|
+
### Removed
|
|
31
|
+
|
|
32
|
+
- **The `scripts/akm-eval` toolkit has moved out of this repository.** The
|
|
33
|
+
read-only measurement toolkit (the case runner and its suites, the twin
|
|
34
|
+
experiment, the real-query verdict for the proactive lane, the state
|
|
35
|
+
analyzers and the curate benchmark) is retired; every live eval is in
|
|
36
|
+
[itlackey/akm-eval](https://github.com/itlackey/akm-eval). Its code is kept
|
|
37
|
+
there, to read and not to run, in `retired/akm-scripts-akm-eval/`, copied from
|
|
38
|
+
commit `f57a7fd44b37`. It imports akm's `src/` by relative path, so it runs
|
|
39
|
+
only in a checkout at that commit. Removed here with it: `scripts/akm-eval/`,
|
|
40
|
+
its tests (`tests/integration/akm-eval/`, `tests/akm-eval-*.test.ts`,
|
|
41
|
+
`tests/curate-metrics.test.ts`) and fixtures (`tests/fixtures/akm-eval/`, and
|
|
42
|
+
the `curate-golden` stash, which only the curate benchmark read), the
|
|
43
|
+
`akm-eval determinism` CI job, and `getMeasurementVerdictsDir`, whose only
|
|
44
|
+
caller was the verdict runner. akm no longer names
|
|
45
|
+
`$STATE/improve/measurement/verdicts/<stash>/`; a file already there is inert.
|
|
46
|
+
`docs/maintainers/eval.md` is now a pointer to the new home.
|
|
47
|
+
|
|
48
|
+
### Fixed
|
|
49
|
+
|
|
50
|
+
- **The drain's judge no longer sees a note with a code block as truncated.**
|
|
51
|
+
The judgment prompt fenced the proposed content (and the live asset, sibling
|
|
52
|
+
proposals and neighbour excerpts) in three backticks, so a note holding its
|
|
53
|
+
own code block closed the fence early and read as cut off; real rejections said
|
|
54
|
+
"ends in an empty code block" or "truncated". Each block now uses a fence longer
|
|
55
|
+
than any backtick run inside it. The judge's reason is also kept on accepts,
|
|
56
|
+
staged accepts and defers (as the gate decision's `judgeReason`, until now
|
|
57
|
+
rejections only), and a judge reply that is not a verdict is stamped
|
|
58
|
+
`judgment-parse-failure`, and a runner failure `judgment-error`, instead of
|
|
59
|
+
looking like a defer.
|
|
60
|
+
|
|
61
|
+
- **Distill writes a lesson only when its memory holds one, says only what the
|
|
62
|
+
memory says, and its judge rejects what a reviewer would.** 2 of the 22 distill
|
|
63
|
+
proposals since 0.9.26 began were accepted, and 17 of the 19 queued on
|
|
64
|
+
2026-10-05 were bad (they restated their memory, filed a dated status as a
|
|
65
|
+
lesson, claimed what the memory does not say, or repeated an asset the library
|
|
66
|
+
holds). Four causes, found in the code and the rejected proposals, and fixed:
|
|
67
|
+
(1) the prompt and schema forced a lesson from every memory, and 18 of the 19
|
|
68
|
+
were records of what was done; the writer now says why a memory holds a
|
|
69
|
+
lesson or none (`reason`, then `decision: lesson|none`, or the word `NONE`),
|
|
70
|
+
defined as a cause and what to do about it, or a rule with its reason, and
|
|
71
|
+
writes only what the memory and its feedback state, in the scope they have; a
|
|
72
|
+
`NONE` is a `skipped` distill (`skipReason: nothing_reusable`, the writer's
|
|
73
|
+
reason in the message) with no proposal and no judge call, and the loop keeps
|
|
74
|
+
its ledger row `unchanged`. (2) The judge asked for "information not already
|
|
75
|
+
present in the source", so an invented claim scored as novel and a faithful
|
|
76
|
+
lesson of a lesson-worthy memory as a restatement, and it passed anything that
|
|
77
|
+
"goes beyond the source" because it "may draw on feedback you are not shown".
|
|
78
|
+
The rubric is now reusable (a rule with its reason, not a record of what was
|
|
79
|
+
done), non-redundancy and grounding (every cause, step, number and limit is
|
|
80
|
+
in the source or its feedback), and the judge is shown the feedback the writer
|
|
81
|
+
saw. (3) A mean hid a decisive score (4 and 1 average 2.5, a review), and a
|
|
82
|
+
reviewer read everything the judge did not reject; any criterion at 2 or
|
|
83
|
+
below, grounding included, is now `quality_rejected`, and the reason names it
|
|
84
|
+
(`grounding 2/5: …`). The "borderline grounding" routing is gone. (4) Neither
|
|
85
|
+
the writer nor the judge could see a knowledge note or a skill that already
|
|
86
|
+
states the rule (the judge saw the 3 lexically nearest lessons, none of them
|
|
87
|
+
related); both now see the lessons, knowledge notes and skills nearest the
|
|
88
|
+
memory, which is the existing `processes.distill.cls` context turned on by
|
|
89
|
+
default (`enabled: false` turns it off). Judge scores are keyed `reusable`
|
|
90
|
+
where they were `novelty`. Measured on the local qwen3.8-27b with akm-eval's
|
|
91
|
+
`evals/distill` (30 fictional memories, 5 runs each side): good lessons 7/14 on
|
|
92
|
+
average (5 to 9) against 3.7/14 (3 to 5), lessons queued for memories that
|
|
93
|
+
deserve none 0.4/16 against 3.3/16. On 37 real memories with their feedback
|
|
94
|
+
(34 reviewed bad, 3 good; 3 runs against 2): a lesson was queued for 11% of the
|
|
95
|
+
bad ones against 44%, and for 6 of 9 good ones against 4 of 6; of the memories
|
|
96
|
+
that pass 0.9.26's skip of bare positive feedback, 20% of the bad against 50%.
|
|
97
|
+
No new settings.
|
|
98
|
+
|
|
99
|
+
- **Reflect no longer plans an asset whose negative feedback is already acted
|
|
100
|
+
on.** A negative `akm feedback` that came with an exact fix (`--replace` and
|
|
101
|
+
`--with`, `--outdated` or `--superseded-by`) makes a `feedback` proposal, and
|
|
102
|
+
once that proposal is accepted the feedback has done its work. Reflect still
|
|
103
|
+
took the ref as having fresh negative feedback, and on 2026-10-07 44 of its 50
|
|
104
|
+
refs were of that kind: the judge refused or the model changed nothing for
|
|
105
|
+
most of them. A negative event with a fix is now left out of the reflect
|
|
106
|
+
cursor when an accepted `feedback` proposal for the ref was created at or
|
|
107
|
+
after it. A negative with no fix, one given after the proposal, and one whose
|
|
108
|
+
proposal is still pending or was rejected plan a reflect as before.
|
|
109
|
+
|
|
110
|
+
- **Consolidate stops re-offering memories a reviewer already turned down, and
|
|
111
|
+
the nightly judge sees what a promotion may duplicate.** About 53 promotions a
|
|
112
|
+
night reached review at ~5% precision, 64-70% of them a memory body already
|
|
113
|
+
proposed or rejected. Four causes, four changes: a memory whose body equals
|
|
114
|
+
that of a consolidate promotion rejected on or after 2026-09-29 is held until
|
|
115
|
+
its body changes, under any name (earlier rejections, the bulk audits of
|
|
116
|
+
2026-08, do not count); a memory the model judged and left alone is held by its
|
|
117
|
+
body hash instead of a 7-day clock, so an unchanged memory is no longer judged
|
|
118
|
+
every week (a row recorded without a hash keeps the 7 days); the coverage gate
|
|
119
|
+
skips a memory when 30% of its text, not 50%, is in a neighbouring knowledge
|
|
120
|
+
doc, which catches paraphrases; and the drain's judgment tier, which judged a
|
|
121
|
+
promotion seeing only the proposal and never `knowledge/`, is now shown the 5
|
|
122
|
+
nearest knowledge notes (ref, description, excerpt) and told to reject a
|
|
123
|
+
promotion they already cover. No new settings.
|
|
124
|
+
|
|
125
|
+
- **A confident `subsumed` or `supersedes` retirement resolves unattended, as a
|
|
126
|
+
`duplicate` already did.** The pair pass staged a retire proposal for the
|
|
127
|
+
triage drain only when the judge's label was `duplicate`; every other retirement
|
|
128
|
+
waited for a person. It now stages any of the three retire labels when the
|
|
129
|
+
second look (what does the retired note hold that the kept one lacks?) comes
|
|
130
|
+
back empty and there is no continuity risk; the retired side's claim list is
|
|
131
|
+
already empty for any proposal, and a `duplicate` must still have an empty list
|
|
132
|
+
on the kept side too. The staged gate reason is the judge's label, and the drain records it. Replay over 360
|
|
133
|
+
judged pairs: 336 safe (0.93): `duplicate` 0.98, `subsumed` 0.92, `supersedes`
|
|
134
|
+
0.875. In production, unstaged `subsumed` retirements were accepted 45 of 54
|
|
135
|
+
times by hand, and in the latest run 41 of 47 pair proposals would have resolved
|
|
136
|
+
without a person. `docs/architecture/internals/improve-workflow.md` said triage
|
|
137
|
+
never auto-accepts a retire proposal, which stopped being true in 0.9.26; it,
|
|
138
|
+
and the matching lines in `improvement.md`, now describe the staging rule.
|
|
139
|
+
|
|
9
140
|
## [0.9.27-alpha.1] - 2026-10-06
|
|
10
141
|
|
|
11
142
|
### Fixed
|
package/LICENSE
CHANGED
|
@@ -225,10 +225,9 @@ statute, judicial order, or regulation then You must: (a) comply with
|
|
|
225
225
|
the terms of this License to the maximum extent possible; and (b)
|
|
226
226
|
describe the limitations and the code they affect. Such description must
|
|
227
227
|
be placed in a text file included with all distributions of the Covered
|
|
228
|
-
Software under
|
|
229
|
-
|
|
230
|
-
|
|
231
|
-
for a recipient of ordinary skill to be able to understand it.
|
|
228
|
+
Software under this License. Except to the extent prohibited by statute
|
|
229
|
+
or regulation, such description must be sufficiently detailed for a
|
|
230
|
+
recipient of ordinary skill to be able to understand it.
|
|
232
231
|
|
|
233
232
|
5. Termination
|
|
234
233
|
--------------
|
|
@@ -1,7 +1,29 @@
|
|
|
1
1
|
You are the akm `distill` distiller.
|
|
2
|
-
|
|
3
|
-
|
|
4
|
-
|
|
2
|
+
You are given a memory and the feedback recorded about it. Decide whether it
|
|
3
|
+
holds a lesson and, if it does, write the lesson.
|
|
4
|
+
|
|
5
|
+
A memory holds a lesson when it states a cause and what to do about it: a
|
|
6
|
+
failure or surprise with its cause and the fix that worked, or a rule with the
|
|
7
|
+
reason it holds. It holds a lesson even when it is short and names one project,
|
|
8
|
+
tool or incident, if the cause and the fix would help someone in a similar
|
|
9
|
+
situation.
|
|
10
|
+
|
|
11
|
+
A memory holds NO lesson when all it states is what was done, shipped, released
|
|
12
|
+
or decided, what is pending or planned, how a system is set up now, or the steps
|
|
13
|
+
of a procedure, with no failure and cause behind it, or when its feedback says
|
|
14
|
+
only that it is out of date or superseded. ANSWER NONE then: the single word and
|
|
15
|
+
nothing else. A reply bound to a JSON schema answers NONE by
|
|
16
|
+
setting `decision` to `none` and leaving the other fields empty. Answer NONE too
|
|
17
|
+
when a related asset listed below the memory already states the rule the memory
|
|
18
|
+
would give.
|
|
19
|
+
|
|
20
|
+
When the memory holds a lesson, write it from what the memory and its feedback
|
|
21
|
+
say, and nothing more:
|
|
22
|
+
- Add no cause, step, rule, number, check or safeguard that neither states.
|
|
23
|
+
- Keep the scope the memory has. A fix verified in one place is a fix for that
|
|
24
|
+
place, and what was not checked stays unchecked. One case is not "always" or
|
|
25
|
+
"never".
|
|
26
|
+
- Be shorter than the memory.
|
|
5
27
|
|
|
6
28
|
YOUR RESPONSE MUST START EXACTLY WITH `---` ON THE VERY FIRST LINE.
|
|
7
29
|
DO NOT output any prose, explanation, or code fences before or after.
|
|
@@ -12,7 +34,7 @@ description: <one complete sentence (ending with `.`) summarising what the lesso
|
|
|
12
34
|
when_to_use: <one complete sentence describing the concrete trigger condition>
|
|
13
35
|
---
|
|
14
36
|
|
|
15
|
-
<lesson body — plain markdown,
|
|
37
|
+
<lesson body — plain markdown, as short as the memory allows>
|
|
16
38
|
|
|
17
39
|
## description field (MANDATORY)
|
|
18
40
|
- A single complete sentence in present tense, 20–400 chars, NO markdown.
|
|
@@ -21,7 +43,7 @@ when_to_use: <one complete sentence describing the concrete trigger condition>
|
|
|
21
43
|
- DO NOT copy a section heading ("Key takeaways", "For example", "Key pitfalls").
|
|
22
44
|
- DO NOT begin with a numbered list marker, code fence, or markdown heading.
|
|
23
45
|
|
|
24
|
-
GOOD: "
|
|
46
|
+
GOOD: "Pin the container image tag, because the `latest` tag moved under the nightly job and its output changed with no code change."
|
|
25
47
|
BAD: "Key pitfalls"
|
|
26
48
|
BAD: "When working with the akm CLI"
|
|
27
49
|
BAD: "For example, you might..."
|
|
@@ -32,5 +54,5 @@ RULES:
|
|
|
32
54
|
- `description` and `when_to_use` MUST differ from each other.
|
|
33
55
|
- The lesson body MUST be non-empty markdown prose. Do NOT restate `description:` or `when_to_use:` inside the body (no `**description:** ...` or `**when_to_use:** ...` lines — the frontmatter is the only place those keys belong).
|
|
34
56
|
- Do NOT emit a second `---` fence after the opening frontmatter — there are exactly two `---` lines in the output, both belonging to the single frontmatter block at the top.
|
|
35
|
-
- Do NOT reproduce the source asset verbatim
|
|
36
|
-
- Output ONLY the lesson file. No preamble, no code fences, no trailing prose.
|
|
57
|
+
- Do NOT reproduce the source asset verbatim.
|
|
58
|
+
- Output ONLY the lesson file. No preamble, no code fences, no trailing prose.
|
|
@@ -6,7 +6,7 @@
|
|
|
6
6
|
* `knowledge/` proposal, ask whether `knowledge/` already says it.
|
|
7
7
|
*
|
|
8
8
|
* The rule: a memory is covered when at least {@link COVERAGE_MIN_CONTAINMENT}
|
|
9
|
-
* (
|
|
9
|
+
* (30%) of its distinct {@link COVERAGE_SHINGLE_WORDS}-word shingles appear in
|
|
10
10
|
* one of the knowledge docs nearest to it. It is a containment of the MEMORY in
|
|
11
11
|
* the doc, not a similarity: a long guide that quotes the memory covers it; a
|
|
12
12
|
* memory that quotes a short doc and adds claims of its own does not.
|
|
@@ -25,9 +25,13 @@
|
|
|
25
25
|
* this gate achieves: it reads only the {@link PAIR_NEIGHBOR_FETCH_K} nearest
|
|
26
26
|
* knowledge docs (below), and a covering doc that ranks lower goes unseen. The
|
|
27
27
|
* recall over that candidate set is unmeasured. The rest of the rejected
|
|
28
|
-
* proposals (paraphrases, partial overlaps) still reach review
|
|
29
|
-
*
|
|
30
|
-
*
|
|
28
|
+
* proposals (paraphrases, partial overlaps) still reach review. 0.5 let
|
|
29
|
+
* paraphrases through: of the 54 promotions minted on 2026-10-07, 18 of the 51
|
|
30
|
+
* later rejected held 30% or more of their text in a neighbouring doc. The cut is
|
|
31
|
+
* now 0.3, between the two measured points: 0.2 (169 of 224 rejected) also
|
|
32
|
+
* skipped 2 of the 103 accepted ones. At 0.3, none of the 5 promotions graded
|
|
33
|
+
* good in the 2026-10-05 review sample would be skipped (their best doc holds
|
|
34
|
+
* at most 1% of them) while 5 of its 15 bad ones would. A wrong skip is a
|
|
31
35
|
* promotion nobody gets to review.
|
|
32
36
|
*
|
|
33
37
|
* Candidates are the memory's {@link PAIR_NEIGHBOR_FETCH_K} nearest knowledge
|
|
@@ -49,7 +53,7 @@ import { PAIR_NEIGHBOR_FETCH_K } from "./pair-pass.js";
|
|
|
49
53
|
/** Words per shingle. */
|
|
50
54
|
export const COVERAGE_SHINGLE_WORDS = 5;
|
|
51
55
|
/** Share of a memory's distinct shingles one knowledge doc must hold for the memory to count as covered. */
|
|
52
|
-
export const COVERAGE_MIN_CONTAINMENT = 0.
|
|
56
|
+
export const COVERAGE_MIN_CONTAINMENT = 0.3;
|
|
53
57
|
const WORD = /[\p{L}\p{N}]+/gu;
|
|
54
58
|
/** The distinct lower-cased word n-grams of `text`; empty when it has fewer than {@link COVERAGE_SHINGLE_WORDS} words. */
|
|
55
59
|
export function wordShingles(text) {
|
|
@@ -72,19 +76,15 @@ export function shingleContainment(memory, doc) {
|
|
|
72
76
|
return shared / memory.size;
|
|
73
77
|
}
|
|
74
78
|
/**
|
|
75
|
-
* The
|
|
76
|
-
*
|
|
77
|
-
*
|
|
78
|
-
* and so no candidates.
|
|
79
|
+
* The {@link PAIR_NEIGHBOR_FETCH_K} knowledge docs in `bundleId` nearest to the
|
|
80
|
+
* memory at `filePath`, nearest first. A memory the index does not know has no
|
|
81
|
+
* stored vector and so no neighbours.
|
|
79
82
|
*/
|
|
80
|
-
|
|
81
|
-
const shingles = wordShingles(body);
|
|
82
|
-
if (shingles.size === 0)
|
|
83
|
-
return undefined;
|
|
83
|
+
function knowledgeNeighbours(db, bundleId, filePath) {
|
|
84
84
|
const entryId = getEntryIdByFilePath(db, filePath);
|
|
85
85
|
if (entryId === undefined)
|
|
86
|
-
return
|
|
87
|
-
|
|
86
|
+
return [];
|
|
87
|
+
const out = [];
|
|
88
88
|
for (const hit of getNeighborsByEntryId(db, entryId, PAIR_NEIGHBOR_FETCH_K, { type: "knowledge", bundleId })) {
|
|
89
89
|
const neighbour = getEntryById(db, hit.id);
|
|
90
90
|
if (!neighbour)
|
|
@@ -96,13 +96,67 @@ export function findCoveringKnowledge(db, bundleId, filePath, body) {
|
|
|
96
96
|
catch {
|
|
97
97
|
continue; // the index outlived the file
|
|
98
98
|
}
|
|
99
|
-
|
|
99
|
+
out.push({
|
|
100
|
+
ref: neighbour.conceptId,
|
|
101
|
+
description: neighbour.entry.description ?? "",
|
|
102
|
+
body: stripFrontmatterBody(raw),
|
|
103
|
+
});
|
|
104
|
+
}
|
|
105
|
+
return out;
|
|
106
|
+
}
|
|
107
|
+
/**
|
|
108
|
+
* The best-covering knowledge doc among the {@link PAIR_NEIGHBOR_FETCH_K}
|
|
109
|
+
* knowledge docs in `bundleId` nearest to the memory. `filePath` is the
|
|
110
|
+
* memory's indexed file; a memory the index does not know has no stored
|
|
111
|
+
* vector and so no candidates.
|
|
112
|
+
*/
|
|
113
|
+
export function findCoveringKnowledge(db, bundleId, filePath, body) {
|
|
114
|
+
const shingles = wordShingles(body);
|
|
115
|
+
if (shingles.size === 0)
|
|
116
|
+
return undefined;
|
|
117
|
+
let best;
|
|
118
|
+
for (const neighbour of knowledgeNeighbours(db, bundleId, filePath)) {
|
|
119
|
+
const containment = shingleContainment(shingles, neighbour.body);
|
|
100
120
|
if (containment >= COVERAGE_MIN_CONTAINMENT && (best === undefined || containment > best.containment)) {
|
|
101
|
-
best = { ref: neighbour.
|
|
121
|
+
best = { ref: neighbour.ref, containment };
|
|
102
122
|
}
|
|
103
123
|
}
|
|
104
124
|
return best;
|
|
105
125
|
}
|
|
126
|
+
/** Knowledge docs a reviewer is shown for a promotion, nearest first. */
|
|
127
|
+
export const NEIGHBOUR_NOTE_COUNT = 5;
|
|
128
|
+
const NEIGHBOUR_EXCERPT_CHARS = 300;
|
|
129
|
+
/**
|
|
130
|
+
* The knowledge notes nearest to the memory at `memoryPath`, for the drain's
|
|
131
|
+
* judge to compare a promotion against: the nearest {@link NEIGHBOUR_NOTE_COUNT}
|
|
132
|
+
* of the same candidates the coverage gate reads. The memory's bundle is the one
|
|
133
|
+
* the index recorded for it. Empty when the index has no vector for the memory
|
|
134
|
+
* or cannot be opened; never throws.
|
|
135
|
+
*/
|
|
136
|
+
export function nearestKnowledgeNotes(memoryPath) {
|
|
137
|
+
let db;
|
|
138
|
+
try {
|
|
139
|
+
db = openExistingDatabase();
|
|
140
|
+
const entryId = getEntryIdByFilePath(db, memoryPath);
|
|
141
|
+
const bundleId = entryId === undefined ? undefined : getEntryById(db, entryId)?.bundleId;
|
|
142
|
+
if (bundleId === undefined)
|
|
143
|
+
return [];
|
|
144
|
+
return knowledgeNeighbours(db, bundleId, memoryPath)
|
|
145
|
+
.slice(0, NEIGHBOUR_NOTE_COUNT)
|
|
146
|
+
.map((n) => ({
|
|
147
|
+
ref: n.ref,
|
|
148
|
+
description: n.description,
|
|
149
|
+
excerpt: n.body.length > NEIGHBOUR_EXCERPT_CHARS ? `${n.body.slice(0, NEIGHBOUR_EXCERPT_CHARS)}...` : n.body,
|
|
150
|
+
}));
|
|
151
|
+
}
|
|
152
|
+
catch {
|
|
153
|
+
return [];
|
|
154
|
+
}
|
|
155
|
+
finally {
|
|
156
|
+
if (db)
|
|
157
|
+
closeDatabase(db);
|
|
158
|
+
}
|
|
159
|
+
}
|
|
106
160
|
/**
|
|
107
161
|
* The gate for one run, holding its own read handle on `index.db` for the
|
|
108
162
|
* promotions that run emits; `undefined` when there is no bundle or no index
|
|
@@ -503,7 +503,7 @@ function checkSection(label, side) {
|
|
|
503
503
|
].join("\n");
|
|
504
504
|
}
|
|
505
505
|
/**
|
|
506
|
-
* The second look a
|
|
506
|
+
* The second look a retirement gets before it may retire unattended: one call
|
|
507
507
|
* that asks only what the retired note holds that the kept one lacks. True
|
|
508
508
|
* only on a clean, empty answer (it caught 2 of 4 duplicates the judge got
|
|
509
509
|
* wrong, and held back none of 109 right ones).
|
|
@@ -658,17 +658,20 @@ async function judgeOne(ctx, candidate) {
|
|
|
658
658
|
}, ctx.opts.proposalsCtx);
|
|
659
659
|
ctx.retired.push(proposal.id);
|
|
660
660
|
ctx.perInitiatorProposed.add(candidate.initiator.ref);
|
|
661
|
-
// A
|
|
662
|
-
//
|
|
663
|
-
//
|
|
664
|
-
//
|
|
665
|
-
|
|
666
|
-
|
|
661
|
+
// A retirement the second look confirms loses nothing retires unattended:
|
|
662
|
+
// the triage drain accepts it under its usual applyMode. The retired side
|
|
663
|
+
// holds no claim of its own for any label (`decideRetirement` mints nothing
|
|
664
|
+
// else); a duplicate must also leave the kept side with none, while a
|
|
665
|
+
// subsumed or superseding successor holds more by definition. Replay
|
|
666
|
+
// precision 336/360 (duplicate 0.98, subsumed 0.92, supersedes 0.875,
|
|
667
|
+
// 2026-10-07); a duplicate alone was 109 of 111 safe on the owner's
|
|
668
|
+
// reviewed pairs (2026-10-04). Anything else waits for a person.
|
|
669
|
+
if ((verdict.relation !== "duplicate" || verdict.onlyInA.length + verdict.onlyInB.length === 0) &&
|
|
667
670
|
!continuityRisk &&
|
|
668
671
|
(await confirmNothingLost(ctx, retired, successor))) {
|
|
669
672
|
recordGateDecision(ctx.stashDir, proposal.id, {
|
|
670
673
|
outcome: "staged",
|
|
671
|
-
reason:
|
|
674
|
+
reason: verdict.relation,
|
|
672
675
|
gate: PAIR_PASS_GATE,
|
|
673
676
|
contentHash: proposalContentHash(proposal),
|
|
674
677
|
}, ctx.opts.proposalsCtx);
|
|
@@ -260,6 +260,39 @@ function loadPendingConsolidateProposalHashes(stashDir, proposalsCtx) {
|
|
|
260
260
|
}
|
|
261
261
|
return hashes;
|
|
262
262
|
}
|
|
263
|
+
/**
|
|
264
|
+
* Rejections decided before this are not a verdict on the memory's text: the
|
|
265
|
+
* 2026-08-02 and 2026-08-18 bulk audits rejected hundreds of promotions
|
|
266
|
+
* wholesale, and holding their bodies would skip good memories for good.
|
|
267
|
+
*/
|
|
268
|
+
const REJECTED_BODY_HOLD_FROM = "2026-09-29";
|
|
269
|
+
/**
|
|
270
|
+
* Body hashes of the memories whose promotion was rejected on review. A
|
|
271
|
+
* proposal minted with `promotionSourceHash` names the memory's raw body; an
|
|
272
|
+
* older one is hashed from its own body, which is the memory's unless
|
|
273
|
+
* sanitization changed it.
|
|
274
|
+
*/
|
|
275
|
+
function loadRejectedPromotionBodyHashes(stashDir, proposalsCtx) {
|
|
276
|
+
const hashes = new Set();
|
|
277
|
+
try {
|
|
278
|
+
for (const p of listProposalsReadOnly(stashDir, { status: "rejected", includeArchive: true }, proposalsCtx)) {
|
|
279
|
+
if (p.source !== "consolidate")
|
|
280
|
+
continue;
|
|
281
|
+
if ((p.review?.decidedAt ?? p.updatedAt) < REJECTED_BODY_HOLD_FROM)
|
|
282
|
+
continue;
|
|
283
|
+
try {
|
|
284
|
+
hashes.add(p.promotionSourceHash ?? contentHash(proposalContent(p), "body"));
|
|
285
|
+
}
|
|
286
|
+
catch {
|
|
287
|
+
// A malformed payload cannot hold a memory.
|
|
288
|
+
}
|
|
289
|
+
}
|
|
290
|
+
}
|
|
291
|
+
catch {
|
|
292
|
+
// Best-effort: a failed read never blocks judging.
|
|
293
|
+
}
|
|
294
|
+
return hashes;
|
|
295
|
+
}
|
|
263
296
|
/**
|
|
264
297
|
* Body hashes of the live knowledge assets, read from disk (the index may lag
|
|
265
298
|
* a just-written asset), so an accepted promotion is not proposed again.
|
|
@@ -469,6 +502,18 @@ export function inspectConsolidationPool(opts, stashDir, warnings, existingKnowl
|
|
|
469
502
|
return !isLedgerBlocked(row, nowIso, changedAt);
|
|
470
503
|
});
|
|
471
504
|
}
|
|
505
|
+
// A memory whose text a reviewer already rejected as a promotion waits for an edit, whatever it is called.
|
|
506
|
+
const rejectedBodies = loadRejectedPromotionBodyHashes(stashDir, opts.proposalsCtx);
|
|
507
|
+
if (rejectedBodies.size > 0) {
|
|
508
|
+
memories = memories.filter((memory) => {
|
|
509
|
+
try {
|
|
510
|
+
return !rejectedBodies.has(contentHash(fs.readFileSync(memory.filePath, "utf8"), "body"));
|
|
511
|
+
}
|
|
512
|
+
catch {
|
|
513
|
+
return true;
|
|
514
|
+
}
|
|
515
|
+
});
|
|
516
|
+
}
|
|
472
517
|
const judgedUnchanged = poolSize - memories.length;
|
|
473
518
|
// Only what retrieval returned or new material improve never processed (#986).
|
|
474
519
|
const retrievalScope = loadRetrievalScope({ proposalsCtx: opts.proposalsCtx, readOnly }, stashDir);
|
|
@@ -768,7 +813,13 @@ async function consolidate(opts, config, stashDir, startMs, stateDb) {
|
|
|
768
813
|
recordLedgerAttempt({ proposalsCtx: opts.proposalsCtx }, [...acc.judgedRefs]
|
|
769
814
|
.filter((ref) => !ctx.promotedSourceRefs.has(ref) &&
|
|
770
815
|
!acc.skipReasonByRef.get(ref)?.skips.some((skip) => skip.reason === "promote_create_failed"))
|
|
771
|
-
.map((ref) => ({
|
|
816
|
+
.map((ref) => ({
|
|
817
|
+
stashDir,
|
|
818
|
+
ref,
|
|
819
|
+
source: "consolidate",
|
|
820
|
+
outcome: "judged_no_action",
|
|
821
|
+
...bodyHashOf(ctx.memoryByRef.get(ref)),
|
|
822
|
+
})));
|
|
772
823
|
return makeConsolidateResult({
|
|
773
824
|
...summary(),
|
|
774
825
|
promoted: ctx.promoted,
|
|
@@ -783,6 +834,17 @@ async function consolidate(opts, config, stashDir, startMs, stateDb) {
|
|
|
783
834
|
},
|
|
784
835
|
});
|
|
785
836
|
}
|
|
837
|
+
/** The memory's current body hash as a ledger input field; empty when it cannot be read (the row then keeps its 7-day window). */
|
|
838
|
+
function bodyHashOf(memory) {
|
|
839
|
+
if (!memory)
|
|
840
|
+
return {};
|
|
841
|
+
try {
|
|
842
|
+
return { contentHash: contentHash(fs.readFileSync(memory.filePath, "utf8"), "body") };
|
|
843
|
+
}
|
|
844
|
+
catch {
|
|
845
|
+
return {};
|
|
846
|
+
}
|
|
847
|
+
}
|
|
786
848
|
/** The conceptId a ref maps to, or undefined for an invalid ref. */
|
|
787
849
|
function conceptIdForRef(ref) {
|
|
788
850
|
try {
|
|
@@ -2,25 +2,24 @@
|
|
|
2
2
|
// License, v. 2.0. If a copy of the MPL was not distributed with this
|
|
3
3
|
// file, You can obtain one at https://mozilla.org/MPL/2.0/.
|
|
4
4
|
/**
|
|
5
|
-
* Distill guards: related lessons
|
|
6
|
-
* overwrite
|
|
7
|
-
* proposal does not contradict the memories it came from.
|
|
5
|
+
* Distill guards: the related lessons, knowledge notes and skills shown to the
|
|
6
|
+
* writer so it does not repeat or overwrite them (CLS context), and a cheap check
|
|
7
|
+
* that a proposal does not contradict the memories it came from.
|
|
8
8
|
*/
|
|
9
9
|
export const DEFAULT_CLS_ADJACENT_COUNT = 3;
|
|
10
|
-
/** The CLS prompt section (each entry capped at
|
|
10
|
+
/** The CLS prompt section (each entry capped at 600 chars); empty when disabled (on unless `enabled: false`) or nothing is related. */
|
|
11
11
|
export function buildClsContext(adjacentItems, config) {
|
|
12
|
-
if (
|
|
12
|
+
if (config.enabled === false || adjacentItems.length === 0)
|
|
13
13
|
return "";
|
|
14
14
|
const lines = [
|
|
15
15
|
"",
|
|
16
|
-
"##
|
|
17
|
-
"The
|
|
18
|
-
"
|
|
19
|
-
"disagree with one, flag it as contradicted (do not ignore it).",
|
|
16
|
+
"## Related assets already in the library",
|
|
17
|
+
"The library already holds these lessons, knowledge notes and skills near this memory. They may be about another subject.",
|
|
18
|
+
"If one of them already states the rule the memory would give, answer NONE. Do not contradict or overwrite them.",
|
|
20
19
|
"",
|
|
21
20
|
];
|
|
22
21
|
for (const item of adjacentItems)
|
|
23
|
-
lines.push(`### ${item.ref}`, item.content.trim().slice(0,
|
|
22
|
+
lines.push(`### ${item.ref}`, item.content.trim().slice(0, 600), "");
|
|
24
23
|
return lines.join("\n");
|
|
25
24
|
}
|
|
26
25
|
/**
|