akm-cli 0.9.16-alpha.2 → 0.9.17-alpha.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +495 -1
- package/dist/assets/prompts/consolidate-system.md +4 -11
- package/dist/assets/prompts/graph-extract-user-prompt.md +5 -5
- package/dist/commands/health/accept-rate.js +6 -0
- package/dist/commands/health/checks.js +54 -0
- package/dist/commands/health/improve-metrics.js +1 -5
- package/dist/commands/health/report-view-model.js +0 -1
- package/dist/commands/health.js +10 -0
- package/dist/commands/improve/consolidate/chunking.js +19 -35
- package/dist/commands/improve/consolidate/merge.js +6 -9
- package/dist/commands/improve/consolidate.js +104 -91
- package/dist/commands/improve/distill/promote-memory.js +40 -2
- package/dist/commands/improve/distill/quality-gate.js +186 -23
- package/dist/commands/improve/distill.js +42 -8
- package/dist/commands/improve/eligibility.js +13 -3
- package/dist/commands/improve/improve-cli.js +32 -9
- package/dist/commands/improve/improve-strategies.js +23 -1
- package/dist/commands/improve/improve.js +121 -84
- package/dist/commands/improve/loop-stages.js +241 -108
- package/dist/commands/improve/preparation.js +50 -17
- package/dist/commands/improve/reflect.js +16 -5
- package/dist/commands/improve/shared.js +0 -10
- package/dist/commands/proposal/drain.js +79 -10
- package/dist/commands/proposal/proposal-types.js +21 -0
- package/dist/commands/proposal/repository.js +108 -29
- package/dist/core/asset/frontmatter.js +106 -1
- package/dist/core/config/schema/improve-processes.js +29 -2
- package/dist/core/improve-result.js +9 -0
- package/dist/core/paths.js +7 -0
- package/dist/indexer/ensure-index.js +52 -7
- package/dist/indexer/graph/graph-extraction.js +82 -8
- package/dist/indexer/passes/memory-inference.js +16 -1
- package/dist/llm/client.js +16 -2
- package/dist/llm/graph-extract.js +162 -18
- package/dist/output/html-render.js +2 -1
- package/dist/output/stdout.js +24 -0
- package/dist/output/text.js +4 -3
- package/dist/scripts/akm-migrate-node.js +20 -4
- package/dist/scripts/akm-migrate.js +20 -4
- package/dist/storage/repositories/index-entries-repository.js +43 -0
- package/dist/storage/repositories/proposals-repository.js +4 -1
- package/dist/storage/state-db-integrity.js +123 -0
- package/dist/workflows/program/schema.js +1 -0
- package/docs/reference/cli.md +4 -3
- package/docs/reference/data-and-telemetry.md +1 -0
- package/package.json +1 -1
- package/schemas/akm-config.json +44 -0
- package/schemas/akm-workflow.json +1 -0
- package/dist/commands/improve/eval-cases.js +0 -52
|
@@ -13,6 +13,7 @@ import { isLlmCredentialAvailable, resolveEngine, } from "../../integrations/age
|
|
|
13
13
|
import { executionEngineDefinitionsFromConfig } from "../../integrations/agent/execution-definitions.js";
|
|
14
14
|
import { loadModelMap, mergeModelMapLayers, parseModelMapLayer, readInstalledModelMapText, resolveModelMapAlias, userModelMapPath, } from "../../integrations/agent/model-map.js";
|
|
15
15
|
import { probeEndpointOnce } from "../../llm/client.js";
|
|
16
|
+
import { STATE_DB_FREELIST_WARN_RATIO, } from "../../storage/state-db-integrity.js";
|
|
16
17
|
import { listKeys } from "../env/env.js";
|
|
17
18
|
import { resolveImprovePlan } from "../improve/improve-strategies.js";
|
|
18
19
|
import { ENGINE_LAST_USED_LOOKBACK_DAYS } from "./engine-usage.js";
|
|
@@ -868,6 +869,59 @@ export const HEALTH_CHECKS = [
|
|
|
868
869
|
evidence: { path: ctx.stateDbPath, durationMs: ctx.probe.durationMs },
|
|
869
870
|
}),
|
|
870
871
|
},
|
|
872
|
+
{
|
|
873
|
+
// R0: nothing looked at state.db's own SQLite-level integrity before
|
|
874
|
+
// this — the round-trip probe above only proves one row can be appended
|
|
875
|
+
// and read back, which stays true on a database that fails
|
|
876
|
+
// `PRAGMA quick_check` elsewhere (corrupt indexes, out-of-order rowids).
|
|
877
|
+
// Also reports the freelist ratio (fraction of pages VACUUM could
|
|
878
|
+
// reclaim) so a bloated-but-uncorrupted file is visible as a warning
|
|
879
|
+
// rather than silence.
|
|
880
|
+
name: "state-db-integrity",
|
|
881
|
+
channel: "hard",
|
|
882
|
+
run: (ctx) => {
|
|
883
|
+
const { ok, lines, error } = ctx.stateDbIntegrity;
|
|
884
|
+
const { ratio: freelistRatio, error: freelistError } = ctx.stateDbFreelist;
|
|
885
|
+
if (!ok) {
|
|
886
|
+
const detail = error ?? lines.join("; ");
|
|
887
|
+
return {
|
|
888
|
+
name: "state-db-integrity",
|
|
889
|
+
kind: "deterministic",
|
|
890
|
+
status: "fail",
|
|
891
|
+
confidence: "high",
|
|
892
|
+
message: `state.db failed PRAGMA quick_check: ${detail}. Repair: back up state.db, then run ` +
|
|
893
|
+
`sqlite3 state.db ".dump" | sqlite3 state.new.db, verify state.new.db passes quick_check, and swap it in.`,
|
|
894
|
+
evidence: { path: ctx.stateDbPath, lines, freelistRatio },
|
|
895
|
+
};
|
|
896
|
+
}
|
|
897
|
+
if (freelistError) {
|
|
898
|
+
return {
|
|
899
|
+
name: "state-db-integrity",
|
|
900
|
+
kind: "deterministic",
|
|
901
|
+
status: "fail",
|
|
902
|
+
confidence: "high",
|
|
903
|
+
message: `state.db passed PRAGMA quick_check, but reading its freelist/page-count failed: ${freelistError}.`,
|
|
904
|
+
evidence: { path: ctx.stateDbPath, lines, freelistError },
|
|
905
|
+
};
|
|
906
|
+
}
|
|
907
|
+
const freelistWarn = freelistRatio > STATE_DB_FREELIST_WARN_RATIO;
|
|
908
|
+
return {
|
|
909
|
+
name: "state-db-integrity",
|
|
910
|
+
kind: "deterministic",
|
|
911
|
+
status: freelistWarn ? "warn" : "pass",
|
|
912
|
+
confidence: "high",
|
|
913
|
+
message: freelistWarn
|
|
914
|
+
? `state.db passed PRAGMA quick_check, but ${(freelistRatio * 100).toFixed(1)}% of its pages are free (reclaimable by VACUUM).`
|
|
915
|
+
: "state.db passed PRAGMA quick_check.",
|
|
916
|
+
evidence: {
|
|
917
|
+
path: ctx.stateDbPath,
|
|
918
|
+
freelistCount: ctx.stateDbFreelist.freelistCount,
|
|
919
|
+
pageCount: ctx.stateDbFreelist.pageCount,
|
|
920
|
+
freelistRatio,
|
|
921
|
+
},
|
|
922
|
+
};
|
|
923
|
+
},
|
|
924
|
+
},
|
|
871
925
|
{
|
|
872
926
|
name: "state-db-migrations",
|
|
873
927
|
channel: "hard",
|
|
@@ -92,7 +92,6 @@ function createUnknownImproveMetrics() {
|
|
|
92
92
|
autoAccept: { promoted: 0, validationFailed: 0 },
|
|
93
93
|
reflectsWithErrorContext: 0,
|
|
94
94
|
coverageGapCount: 0,
|
|
95
|
-
evalCasesWritten: 0,
|
|
96
95
|
deadUrlCount: 0,
|
|
97
96
|
deadUrlsChecked: 0,
|
|
98
97
|
deadUrlsTotal: 0,
|
|
@@ -344,14 +343,13 @@ function applyDistillSkippedAggregate(metrics, result) {
|
|
|
344
343
|
}
|
|
345
344
|
}
|
|
346
345
|
}
|
|
347
|
-
/** autoAccept + the small envelope-level counters (reflectsWithErrorContext, coverageGaps,
|
|
346
|
+
/** autoAccept + the small envelope-level counters (reflectsWithErrorContext, coverageGaps, deadUrls). */
|
|
348
347
|
function applyMiscCounters(metrics, result) {
|
|
349
348
|
metrics.autoAccept.promoted += toFiniteNumber(result.gateAutoAcceptedCount);
|
|
350
349
|
metrics.autoAccept.validationFailed += toFiniteNumber(result.gateAutoAcceptFailedCount);
|
|
351
350
|
metrics.reflectsWithErrorContext += toFiniteNumber(result.reflectsWithErrorContext);
|
|
352
351
|
if (Array.isArray(result.coverageGaps))
|
|
353
352
|
metrics.coverageGapCount += result.coverageGaps.length;
|
|
354
|
-
metrics.evalCasesWritten += toFiniteNumber(result.evalCasesWritten);
|
|
355
353
|
if (Array.isArray(result.deadUrls))
|
|
356
354
|
metrics.deadUrlCount += result.deadUrls.length;
|
|
357
355
|
const deadUrlCoverage = result.deadUrlCoverage;
|
|
@@ -593,7 +591,6 @@ function mergeImproveMetrics(dst, src) {
|
|
|
593
591
|
dst.autoAccept.validationFailed += src.autoAccept.validationFailed;
|
|
594
592
|
dst.reflectsWithErrorContext += src.reflectsWithErrorContext;
|
|
595
593
|
dst.coverageGapCount += src.coverageGapCount;
|
|
596
|
-
dst.evalCasesWritten += src.evalCasesWritten;
|
|
597
594
|
dst.deadUrlCount += src.deadUrlCount;
|
|
598
595
|
dst.deadUrlsChecked += src.deadUrlsChecked;
|
|
599
596
|
dst.deadUrlsTotal += src.deadUrlsTotal;
|
|
@@ -801,7 +798,6 @@ export function projectImproveRunSummary(row, wallTimeMs, taskId) {
|
|
|
801
798
|
memoryInference: perRow.memoryInference,
|
|
802
799
|
graphExtraction: perRow.graphExtraction,
|
|
803
800
|
reflectsWithErrorContext: perRow.reflectsWithErrorContext,
|
|
804
|
-
evalCasesWritten: perRow.evalCasesWritten,
|
|
805
801
|
orphansPurged,
|
|
806
802
|
lintFixed,
|
|
807
803
|
lintFlagged,
|
package/dist/commands/health.js
CHANGED
|
@@ -18,6 +18,7 @@ import { countImproveRunsSince } from "../storage/repositories/improve-runs-repo
|
|
|
18
18
|
import { closeDatabase, openReadonlyExistingDatabase } from "../storage/repositories/index-connection.js";
|
|
19
19
|
import { getAllEntries } from "../storage/repositories/index-entries-repository.js";
|
|
20
20
|
import { queryTaskHistory } from "../storage/repositories/task-history-repository.js";
|
|
21
|
+
import { getStateDbFreelistInfo, runStateDbQuickCheck } from "../storage/state-db-integrity.js";
|
|
21
22
|
import { pkgVersion } from "../version.js";
|
|
22
23
|
import { collectImproveAdvisories } from "./health/advisories.js";
|
|
23
24
|
import { HEALTH_CHECKS, probeActiveImproveStrategy, runHealthEngineProbes, runPendingStateMigrationsCheck, SESSION_EXTRACTION_LEDGER_WINDOW_DAYS, } from "./health/checks.js";
|
|
@@ -124,6 +125,11 @@ function gatherTaskHistoryPhase(db, logsDb, since, stateDbPath, now) {
|
|
|
124
125
|
const requiredTables = ["events", "proposals", "schema_migrations", "task_history"];
|
|
125
126
|
const missingTables = requiredTables.filter((name) => !tableNames.includes(name));
|
|
126
127
|
const probe = probeStateDbRoundTrip(stateDbPath);
|
|
128
|
+
// R0: read-only, independent of the round-trip probe above — quick_check
|
|
129
|
+
// catches corruption a successful append/read cannot (out-of-order rowids,
|
|
130
|
+
// bad index entry counts), and the freelist reading is purely informational.
|
|
131
|
+
const stateDbIntegrity = runStateDbQuickCheck(stateDbPath);
|
|
132
|
+
const stateDbFreelist = getStateDbFreelistInfo(stateDbPath);
|
|
127
133
|
const taskRows = queryTaskHistory(db, { since });
|
|
128
134
|
const { withLogs: taskRowsWithLogs, backed: existingLogRows } = partitionLogBackedRows(taskRows, logsDb);
|
|
129
135
|
const failedTaskRows = taskRows.filter((row) => row.status === "failed");
|
|
@@ -145,6 +151,8 @@ function gatherTaskHistoryPhase(db, logsDb, since, stateDbPath, now) {
|
|
|
145
151
|
tableNames,
|
|
146
152
|
missingTables,
|
|
147
153
|
probe,
|
|
154
|
+
stateDbIntegrity,
|
|
155
|
+
stateDbFreelist,
|
|
148
156
|
taskRowCount: taskRows.length,
|
|
149
157
|
taskRowsWithLogsCount: taskRowsWithLogs.length,
|
|
150
158
|
existingLogRowsCount: existingLogRows.length,
|
|
@@ -580,6 +588,8 @@ export async function akmHealth(options = {}) {
|
|
|
580
588
|
tableNames,
|
|
581
589
|
missingTables,
|
|
582
590
|
probe,
|
|
591
|
+
stateDbIntegrity: taskHistory.stateDbIntegrity,
|
|
592
|
+
stateDbFreelist: taskHistory.stateDbFreelist,
|
|
583
593
|
taskRowCount: taskHistory.taskRowCount,
|
|
584
594
|
taskFailRate: taskHistory.taskFailRate,
|
|
585
595
|
taskRowsWithLogsCount: taskHistory.taskRowsWithLogsCount,
|
|
@@ -6,7 +6,7 @@
|
|
|
6
6
|
// no orchestrator coupling.
|
|
7
7
|
import fs from "node:fs";
|
|
8
8
|
import { parseFrontmatter } from "../../../core/asset/frontmatter.js";
|
|
9
|
-
import { cacheHash } from "../content-hash.js";
|
|
9
|
+
import { cacheHash, stripFrontmatterBody } from "../content-hash.js";
|
|
10
10
|
/**
|
|
11
11
|
* Conservative chars-per-token estimate used when computing prompt budgets.
|
|
12
12
|
* English text averages roughly 4 chars/token for most LLM tokenizers. We use
|
|
@@ -66,25 +66,19 @@ export function computeSafeChunkSize(contextLength, bodyTruncation, maxChunkSize
|
|
|
66
66
|
/**
|
|
67
67
|
* Build the per-chunk user prompt fed to the consolidate LLM.
|
|
68
68
|
*
|
|
69
|
-
* Each memory is annotated with two flags
|
|
70
|
-
*
|
|
71
|
-
*
|
|
72
|
-
* forbids proposing delete. ~60 wasted LLM verdicts/4h on this user's
|
|
73
|
-
* stack before this annotation.
|
|
69
|
+
* Each memory is annotated with two flags:
|
|
70
|
+
* - `(captureMode: hot)` — user-explicit memory, surfaced to the model as
|
|
71
|
+
* provenance context.
|
|
74
72
|
* - `(already queued)` — the memory's body hash matches a pending
|
|
75
|
-
* consolidate proposal; system prompt rule
|
|
76
|
-
* promote
|
|
73
|
+
* consolidate proposal; the system prompt's PROMOTE rule forbids
|
|
74
|
+
* proposing promote for these. ~107/4h before this annotation.
|
|
77
75
|
*
|
|
78
76
|
* Both annotations are visible to the LLM. `pendingProposalBodyHashes`
|
|
79
77
|
* is precomputed once per run by `loadPendingConsolidateProposalHashes`
|
|
80
78
|
* so the cost stays O(memories) inside the chunk loop.
|
|
81
79
|
*/
|
|
82
|
-
export function buildChunkPrompt(sourceName, memories, chunkIndex, totalChunks, bodyTruncation, pendingProposalBodyHashes = new Set()
|
|
83
|
-
const start = memories[0] ? `memories/${memories[0].name}` : "";
|
|
84
|
-
const lastMemory = memories[memories.length - 1];
|
|
85
|
-
const end = lastMemory ? `memories/${lastMemory.name}` : "";
|
|
80
|
+
export function buildChunkPrompt(sourceName, memories, chunkIndex, totalChunks, bodyTruncation, pendingProposalBodyHashes = new Set()) {
|
|
86
81
|
const annotationsByIndex = [];
|
|
87
|
-
const hotRefs = [];
|
|
88
82
|
for (const m of memories) {
|
|
89
83
|
let body = "";
|
|
90
84
|
try {
|
|
@@ -93,41 +87,31 @@ export function buildChunkPrompt(sourceName, memories, chunkIndex, totalChunks,
|
|
|
93
87
|
catch {
|
|
94
88
|
body = "(unreadable)";
|
|
95
89
|
}
|
|
90
|
+
// Hot/queued detection stays on the raw body — frontmatter is exactly
|
|
91
|
+
// what parseFrontmatter and the pending-proposal hash domain need.
|
|
96
92
|
const parsed = parseFrontmatter(body);
|
|
97
93
|
const isHot = parsed.data.captureMode === "hot";
|
|
98
94
|
// Use cacheHash (case-preserving stripped body) to match the domain used
|
|
99
95
|
// by loadPendingConsolidateProposalHashes and the body-embedding cache.
|
|
100
96
|
const bodyHash = cacheHash(body);
|
|
101
97
|
const isAlreadyQueued = pendingProposalBodyHashes.has(bodyHash);
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
98
|
+
// The excerpt shown to the LLM is the body ONLY — frontmatter can run
|
|
99
|
+
// longer than bodyTruncation (~21% of memories), which used to leave the
|
|
100
|
+
// model judging metadata instead of content. stripFrontmatterBody falls
|
|
101
|
+
// back to raw.trim() on unparseable/absent frontmatter, so this is a
|
|
102
|
+
// no-op for the "(unreadable)" placeholder.
|
|
103
|
+
const excerpt = stripFrontmatterBody(body);
|
|
104
|
+
annotationsByIndex.push({ isHot, isAlreadyQueued, excerpt });
|
|
105
105
|
}
|
|
106
106
|
const lines = [
|
|
107
107
|
`Source: ${sourceName}`,
|
|
108
|
-
`Chunk ${chunkIndex + 1} of ${totalChunks}
|
|
108
|
+
`Chunk ${chunkIndex + 1} of ${totalChunks} (${memories.length} memories):`,
|
|
109
109
|
"",
|
|
110
110
|
];
|
|
111
|
-
if (standardsContext.trim()) {
|
|
112
|
-
lines.push("Standards to follow (the rulebook for this target):");
|
|
113
|
-
lines.push(standardsContext.trim());
|
|
114
|
-
lines.push("");
|
|
115
|
-
}
|
|
116
|
-
// Top-of-prompt protection block for hot refs. Neutral phrasing — avoid
|
|
117
|
-
// op-words like "promote", "merge", "contradict" so the model doesn't
|
|
118
|
-
// accidentally treat the warning as a hint to use that op elsewhere
|
|
119
|
-
// (variant B leaked the word "contradict" into the control sample
|
|
120
|
-
// during the diagnostic).
|
|
121
|
-
if (hotRefs.length > 0) {
|
|
122
|
-
lines.push("⛔ DO NOT propose any `delete` operation for these refs — they are user-explicit (captureMode: hot) and the downstream guard refuses them regardless. Proposing delete for any of these only wastes tokens.");
|
|
123
|
-
for (const ref of hotRefs)
|
|
124
|
-
lines.push(` - ${ref}`);
|
|
125
|
-
lines.push("");
|
|
126
|
-
}
|
|
127
111
|
for (let i = 0; i < memories.length; i++) {
|
|
128
112
|
const m = memories[i];
|
|
129
113
|
// `annotationsByIndex` has exactly one entry per memory (built in the loop above).
|
|
130
|
-
const { isHot, isAlreadyQueued,
|
|
114
|
+
const { isHot, isAlreadyQueued, excerpt } = annotationsByIndex[i];
|
|
131
115
|
const annotations = [];
|
|
132
116
|
if (isHot)
|
|
133
117
|
annotations.push("captureMode: hot");
|
|
@@ -138,7 +122,7 @@ export function buildChunkPrompt(sourceName, memories, chunkIndex, totalChunks,
|
|
|
138
122
|
lines.push(`Description: ${m.description || "(none)"}`);
|
|
139
123
|
lines.push(`Tags: ${m.tags.length > 0 ? m.tags.join(", ") : "(none)"}`);
|
|
140
124
|
lines.push("---");
|
|
141
|
-
lines.push(
|
|
125
|
+
lines.push(excerpt.slice(0, bodyTruncation));
|
|
142
126
|
lines.push("");
|
|
143
127
|
}
|
|
144
128
|
return lines.join("\n");
|
|
@@ -1,22 +1,19 @@
|
|
|
1
1
|
// This Source Code Form is subject to the terms of the Mozilla Public
|
|
2
2
|
// License, v. 2.0. If a copy of the MPL was not distributed with this
|
|
3
3
|
// file, You can obtain one at https://mozilla.org/MPL/2.0/.
|
|
4
|
+
// R12a: merge/delete/contradict are no longer requested (see
|
|
5
|
+
// CONSOLIDATE_PLAN_JSON_SCHEMA / consolidate-system.md) — only promote is
|
|
6
|
+
// ever executed. An op of one of those retired shapes — e.g. from a model
|
|
7
|
+
// that ignores the schema — is rejected here so it degrades to the generic
|
|
8
|
+
// "skipping invalid operation" warning in the chunk-judge loop rather than
|
|
9
|
+
// being treated as an actionable (if advisory) plan entry.
|
|
4
10
|
export function isValidOp(op) {
|
|
5
11
|
if (typeof op !== "object" || op === null)
|
|
6
12
|
return false;
|
|
7
13
|
const o = op;
|
|
8
|
-
if (o.op === "merge") {
|
|
9
|
-
return typeof o.primary === "string" && Array.isArray(o.secondaries);
|
|
10
|
-
}
|
|
11
|
-
if (o.op === "delete") {
|
|
12
|
-
return typeof o.ref === "string";
|
|
13
|
-
}
|
|
14
14
|
if (o.op === "promote") {
|
|
15
15
|
return typeof o.ref === "string" && typeof o.knowledgeRef === "string";
|
|
16
16
|
}
|
|
17
|
-
if (o.op === "contradict") {
|
|
18
|
-
return typeof o.ref === "string" && typeof o.contradictedByRef === "string";
|
|
19
|
-
}
|
|
20
17
|
return false;
|
|
21
18
|
}
|
|
22
19
|
export function mergePlans(chunks, knownRefs) {
|
|
@@ -11,7 +11,6 @@ import { parseFrontmatter } from "../../core/asset/frontmatter.js";
|
|
|
11
11
|
import { conceptIdFromTypeName, displayRef, parseRefInput } from "../../core/asset/resolve-ref.js";
|
|
12
12
|
import { getImproveProcessConfig, loadConfig } from "../../core/config/config.js";
|
|
13
13
|
import { parseEmbeddedJsonResponse } from "../../core/parse.js";
|
|
14
|
-
import { resolveStandardsContext } from "../../core/standards/resolve-standards-context.js";
|
|
15
14
|
import { openStateDatabase } from "../../core/state-db.js";
|
|
16
15
|
import { parseSinceToIsoLenient } from "../../core/time.js";
|
|
17
16
|
import { warn, warnVerbose } from "../../core/warn.js";
|
|
@@ -48,15 +47,18 @@ const CONSOLIDATE_SYSTEM_PROMPT = consolidateSystemPrompt;
|
|
|
48
47
|
/**
|
|
49
48
|
* JSON Schema for structured consolidate plans (PR 1 of the asset-writers
|
|
50
49
|
* decision — see knowledge/projects/akm/asset-writers-investigation/00-synthesis).
|
|
51
|
-
* Mirrors the {ops[]
|
|
52
|
-
*
|
|
53
|
-
*
|
|
54
|
-
*
|
|
50
|
+
* Mirrors the {ops[]} shape currently described in CONSOLIDATE_SYSTEM_PROMPT.
|
|
51
|
+
* Providers with `supportsJsonSchema: true` enforce the shape upstream so the
|
|
52
|
+
* chunk-level "invalid plan from AI — skipping" branch in `runConsolidate`
|
|
53
|
+
* becomes unreachable on schema-honouring providers.
|
|
55
54
|
*
|
|
56
|
-
*
|
|
57
|
-
*
|
|
58
|
-
*
|
|
59
|
-
* working as a fallback parser for
|
|
55
|
+
* Promote-only (R12a): `merge`/`delete`/`contradict` were advisory-only — the
|
|
56
|
+
* apply loop only ever executed `promote` — and cost 21-30k completion tokens
|
|
57
|
+
* per run for output nothing acted on. `warnings` is dropped for the same
|
|
58
|
+
* reason. `parseEmbeddedJsonResponse` keeps working as a fallback parser for
|
|
59
|
+
* providers that ignore the schema; `isValidOp` rejects any op shape other
|
|
60
|
+
* than `promote` so a non-compliant response degrades to a skipped-op warning
|
|
61
|
+
* rather than being treated as an actionable plan.
|
|
60
62
|
*/
|
|
61
63
|
export const CONSOLIDATE_PLAN_JSON_SCHEMA = {
|
|
62
64
|
type: "object",
|
|
@@ -65,70 +67,21 @@ export const CONSOLIDATE_PLAN_JSON_SCHEMA = {
|
|
|
65
67
|
properties: {
|
|
66
68
|
operations: {
|
|
67
69
|
type: "array",
|
|
68
|
-
description: "Ordered list of
|
|
70
|
+
description: "Ordered list of promote operations the planner proposes.",
|
|
69
71
|
items: {
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
maxItems: 1,
|
|
82
|
-
items: { type: "string", minLength: 1 },
|
|
83
|
-
},
|
|
84
|
-
mergeStrategy: { type: "string", minLength: 1 },
|
|
85
|
-
confidence: { type: "number", minimum: 0, maximum: 1 },
|
|
86
|
-
},
|
|
87
|
-
},
|
|
88
|
-
{
|
|
89
|
-
type: "object",
|
|
90
|
-
required: ["op", "ref", "reason"],
|
|
91
|
-
additionalProperties: false,
|
|
92
|
-
properties: {
|
|
93
|
-
op: { type: "string", enum: ["delete"] },
|
|
94
|
-
ref: { type: "string", minLength: 1 },
|
|
95
|
-
reason: { type: "string", minLength: 1 },
|
|
96
|
-
confidence: { type: "number", minimum: 0, maximum: 1 },
|
|
97
|
-
},
|
|
98
|
-
},
|
|
99
|
-
{
|
|
100
|
-
type: "object",
|
|
101
|
-
required: ["op", "ref", "knowledgeRef", "reason"],
|
|
102
|
-
additionalProperties: false,
|
|
103
|
-
properties: {
|
|
104
|
-
op: { type: "string", enum: ["promote"] },
|
|
105
|
-
ref: { type: "string", minLength: 1 },
|
|
106
|
-
knowledgeRef: { type: "string", minLength: 1 },
|
|
107
|
-
reason: { type: "string", minLength: 1 },
|
|
108
|
-
description: { type: "string" },
|
|
109
|
-
confidence: { type: "number", minimum: 0, maximum: 1 },
|
|
110
|
-
},
|
|
111
|
-
},
|
|
112
|
-
{
|
|
113
|
-
type: "object",
|
|
114
|
-
required: ["op", "ref", "contradictedByRef", "reason"],
|
|
115
|
-
additionalProperties: false,
|
|
116
|
-
properties: {
|
|
117
|
-
op: { type: "string", enum: ["contradict"] },
|
|
118
|
-
ref: { type: "string", minLength: 1 },
|
|
119
|
-
contradictedByRef: { type: "string", minLength: 1 },
|
|
120
|
-
reason: { type: "string", minLength: 1 },
|
|
121
|
-
confidence: { type: "number", minimum: 0, maximum: 1 },
|
|
122
|
-
},
|
|
123
|
-
},
|
|
124
|
-
],
|
|
72
|
+
type: "object",
|
|
73
|
+
required: ["op", "ref", "knowledgeRef", "reason"],
|
|
74
|
+
additionalProperties: false,
|
|
75
|
+
properties: {
|
|
76
|
+
op: { type: "string", enum: ["promote"] },
|
|
77
|
+
ref: { type: "string", minLength: 1 },
|
|
78
|
+
knowledgeRef: { type: "string", minLength: 1 },
|
|
79
|
+
reason: { type: "string", minLength: 1, maxLength: 200 },
|
|
80
|
+
description: { type: "string" },
|
|
81
|
+
confidence: { type: "number", minimum: 0, maximum: 1 },
|
|
82
|
+
},
|
|
125
83
|
},
|
|
126
84
|
},
|
|
127
|
-
warnings: {
|
|
128
|
-
type: "array",
|
|
129
|
-
description: "Optional list of human-readable concerns the planner wants to surface.",
|
|
130
|
-
items: { type: "string" },
|
|
131
|
-
},
|
|
132
85
|
},
|
|
133
86
|
};
|
|
134
87
|
async function clusterMemoriesBySimilarity(memories, config, stateDb, signal) {
|
|
@@ -574,7 +527,7 @@ function resolveConsolidationSourceOwner(opts, stashDir) {
|
|
|
574
527
|
* Read and narrow the exact pool the live pass consumes, without embedding,
|
|
575
528
|
* LLM, proposal, event, or asset writes. Used by both preview and execution.
|
|
576
529
|
*/
|
|
577
|
-
export function inspectConsolidationPool(opts, stashDir, warnings, access) {
|
|
530
|
+
export function inspectConsolidationPool(opts, stashDir, warnings, existingKnowledgeBodyHashes = new Set(), access) {
|
|
578
531
|
const readOnly = access?.readOnly === true;
|
|
579
532
|
const sourceOwner = resolveConsolidationSourceOwner(opts, stashDir);
|
|
580
533
|
let memories = loadMemoriesForSource(sourceOwner, warnings, readOnly);
|
|
@@ -591,6 +544,13 @@ export function inspectConsolidationPool(opts, stashDir, warnings, access) {
|
|
|
591
544
|
if (opts.limit === undefined && memories.length > 150) {
|
|
592
545
|
warnings.push(`Consolidation: pool has ${memories.length} memories and no limit is set. Consider adding a limit to your consolidate config to prevent timeouts on slow LLM endpoints.`);
|
|
593
546
|
}
|
|
547
|
+
// Drop already-promoted memories before the limit cap selects its window,
|
|
548
|
+
// so the cap picks from memories the run can actually act on (R2-1: with
|
|
549
|
+
// the pre-filter applied afterward, the cap's oldest-modified-first window
|
|
550
|
+
// was drawn from the unfiltered pool, re-selecting and re-dropping the same
|
|
551
|
+
// permanently-undeletable duplicates on every run).
|
|
552
|
+
const { memories: prefiltered, prefilteredAlreadyPromoted } = prefilterAlreadyPromotedMemories(memories, existingKnowledgeBodyHashes);
|
|
553
|
+
memories = prefiltered;
|
|
594
554
|
if (opts.limit !== undefined && memories.length > opts.limit) {
|
|
595
555
|
const mtimeOf = (memory) => {
|
|
596
556
|
try {
|
|
@@ -605,18 +565,60 @@ export function inspectConsolidationPool(opts, stashDir, warnings, access) {
|
|
|
605
565
|
warnings.push(`Consolidation: pool capped at ${opts.limit} of ${memories.length} memories (limit option, oldest-modified first).`);
|
|
606
566
|
memories = memories.slice(0, opts.limit);
|
|
607
567
|
}
|
|
608
|
-
return { poolSize, candidatePoolSize: memories.length, dedupPoolSize, memories };
|
|
568
|
+
return { poolSize, candidatePoolSize: memories.length, dedupPoolSize, memories, prefilteredAlreadyPromoted };
|
|
569
|
+
}
|
|
570
|
+
/**
|
|
571
|
+
* Drop memories whose body already exists verbatim in `knowledge/` before any
|
|
572
|
+
* chunking or LLM work. Hashes the raw file with `cacheHash` (case-preserving
|
|
573
|
+
* stripped body), the same way `loadExistingKnowledgeBodyHashes` built
|
|
574
|
+
* `existingKnowledgeBodyHashes` — so the two sides of the comparison are
|
|
575
|
+
* computed identically. (H1: the post-LLM check, `shouldSkipPromotionBody-
|
|
576
|
+
* Duplicate`, used to hash `cacheHash(parseFrontmatter(memoryContent).content
|
|
577
|
+
* .trim())` — double-stripping frontmatter and diverging from this pre-filter
|
|
578
|
+
* for a memory body starting with its own `---` block. It now hashes
|
|
579
|
+
* `cacheHash(memoryContent)` directly, matching this single-strip domain.)
|
|
580
|
+
* An unreadable memory is kept (fail-safe: let the later passes surface the
|
|
581
|
+
* read error).
|
|
582
|
+
*/
|
|
583
|
+
function prefilterAlreadyPromotedMemories(memories, existingKnowledgeBodyHashes) {
|
|
584
|
+
if (existingKnowledgeBodyHashes.size === 0)
|
|
585
|
+
return { memories, prefilteredAlreadyPromoted: 0 };
|
|
586
|
+
const kept = [];
|
|
587
|
+
let prefilteredAlreadyPromoted = 0;
|
|
588
|
+
for (const memory of memories) {
|
|
589
|
+
let raw;
|
|
590
|
+
try {
|
|
591
|
+
raw = fs.readFileSync(memory.filePath, "utf8");
|
|
592
|
+
}
|
|
593
|
+
catch {
|
|
594
|
+
kept.push(memory);
|
|
595
|
+
continue;
|
|
596
|
+
}
|
|
597
|
+
if (existingKnowledgeBodyHashes.has(cacheHash(raw))) {
|
|
598
|
+
prefilteredAlreadyPromoted++;
|
|
599
|
+
}
|
|
600
|
+
else {
|
|
601
|
+
kept.push(memory);
|
|
602
|
+
}
|
|
603
|
+
}
|
|
604
|
+
return { memories: kept, prefilteredAlreadyPromoted };
|
|
609
605
|
}
|
|
610
606
|
/**
|
|
611
607
|
* Pass 1 — narrow the memory pool before any LLM work: drop stale DB entries,
|
|
612
|
-
* apply incremental-since narrowing,
|
|
613
|
-
*
|
|
614
|
-
*
|
|
615
|
-
*
|
|
608
|
+
* apply incremental-since narrowing, pre-filter memories already promoted
|
|
609
|
+
* verbatim into `knowledge/` (R5 (b) — discovered previously only after the
|
|
610
|
+
* LLM chunk call, paying for the judgement on ~84% of the pool just to skip
|
|
611
|
+
* it), and cap to `opts.limit` (oldest-modified first, drawn from the
|
|
612
|
+
* pre-filtered pool — R2-1). Returns an early envelope when the pool empties
|
|
613
|
+
* at any stage; otherwise returns the narrowed pool and the state the
|
|
614
|
+
* plan/apply passes consume.
|
|
616
615
|
*/
|
|
617
|
-
async function narrowConsolidationPool(opts, stashDir, startMs, warnings) {
|
|
618
|
-
const snapshot = inspectConsolidationPool(opts, stashDir, warnings);
|
|
619
|
-
const memories = snapshot
|
|
616
|
+
async function narrowConsolidationPool(opts, stashDir, startMs, warnings, existingKnowledgeBodyHashes) {
|
|
617
|
+
const snapshot = inspectConsolidationPool(opts, stashDir, warnings, existingKnowledgeBodyHashes);
|
|
618
|
+
const { memories, prefilteredAlreadyPromoted } = snapshot;
|
|
619
|
+
if (prefilteredAlreadyPromoted > 0) {
|
|
620
|
+
warnings.push(`Consolidation: pre-filtered ${prefilteredAlreadyPromoted} memor${prefilteredAlreadyPromoted === 1 ? "y" : "ies"} whose body already exists verbatim in knowledge/ before chunking.`);
|
|
621
|
+
}
|
|
620
622
|
// (The former WS-3b Step 0a homeostatic demotion pass was removed — R4:
|
|
621
623
|
// it was default-off and self-undoing (the next salience recompute
|
|
622
624
|
// unconditionally overwrote the demoted values). Continuous decay now lives
|
|
@@ -629,10 +631,11 @@ async function narrowConsolidationPool(opts, stashDir, startMs, warnings) {
|
|
|
629
631
|
target: opts.target ?? stashDir,
|
|
630
632
|
warnings,
|
|
631
633
|
durationMs: Date.now() - startMs,
|
|
634
|
+
prefilteredAlreadyPromoted,
|
|
632
635
|
}),
|
|
633
636
|
};
|
|
634
637
|
}
|
|
635
|
-
return { done: false, memories, dedupPoolSize: snapshot.dedupPoolSize };
|
|
638
|
+
return { done: false, memories, dedupPoolSize: snapshot.dedupPoolSize, prefilteredAlreadyPromoted };
|
|
636
639
|
}
|
|
637
640
|
/**
|
|
638
641
|
* Pass 2 — turn the narrowed pool into an executable plan. Sizes chunks to the
|
|
@@ -681,7 +684,7 @@ function recordChunkJudgedNoAction(chunk, ops, accounting) {
|
|
|
681
684
|
* are byte-identical, and every counter-increment point is unmoved.
|
|
682
685
|
*/
|
|
683
686
|
async function judgeConsolidationChunks(args) {
|
|
684
|
-
const { chunks, opts, config, llmRunner, lease, sourceName, bodyTruncation, pendingProposalBodyHashes,
|
|
687
|
+
const { chunks, opts, config, llmRunner, lease, sourceName, bodyTruncation, pendingProposalBodyHashes, warnings, accounting, } = args;
|
|
685
688
|
const chunkOpsArrays = [];
|
|
686
689
|
// judgedNoAction tracks memories the LLM saw inside a chunk but proposed
|
|
687
690
|
// no op for. Computed per chunk as `chunk.length − unique(targetRefs in ops)`.
|
|
@@ -748,7 +751,7 @@ async function judgeConsolidationChunks(args) {
|
|
|
748
751
|
continue;
|
|
749
752
|
}
|
|
750
753
|
warn(`[consolidate] chunk ${chunkIdx + 1}/${chunks.length} (${chunk.length} memories) …`);
|
|
751
|
-
const userPrompt = buildChunkPrompt(sourceName, chunk, chunkIdx, chunks.length, bodyTruncation, pendingProposalBodyHashes
|
|
754
|
+
const userPrompt = buildChunkPrompt(sourceName, chunk, chunkIdx, chunks.length, bodyTruncation, pendingProposalBodyHashes);
|
|
752
755
|
// Single chunk LLM call, wrapped in the feature gate. Deduplicated across
|
|
753
756
|
// the first attempt and the retry below (the two blocks were byte-identical
|
|
754
757
|
// apart from their fallback error string). responseSchema lift (PR 1,
|
|
@@ -971,10 +974,6 @@ async function planConsolidation(opts, config, stashDir, _startMs, memories, war
|
|
|
971
974
|
const pendingProposalBodyHashes = loadPendingConsolidateProposalHashes(stashDir);
|
|
972
975
|
warn(`[consolidate] ${budgetedMemories.length} memories / ${chunks.length} chunk(s) / chunk_size=${chunkSize}` +
|
|
973
976
|
` / pending-proposal hashes: ${pendingProposalBodyHashes.size}`);
|
|
974
|
-
// Consolidate output merges memories (non-wiki) → stash authoring standards.
|
|
975
|
-
// Resolved ONCE per run and passed to each chunk prompt (facts not re-read
|
|
976
|
-
// per chunk).
|
|
977
|
-
const standardsContext = resolveStandardsContext("memories/_consolidated", stashDir);
|
|
978
977
|
const chunkOpsArrays = await judgeConsolidationChunks({
|
|
979
978
|
chunks,
|
|
980
979
|
opts,
|
|
@@ -984,7 +983,6 @@ async function planConsolidation(opts, config, stashDir, _startMs, memories, war
|
|
|
984
983
|
sourceName,
|
|
985
984
|
bodyTruncation,
|
|
986
985
|
pendingProposalBodyHashes,
|
|
987
|
-
standardsContext,
|
|
988
986
|
warnings,
|
|
989
987
|
accounting,
|
|
990
988
|
});
|
|
@@ -1008,11 +1006,18 @@ async function planConsolidation(opts, config, stashDir, _startMs, memories, war
|
|
|
1008
1006
|
}
|
|
1009
1007
|
}
|
|
1010
1008
|
async function akmConsolidateInner(opts, config, stashDir, startMs, warnings, sharedStateDb) {
|
|
1009
|
+
// Loaded once and shared with the pre-filter (narrowConsolidationPool) and
|
|
1010
|
+
// the post-LLM promote-dedup check (shouldSkipPromotionBodyDuplicate) below
|
|
1011
|
+
// — knowledge/ can hold thousands of files, so walking it twice per run
|
|
1012
|
+
// would double that cost for no benefit. When the caller already walked
|
|
1013
|
+
// knowledge/ for the pool preview (runConsolidationPass, R2-1/R3-1), reuse
|
|
1014
|
+
// that set instead of walking it again here.
|
|
1015
|
+
const existingKnowledgeBodyHashes = opts.existingKnowledgeBodyHashes ?? loadExistingKnowledgeBodyHashes(stashDir);
|
|
1011
1016
|
// -- Pass 1: narrow the memory pool (may early-return an envelope) ----------
|
|
1012
|
-
const narrowed = await narrowConsolidationPool(opts, stashDir, startMs, warnings);
|
|
1017
|
+
const narrowed = await narrowConsolidationPool(opts, stashDir, startMs, warnings, existingKnowledgeBodyHashes);
|
|
1013
1018
|
if (narrowed.done)
|
|
1014
1019
|
return narrowed.result;
|
|
1015
|
-
const { memories, dedupPoolSize } = narrowed;
|
|
1020
|
+
const { memories, dedupPoolSize, prefilteredAlreadyPromoted } = narrowed;
|
|
1016
1021
|
// -- Pass 2: build the LLM plan (populates the shared accounting counters) ---
|
|
1017
1022
|
const accounting = createConsolidateAccounting();
|
|
1018
1023
|
const { allOps, totalChunks, llmPoolSize, deferredMemories, embedTelemetry, sourceName } = await planConsolidation(opts, config, stashDir, startMs, memories, warnings, sharedStateDb, accounting);
|
|
@@ -1035,6 +1040,7 @@ async function akmConsolidateInner(opts, config, stashDir, startMs, warnings, sh
|
|
|
1035
1040
|
planned: allOps,
|
|
1036
1041
|
warnings,
|
|
1037
1042
|
durationMs: Date.now() - startMs,
|
|
1043
|
+
prefilteredAlreadyPromoted,
|
|
1038
1044
|
});
|
|
1039
1045
|
}
|
|
1040
1046
|
warn(`[consolidate] plan: ${allOps.length} operation(s)`);
|
|
@@ -1052,7 +1058,7 @@ async function akmConsolidateInner(opts, config, stashDir, startMs, warnings, sh
|
|
|
1052
1058
|
memoryByRef,
|
|
1053
1059
|
promoted,
|
|
1054
1060
|
promotedSourceRefs: new Set(),
|
|
1055
|
-
existingKnowledgeBodyHashes
|
|
1061
|
+
existingKnowledgeBodyHashes,
|
|
1056
1062
|
promotionFailures,
|
|
1057
1063
|
warnings,
|
|
1058
1064
|
pushSkipReason: accounting.pushSkipReason,
|
|
@@ -1077,6 +1083,7 @@ async function akmConsolidateInner(opts, config, stashDir, startMs, warnings, sh
|
|
|
1077
1083
|
planned: allOps,
|
|
1078
1084
|
warnings,
|
|
1079
1085
|
durationMs: Date.now() - startMs,
|
|
1086
|
+
prefilteredAlreadyPromoted,
|
|
1080
1087
|
perfTelemetry: {
|
|
1081
1088
|
dedupPoolSize,
|
|
1082
1089
|
llmPoolSize,
|
|
@@ -1203,7 +1210,13 @@ export async function emitPromotionProposal(op, ctx) {
|
|
|
1203
1210
|
// different knowledgeRef slug.
|
|
1204
1211
|
// Use cacheHash (case-preserving stripped body) to match the canonical
|
|
1205
1212
|
// hash domain used by the body-embedding cache and pending-proposal set.
|
|
1206
|
-
|
|
1213
|
+
// H1: hash `memoryContent` (one frontmatter strip, inside cacheHash)
|
|
1214
|
+
// rather than the already-stripped `sourceBody` — hashing sourceBody here
|
|
1215
|
+
// double-stripped (parseFrontmatter ran once above to produce sourceBody,
|
|
1216
|
+
// then cacheHash's internal stripFrontmatterBody ran again), diverging from
|
|
1217
|
+
// the single-strip domain `loadExistingKnowledgeBodyHashes` and the
|
|
1218
|
+
// pre-filter use whenever a body starts with its own `---` block.
|
|
1219
|
+
const bodyHash = cacheHash(memoryContent);
|
|
1207
1220
|
if (shouldSkipPromotionBodyDuplicate({ bodyHash, op, knowledgeRef, ctx }))
|
|
1208
1221
|
return;
|
|
1209
1222
|
try {
|