akm-cli 0.9.17-alpha.2 → 0.9.17-alpha.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +756 -0
- package/dist/akm +94 -196
- package/dist/cli/shared.js +6 -2
- package/dist/cli.js +22 -9
- package/dist/commands/agent/agent-dispatch.js +1 -1
- package/dist/commands/command/command-execution.js +24 -62
- package/dist/commands/feedback-cli.js +0 -1
- package/dist/commands/health/accept-rate.js +2 -2
- package/dist/commands/health/checks.js +30 -75
- package/dist/commands/health/config-skew.js +38 -0
- package/dist/commands/health/egress.js +54 -0
- package/dist/commands/health/html-report.js +0 -38
- package/dist/commands/health/improve-metrics.js +123 -562
- package/dist/commands/health/plugin-staleness.js +53 -3
- package/dist/commands/health/renderers.js +12 -4
- package/dist/commands/health/report-view-model.js +11 -106
- package/dist/commands/health/types-improve.js +4 -19
- package/dist/commands/health/windows.js +64 -73
- package/dist/commands/health.js +122 -143
- package/dist/commands/improve/consolidate/chunking.js +25 -100
- package/dist/commands/improve/consolidate/sanitize.js +54 -149
- package/dist/commands/improve/consolidate.js +538 -1075
- package/dist/commands/improve/content-hash.js +16 -24
- package/dist/commands/improve/distill/content-repair.js +18 -100
- package/dist/commands/improve/distill-guards.js +20 -81
- package/dist/commands/improve/distill-promotion-policy.js +23 -243
- package/dist/commands/improve/distill.js +608 -1075
- package/dist/commands/improve/eligibility.js +126 -400
- package/dist/commands/improve/execution.js +3 -5
- package/dist/commands/improve/extract.js +487 -1046
- package/dist/commands/improve/feedback-valence.js +0 -25
- package/dist/commands/improve/improve-cli.js +29 -166
- package/dist/commands/improve/improve-result-file.js +10 -66
- package/dist/commands/improve/improve-strategies.js +12 -7
- package/dist/commands/improve/improve-usage-report.js +18 -64
- package/dist/commands/improve/improve.js +443 -1063
- package/dist/commands/improve/ledger.js +114 -0
- package/dist/commands/improve/locks.js +2 -8
- package/dist/commands/improve/loop-stages.js +459 -1172
- package/dist/commands/improve/memory/derived-ref.js +12 -77
- package/dist/commands/improve/memory/memory-belief.js +14 -118
- package/dist/commands/improve/memory/memory-improve.js +4 -3
- package/dist/commands/improve/outcome-loop.js +28 -156
- package/dist/commands/improve/planner.js +5 -10
- package/dist/commands/improve/preparation.js +851 -2339
- package/dist/commands/improve/proactive-maintenance.js +34 -101
- package/dist/commands/improve/reflect-noise.js +104 -280
- package/dist/commands/improve/reflect.js +621 -1367
- package/dist/commands/improve/salience.js +46 -232
- package/dist/commands/improve/session-asset.js +19 -100
- package/dist/commands/improve/stage.js +323 -0
- package/dist/commands/proposal/drain.js +251 -644
- package/dist/commands/proposal/proposal-cli.js +3 -18
- package/dist/commands/proposal/proposal-types.js +20 -41
- package/dist/commands/proposal/proposal.js +1 -2
- package/dist/commands/proposal/propose.js +134 -160
- package/dist/commands/proposal/repository.js +502 -1487
- package/dist/commands/proposal/validators/proposal-quality-validators.js +71 -174
- package/dist/commands/proposal/validators/proposal-validators.js +1 -1
- package/dist/commands/proposal/validators/proposals.js +13 -89
- package/dist/commands/read/curate.js +63 -413
- package/dist/commands/read/search-cli.js +16 -33
- package/dist/commands/read/search.js +17 -23
- package/dist/commands/read/show.js +2 -13
- package/dist/commands/sources/bundle-cli.js +25 -2
- package/dist/commands/sources/bundle-config-ops.js +7 -0
- package/dist/commands/sources/dangerous-env-audit.js +1 -2
- package/dist/commands/sources/info.js +2 -11
- package/dist/commands/sources/installed-stashes.js +197 -746
- package/dist/commands/sources/schema-repair.js +98 -129
- package/dist/commands/sources/source-add.js +62 -12
- package/dist/commands/sources/stash-cli.js +1 -1
- package/dist/commands/tasks/explain.js +10 -13
- package/dist/commands/tasks/tasks-cli.js +9 -8
- package/dist/commands/tasks/tasks.js +326 -930
- package/dist/commands/tasks/validate.js +42 -21
- package/dist/commands/workflow/plan.js +22 -29
- package/dist/commands/workflow-cli.js +4 -4
- package/dist/core/adapter/adapters/akm-adapter.js +0 -1
- package/dist/core/adapter/adapters/akm-lint.js +2 -3
- package/dist/core/adapter/adapters/akm-metadata.js +11 -12
- package/dist/core/adapter/adapters/akm-workflow-adapter.js +1 -1
- package/dist/core/adapter/execution-source.js +17 -29
- package/dist/core/asset/resolve-ref.js +1 -1
- package/dist/core/bundle-id.js +42 -5
- package/dist/core/bundle-rename.js +291 -0
- package/dist/core/config/config-io.js +1 -2
- package/dist/core/config/config-schema.js +1 -33
- package/dist/core/config/config-walker.js +1 -1
- package/dist/core/config/config.js +163 -68
- package/dist/core/config/legacy-source-shape-shim.js +38 -9
- package/dist/core/config/schema/embedding.js +20 -5
- package/dist/core/config/schema/engines.js +5 -0
- package/dist/core/config/schema/execution.js +1 -1
- package/dist/core/config/schema/experimental.js +1 -1
- package/dist/core/config/schema/improve-processes.js +21 -95
- package/dist/core/config/schema/improve.js +4 -42
- package/dist/core/config/schema/scheduler.js +12 -12
- package/dist/core/config/schema/search.js +6 -22
- package/dist/core/env-secret-ref.js +0 -1
- package/dist/core/errors.js +8 -9
- package/dist/core/file-lock.js +76 -173
- package/dist/core/logs-db.js +2 -2
- package/dist/core/paths.js +0 -27
- package/dist/core/redaction.js +109 -2
- package/dist/core/run-lock.js +2 -5
- package/dist/core/spawn-env.js +1 -1
- package/dist/core/state/migrations.js +108 -61
- package/dist/core/state-db-scope.js +2 -4
- package/dist/core/state-db.js +126 -692
- package/dist/core/type-presentation.js +1 -9
- package/dist/core/write-source.js +293 -1012
- package/dist/execution/input-contract.js +1 -1
- package/dist/execution/resolved-request.js +135 -689
- package/dist/execution/source.js +63 -257
- package/dist/execution/target-ref.js +1 -1
- package/dist/indexer/bundle-identity-guard.js +2 -2
- package/dist/indexer/db/graph-db.js +106 -46
- package/dist/indexer/ensure-index.js +44 -85
- package/dist/indexer/graph/graph-extraction.js +340 -562
- package/dist/indexer/graph/graph-related.js +130 -0
- package/dist/indexer/index-rebuild-lock.js +3 -11
- package/dist/indexer/index-writer-lock.js +8 -17
- package/dist/indexer/index-written-assets.js +139 -151
- package/dist/indexer/indexer.js +524 -846
- package/dist/indexer/materialize-embeddings.js +60 -397
- package/dist/indexer/passes/memory-inference.js +81 -90
- package/dist/indexer/passes/metadata.js +132 -200
- package/dist/indexer/read-preflight.js +0 -7
- package/dist/indexer/scan/doc-to-entry.js +1 -3
- package/dist/indexer/scan/drain-dir.js +1 -1
- package/dist/indexer/search/db-search.js +181 -590
- package/dist/indexer/search/fts-query.js +30 -41
- package/dist/indexer/search/ranking.js +28 -154
- package/dist/indexer/search/search-attribution.js +12 -32
- package/dist/indexer/search/search-fields.js +11 -15
- package/dist/indexer/search/search-hit-enrichers.js +54 -85
- package/dist/indexer/search/search-source.js +1 -4
- package/dist/indexer/usage/usage-events.js +2 -7
- package/dist/integrations/agent/engine-fallback.js +23 -40
- package/dist/integrations/agent/engine-resolution.js +93 -183
- package/dist/integrations/agent/execution.js +507 -0
- package/dist/integrations/agent/model-map.js +28 -156
- package/dist/integrations/agent/request-lowering.js +66 -141
- package/dist/integrations/agent/runner-dispatch.js +143 -321
- package/dist/integrations/agent/runner.js +54 -14
- package/dist/integrations/lockfile.js +53 -101
- package/dist/llm/embedders/deterministic.js +2 -3
- package/dist/llm/embedders/profile.js +71 -0
- package/dist/llm/embedders/remote.js +10 -15
- package/dist/llm/graph-extract.js +3 -12
- package/dist/llm/index-passes.js +3 -5
- package/dist/llm/memory-infer.js +1 -2
- package/dist/llm/metadata-enhance.js +1 -2
- package/dist/llm/structured-call.js +5 -24
- package/dist/output/generic-render.js +23 -11
- package/dist/output/html-render.js +13 -10
- package/dist/output/render-registry.js +3 -32
- package/dist/output/shapes/helpers.js +2 -34
- package/dist/output/shapes/passthrough.js +1 -9
- package/dist/{indexer/search/ranking-types.js → output/text/bundle-rename.js} +4 -1
- package/dist/output/text/command-format.js +60 -23
- package/dist/output/text/helpers.js +1 -1
- package/dist/output/text/migrate.js +5 -14
- package/dist/output/text/proposal-format.js +1 -2
- package/dist/output/text/workflow-format.js +0 -32
- package/dist/output/text.js +2 -0
- package/dist/registry/factory.js +4 -19
- package/dist/registry/network.js +66 -220
- package/dist/registry/providers/index.js +0 -2
- package/dist/registry/providers/skills-sh.js +3 -14
- package/dist/registry/providers/static-index.js +24 -26
- package/dist/registry/resolve.js +55 -131
- package/dist/scripts/akm-migrate-node.js +43937 -93313
- package/dist/scripts/akm-migrate.js +43697 -93071
- package/dist/setup/registry-stash-loader.js +4 -13
- package/dist/setup/semantic-assets.js +3 -44
- package/dist/setup/setup.js +1 -1
- package/dist/setup/steps/tasks.js +25 -15
- package/dist/sources/provider-factory.js +17 -18
- package/dist/sources/providers/filesystem.js +2 -3
- package/dist/sources/providers/git-install.js +7 -1
- package/dist/sources/providers/git-provider.js +0 -3
- package/dist/sources/providers/git-stash.js +0 -17
- package/dist/sources/providers/npm.js +2 -4
- package/dist/sources/providers/provider-utils.js +5 -10
- package/dist/sources/providers/website.js +0 -2
- package/dist/sources/snapshot-fetchers/website-ingest.js +1 -1
- package/dist/sources/website-url.js +2 -2
- package/dist/storage/database.js +9 -35
- package/dist/storage/repositories/improve-ledger-repository.js +168 -0
- package/dist/storage/repositories/index-connection.js +34 -70
- package/dist/storage/repositories/index-entries-repository.js +69 -111
- package/dist/storage/repositories/index-entry-mapper.js +1 -2
- package/dist/storage/repositories/index-entry-schema.js +83 -269
- package/dist/storage/repositories/index-fts-repository.js +86 -256
- package/dist/storage/repositories/index-llm-cache-repository.js +17 -0
- package/dist/storage/repositories/index-meta-repository.js +6 -4
- package/dist/storage/repositories/index-schema.js +192 -220
- package/dist/storage/repositories/index-utility-repository.js +8 -29
- package/dist/storage/repositories/index-vec-repository.js +133 -414
- package/dist/storage/repositories/outcome-repository.js +2 -1
- package/dist/storage/repositories/proposals-repository.js +35 -0
- package/dist/storage/repositories/registry-index-cache-repository.js +100 -0
- package/dist/storage/repositories/task-history-repository.js +26 -4
- package/dist/storage/repositories/workflow-runs-repository.js +53 -244
- package/dist/storage/sqlite-migrations.js +136 -0
- package/dist/storage/sqlite-pragmas.js +11 -9
- package/dist/storage/sqlite-transaction.js +170 -0
- package/dist/storage/state-db-integrity.js +34 -27
- package/dist/tasks/activation-config.js +134 -62
- package/dist/tasks/backends/cron.js +129 -277
- package/dist/tasks/backends/exec-utils.js +2 -5
- package/dist/tasks/backends/launchd.js +125 -745
- package/dist/tasks/backends/schtasks.js +101 -620
- package/dist/tasks/prepare/prepare-support.js +5 -15
- package/dist/tasks/prepare/prepare.js +0 -2
- package/dist/tasks/resolve-akm-bin.js +20 -79
- package/dist/tasks/run/attempt-lifecycle.js +0 -1
- package/dist/tasks/scheduler-binding.js +18 -238
- package/dist/tasks/scheduler-invocation.js +52 -52
- package/dist/tasks/scheduler-lock.js +53 -0
- package/dist/tasks/scheduler-sync.js +363 -679
- package/dist/tasks/source/parse-task-source.js +160 -10
- package/dist/tasks/source/task-source-v3-frozen.js +3 -4
- package/dist/tasks/source/task-to-v4.js +2 -2
- package/dist/workflows/authoring/authoring.js +3 -12
- package/dist/workflows/compile.js +211 -0
- package/dist/workflows/concurrency-policy.js +13 -74
- package/dist/workflows/exec/child-invocation.js +3 -17
- package/dist/workflows/exec/child-workflow.js +32 -141
- package/dist/workflows/exec/dispatch-redaction.js +13 -53
- package/dist/workflows/exec/environment.js +98 -0
- package/dist/workflows/exec/exec-unit.js +33 -140
- package/dist/workflows/exec/frozen-judge.js +7 -59
- package/dist/workflows/exec/native-executor.js +82 -341
- package/dist/workflows/exec/param-secrets.js +29 -47
- package/dist/workflows/exec/run-workflow.js +154 -387
- package/dist/workflows/exec/scheduler.js +9 -36
- package/dist/workflows/exec/step-work.js +127 -430
- package/dist/workflows/exec/unit-dispatch.js +11 -63
- package/dist/workflows/exec/unit-writer.js +8 -52
- package/dist/workflows/exec/worktree.js +39 -273
- package/dist/workflows/freeze/child-output-references.js +4 -15
- package/dist/workflows/freeze/environment.js +99 -92
- package/dist/workflows/freeze/freeze.js +172 -0
- package/dist/workflows/freeze/step-values.js +19 -21
- package/dist/workflows/freeze/targets/child-workflow.js +23 -92
- package/dist/workflows/freeze/targets/command.js +10 -33
- package/dist/workflows/freeze/targets/script.js +5 -12
- package/dist/workflows/freeze/targets/shell.js +3 -6
- package/dist/workflows/freeze/targets/task.js +25 -80
- package/dist/workflows/freeze/task-bindings.js +20 -67
- package/dist/workflows/{source-ir/github-yaml.js → github-yaml.js} +88 -206
- package/dist/workflows/ir/params.js +6 -51
- package/dist/workflows/ir/plan-hash.js +2 -34
- package/dist/workflows/parser.js +140 -43
- package/dist/{commands/improve/consolidate/types.js → workflows/plan.js} +2 -1
- package/dist/workflows/renderer.js +36 -69
- package/dist/workflows/resource-limits.js +12 -120
- package/dist/workflows/runtime/agent-identity.js +8 -40
- package/dist/workflows/runtime/run-outputs.js +3 -6
- package/dist/workflows/runtime/run-plan.js +316 -0
- package/dist/workflows/runtime/runs.js +48 -200
- package/dist/workflows/runtime/workflow-asset-loader.js +24 -57
- package/dist/workflows/{source-ir/semantics.js → source-semantics.js} +16 -20
- package/dist/workflows/validate-summary.js +2 -7
- package/docs/integration/bundling-akm.md +49 -42
- package/docs/migration/README.md +1 -0
- package/docs/migration/release-notes/0.9.17.md +41 -0
- package/docs/migration/v0.9.1-to-v0.9.2.md +19 -7
- package/docs/reference/cli.md +182 -125
- package/docs/reference/configuration.md +49 -56
- package/docs/reference/data-and-telemetry.md +19 -20
- package/docs/reference/tasks.md +86 -38
- package/docs/reference/workflow-schema.md +14 -18
- package/docs/reference/workflows.md +6 -9
- package/package.json +1 -1
- package/schemas/akm-config.json +87 -406
- package/dist/commands/health/advisories.js +0 -150
- package/dist/commands/health/metrics.js +0 -329
- package/dist/commands/health/surfaces.js +0 -102
- package/dist/commands/improve/anti-collapse.js +0 -83
- package/dist/commands/improve/collapse-detector.js +0 -432
- package/dist/commands/improve/consolidate/eligibility.js +0 -48
- package/dist/commands/improve/consolidate/merge.js +0 -146
- package/dist/commands/improve/distill/promote-memory.js +0 -329
- package/dist/commands/improve/distill/quality-gate.js +0 -500
- package/dist/commands/improve/memory/memory-contradiction-detect.js +0 -291
- package/dist/commands/improve/proposal-envelope.js +0 -31
- package/dist/commands/improve/run-context.js +0 -123
- package/dist/commands/improve/shared.js +0 -21
- package/dist/commands/improve/source-identity.js +0 -28
- package/dist/commands/improve/triage.js +0 -96
- package/dist/commands/proposal/drain-policies.js +0 -151
- package/dist/commands/sources/update-transaction.js +0 -220
- package/dist/core/action-contributors.js +0 -28
- package/dist/core/config/config-version-shim.js +0 -101
- package/dist/core/config/retired-experimental-keys-shim.js +0 -62
- package/dist/core/fs-txn.js +0 -405
- package/dist/core/lexical-score.js +0 -25
- package/dist/core/maintenance-barrier.js +0 -167
- package/dist/execution/executable-identity.js +0 -105
- package/dist/execution/guarded-source.js +0 -427
- package/dist/indexer/graph/graph-boost.js +0 -427
- package/dist/indexer/graph/graph-dedup.js +0 -95
- package/dist/indexer/search/name-match.js +0 -35
- package/dist/indexer/search/ranking-contributors.js +0 -515
- package/dist/indexer/walk/project-context.js +0 -192
- package/dist/integrations/agent/execution-cascade.js +0 -566
- package/dist/integrations/agent/execution-definitions.js +0 -202
- package/dist/integrations/agent/execution-lowering.js +0 -841
- package/dist/integrations/agent/execution-preparation.js +0 -98
- package/dist/integrations/agent/inline-execution.js +0 -74
- package/dist/registry/create-provider-registry.js +0 -29
- package/dist/registry/pinned-request-helper.js +0 -247
- package/dist/registry/pinned-transport.js +0 -717
- package/dist/sources/providers/index.js +0 -14
- package/dist/storage/engines/sqlite-migrations.js +0 -271
- package/dist/storage/repositories/canaries-repository.js +0 -107
- package/dist/storage/repositories/embedding-salvage-repository.js +0 -184
- package/dist/storage/repositories/registry-cache.js +0 -113
- package/dist/tasks/scheduler-sync-preview.js +0 -52
- package/dist/workflows/freeze/resolve-steps.js +0 -86
- package/dist/workflows/freeze/source-freeze.js +0 -64
- package/dist/workflows/ir/compile.js +0 -321
- package/dist/workflows/ir/environment-v4.js +0 -330
- package/dist/workflows/ir/freeze-v4.js +0 -153
- package/dist/workflows/ir/schema-v4.js +0 -745
- package/dist/workflows/ir/schema.js +0 -354
- package/dist/workflows/program/schema.js +0 -78
- package/dist/workflows/runtime/checkin.js +0 -57
- package/dist/workflows/runtime/plan-classifier.js +0 -196
- package/dist/workflows/runtime/unit-checkin.js +0 -45
- package/dist/workflows/runtime/unit-phases.js +0 -20
- package/dist/workflows/schema.js +0 -4
- package/dist/workflows/source-ir/compile.js +0 -200
- package/dist/workflows/source-ir/program.js +0 -50
- package/dist/workflows/source-ir/result.js +0 -26
- package/dist/workflows/source-ir/schema.js +0 -786
- package/dist/workflows/source-ir/triggers.js +0 -79
- package/dist/workflows/source-ir/uses.js +0 -40
- package/dist/workflows/validator.js +0 -60
|
@@ -1,7 +1,16 @@
|
|
|
1
1
|
// This Source Code Form is subject to the terms of the Mozilla Public
|
|
2
2
|
// License, v. 2.0. If a copy of the MPL was not distributed with this
|
|
3
3
|
// file, You can obtain one at https://mozilla.org/MPL/2.0/.
|
|
4
|
-
|
|
4
|
+
/**
|
|
5
|
+
* `akm consolidate` — show the model the memory pool in chunks of similar
|
|
6
|
+
* memories and queue a knowledge proposal for each memory it says should be
|
|
7
|
+
* promoted. Promotion is the only operation: it emits a reviewable proposal and
|
|
8
|
+
* never touches the memory. Memories the improve ledger judged recently and
|
|
9
|
+
* that have not changed since are not judged again.
|
|
10
|
+
*
|
|
11
|
+
* Accounting invariant: `processed == promoted + judgedNoAction +
|
|
12
|
+
* Σ(skipReasons) + failedChunkMemories`.
|
|
13
|
+
*/
|
|
5
14
|
import fs from "node:fs";
|
|
6
15
|
import path from "node:path";
|
|
7
16
|
import consolidateSystemPrompt from "../../assets/prompts/consolidate-system.md" with { type: "text" };
|
|
@@ -17,48 +26,56 @@ import { warn, warnVerbose } from "../../core/warn.js";
|
|
|
17
26
|
import { resolveWriteTarget } from "../../core/write-source.js";
|
|
18
27
|
import { deriveInstallations } from "../../indexer/installations.js";
|
|
19
28
|
import { resolveSourceEntries } from "../../indexer/search/search-source.js";
|
|
20
|
-
import {
|
|
29
|
+
import { assertRunnerCredentials } from "../../integrations/agent/runner-dispatch.js";
|
|
21
30
|
import { cosineSimilarity, embedBatch, resolveEmbeddingModelId } from "../../llm/embedder.js";
|
|
22
|
-
import { callStructured, preflightStructuredLlmRunner } from "../../llm/structured-call.js";
|
|
23
31
|
import { getBodyEmbeddings, upsertBodyEmbeddings } from "../../storage/repositories/embeddings-repository.js";
|
|
24
32
|
import { closeDatabase, openExistingDatabase, openReadonlyExistingDatabase, } from "../../storage/repositories/index-connection.js";
|
|
25
33
|
import { findEntryIdByRef, getAllEntries, getEntryById } from "../../storage/repositories/index-entries-repository.js";
|
|
26
34
|
import { getNeighborsByEntryId } from "../../storage/repositories/index-vec-repository.js";
|
|
27
|
-
import {
|
|
28
|
-
import { hasSupersededStatus, validateProposalFrontmatter } from "../proposal/validators/proposal-quality-validators.js";
|
|
29
|
-
import { DEFAULT_RANDOM_CLUSTER_FRACTION } from "./anti-collapse.js";
|
|
30
|
-
import { cacheHash } from "./content-hash.js";
|
|
31
|
-
import { resolveImproveLlmExecution } from "./execution.js";
|
|
32
|
-
import { resolveImproveStrategy, resolveProcessEnabled } from "./improve-strategies.js";
|
|
33
|
-
import { emitProposal } from "./proposal-envelope.js";
|
|
34
|
-
import { createRunContext } from "./run-context.js";
|
|
35
|
-
// Chunk sizing + per-chunk prompt assembly live in ./consolidate/chunking.
|
|
35
|
+
import { listProposals, listProposalsReadOnly, proposalContent } from "../proposal/repository.js";
|
|
36
|
+
import { hasHotCaptureMode, hasSupersededStatus, validateProposalFrontmatter, } from "../proposal/validators/proposal-quality-validators.js";
|
|
36
37
|
import { buildChunkPrompt, computeSafeChunkSize, DEFAULT_CONTEXT_LENGTH_TOKENS } from "./consolidate/chunking.js";
|
|
37
|
-
// Eligibility / safety predicates live in ./consolidate/eligibility.
|
|
38
|
-
import { isConsolidationEligibleMemoryName, isHotCapturedMemory } from "./consolidate/eligibility.js";
|
|
39
|
-
// Plan parsing / merging (pure op-reconciliation algebra) lives in
|
|
40
|
-
// ./consolidate/merge.
|
|
41
|
-
import { isValidOp, mergePlans } from "./consolidate/merge.js";
|
|
42
|
-
// LLM-output sanitization (pure string/frontmatter transforms) lives in
|
|
43
|
-
// ./consolidate/sanitize.
|
|
44
38
|
import { sanitizeMergedContent } from "./consolidate/sanitize.js";
|
|
45
|
-
|
|
46
|
-
|
|
39
|
+
import { contentHash } from "./content-hash.js";
|
|
40
|
+
import { resolveImproveStrategy, resolveProcessEnabled } from "./improve-strategies.js";
|
|
41
|
+
import { isLedgerBlocked, ledgerKey, loadLedgerSnapshot, recordLedgerAttempt } from "./ledger.js";
|
|
42
|
+
import { callStage, mintProposal, noticeSet, stageRunner } from "./stage.js";
|
|
43
|
+
/** A plan op worth acting on. Retired advisory ops (merge/delete/contradict) are dropped, never thrown on. */
|
|
44
|
+
export function isValidOp(op) {
|
|
45
|
+
if (typeof op !== "object" || op === null)
|
|
46
|
+
return false;
|
|
47
|
+
const o = op;
|
|
48
|
+
return o.op === "promote" && typeof o.ref === "string" && typeof o.knowledgeRef === "string";
|
|
49
|
+
}
|
|
50
|
+
/** Reconcile the per-chunk plans: one promotion per source memory, the last chunk's wins. */
|
|
51
|
+
export function mergePlans(chunks) {
|
|
52
|
+
const byRef = new Map();
|
|
53
|
+
for (const chunk of chunks)
|
|
54
|
+
for (const op of chunk)
|
|
55
|
+
byRef.set(op.ref, op);
|
|
56
|
+
return [...byRef.values()];
|
|
57
|
+
}
|
|
58
|
+
export function isConsolidationEligibleMemoryName(name) {
|
|
59
|
+
return !name.endsWith(".derived");
|
|
60
|
+
}
|
|
47
61
|
/**
|
|
48
|
-
*
|
|
49
|
-
*
|
|
50
|
-
*
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
+
* A `captureMode: hot` memory (written deliberately with `akm remember`). A
|
|
63
|
+
* missing file is not hot; an unreadable one is treated as hot — the check is
|
|
64
|
+
* a protection and must not fail open.
|
|
65
|
+
*/
|
|
66
|
+
export function isHotCapturedMemory(filePath) {
|
|
67
|
+
if (!fs.existsSync(filePath))
|
|
68
|
+
return false;
|
|
69
|
+
try {
|
|
70
|
+
return hasHotCaptureMode(parseFrontmatter(fs.readFileSync(filePath, "utf8")).data);
|
|
71
|
+
}
|
|
72
|
+
catch {
|
|
73
|
+
return true;
|
|
74
|
+
}
|
|
75
|
+
}
|
|
76
|
+
/**
|
|
77
|
+
* Structured-output schema for a plan. Promote-only: merge/delete/contradict
|
|
78
|
+
* were advisory, never executed, and cost thousands of completion tokens.
|
|
62
79
|
*/
|
|
63
80
|
export const CONSOLIDATE_PLAN_JSON_SCHEMA = {
|
|
64
81
|
type: "object",
|
|
@@ -84,125 +101,74 @@ export const CONSOLIDATE_PLAN_JSON_SCHEMA = {
|
|
|
84
101
|
},
|
|
85
102
|
},
|
|
86
103
|
};
|
|
104
|
+
/**
|
|
105
|
+
* Order memories so similar ones sit together and land in the same chunk:
|
|
106
|
+
* a greedy nearest-neighbour chain over description+tag embeddings (cached in
|
|
107
|
+
* `body_embeddings` under the text's hash). Keeps the original order without
|
|
108
|
+
* an embedding config, for fewer than three memories, or when embedding fails.
|
|
109
|
+
*/
|
|
87
110
|
async function clusterMemoriesBySimilarity(memories, config, stateDb, signal) {
|
|
88
|
-
const
|
|
111
|
+
const telemetry = { embedMs: 0, cacheHits: 0, cacheMisses: 0 };
|
|
89
112
|
if (memories.length < 3 || !config.embedding)
|
|
90
|
-
return { ordered: memories, embedTelemetry:
|
|
91
|
-
// WS-3a: cluster uses description+tags as the embedding input (NOT the raw
|
|
92
|
-
// body) — this is intentionally different from the dedup/body cache because
|
|
93
|
-
// the clustering goal is semantic grouping, not dedup twin detection.
|
|
94
|
-
// The body_embeddings cache is keyed by cacheHash(body); clustering inputs
|
|
95
|
-
// are keyed by cacheHash(description+tags text). Re-use the same table with
|
|
96
|
-
// a distinct hash so the two lookup sets never collide.
|
|
113
|
+
return { ordered: memories, embedTelemetry: telemetry };
|
|
97
114
|
const modelId = resolveEmbeddingModelId(config.embedding);
|
|
98
|
-
const texts = memories.map((m) =>
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
parts.push(m.description);
|
|
102
|
-
if (m.tags.length > 0)
|
|
103
|
-
parts.push(m.tags.join(" "));
|
|
104
|
-
return parts.join(". ") || m.name;
|
|
105
|
-
});
|
|
106
|
-
// Compute content hashes for the cluster texts (not bodies — different input).
|
|
107
|
-
const contentHashes = texts.map((t) => createHash("sha256").update(t, "utf8").digest("hex"));
|
|
108
|
-
// WS-5: track embed cache hits/misses for perf telemetry.
|
|
109
|
-
let embedMs = 0;
|
|
110
|
-
let cacheHits = 0;
|
|
111
|
-
let cacheMisses = 0;
|
|
112
|
-
let cachedVecs = new Map();
|
|
115
|
+
const texts = memories.map((m) => [m.description, m.tags.join(" ")].filter(Boolean).join(". ") || m.name);
|
|
116
|
+
const hashes = texts.map((t) => contentHash(t));
|
|
117
|
+
let cached = new Map();
|
|
113
118
|
if (stateDb) {
|
|
114
119
|
try {
|
|
115
|
-
|
|
120
|
+
cached = getBodyEmbeddings(stateDb, hashes, modelId);
|
|
116
121
|
}
|
|
117
122
|
catch {
|
|
118
|
-
|
|
119
|
-
cachedVecs = new Map();
|
|
123
|
+
cached = new Map();
|
|
120
124
|
}
|
|
121
125
|
}
|
|
122
|
-
const missIndices = [];
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
missTexts.push(texts[i]);
|
|
128
|
-
cacheMisses++;
|
|
129
|
-
}
|
|
130
|
-
else {
|
|
131
|
-
cacheHits++;
|
|
132
|
-
}
|
|
133
|
-
}
|
|
134
|
-
let missVecs = [];
|
|
135
|
-
if (missTexts.length > 0) {
|
|
126
|
+
const missIndices = hashes.flatMap((hash, i) => (cached.has(hash) ? [] : [i]));
|
|
127
|
+
telemetry.cacheHits = memories.length - missIndices.length;
|
|
128
|
+
telemetry.cacheMisses = missIndices.length;
|
|
129
|
+
const vectors = new Map(cached);
|
|
130
|
+
if (missIndices.length > 0) {
|
|
136
131
|
const embedStart = Date.now();
|
|
132
|
+
let missVecs;
|
|
137
133
|
try {
|
|
138
|
-
missVecs = await embedBatch(
|
|
134
|
+
missVecs = await embedBatch(missIndices.map((i) => texts[i]), config.embedding, signal);
|
|
139
135
|
}
|
|
140
136
|
catch {
|
|
141
|
-
|
|
142
|
-
return { ordered: memories, embedTelemetry: { embedMs, cacheHits, cacheMisses } };
|
|
137
|
+
return { ordered: memories, embedTelemetry: telemetry };
|
|
143
138
|
}
|
|
144
139
|
finally {
|
|
145
|
-
embedMs += Date.now() - embedStart;
|
|
140
|
+
telemetry.embedMs += Date.now() - embedStart;
|
|
146
141
|
}
|
|
147
|
-
|
|
148
|
-
|
|
149
|
-
|
|
150
|
-
|
|
142
|
+
const fresh = missIndices.flatMap((idx, pos) => {
|
|
143
|
+
const embedding = missVecs[pos];
|
|
144
|
+
return embedding ? [{ contentHash: hashes[idx], embedding, modelId }] : [];
|
|
145
|
+
});
|
|
146
|
+
for (const entry of fresh)
|
|
147
|
+
vectors.set(entry.contentHash, entry.embedding);
|
|
148
|
+
// A document the embedder skipped has no vector to cache.
|
|
149
|
+
if (stateDb && missVecs.length === missIndices.length) {
|
|
151
150
|
try {
|
|
152
|
-
|
|
153
|
-
const embedding = missVecs[pos];
|
|
154
|
-
return embedding ? [{ contentHash: contentHashes[idx], embedding, modelId }] : [];
|
|
155
|
-
});
|
|
156
|
-
upsertBodyEmbeddings(stateDb, toUpsert);
|
|
151
|
+
upsertBodyEmbeddings(stateDb, fresh);
|
|
157
152
|
}
|
|
158
153
|
catch {
|
|
159
|
-
//
|
|
154
|
+
// Cache writes are best-effort.
|
|
160
155
|
}
|
|
161
156
|
}
|
|
162
157
|
}
|
|
163
|
-
|
|
164
|
-
|
|
165
|
-
|
|
166
|
-
const assembled = [];
|
|
167
|
-
let ok = true;
|
|
168
|
-
for (let i = 0; i < memories.length; i++) {
|
|
169
|
-
const hash = contentHashes[i];
|
|
170
|
-
const cached = cachedVecs.get(hash);
|
|
171
|
-
if (cached) {
|
|
172
|
-
assembled.push(cached);
|
|
173
|
-
continue;
|
|
174
|
-
}
|
|
175
|
-
const missPos = missIndices.indexOf(i);
|
|
176
|
-
const vec = missPos >= 0 ? missVecs[missPos] : undefined;
|
|
177
|
-
if (vec) {
|
|
178
|
-
assembled.push(vec);
|
|
179
|
-
}
|
|
180
|
-
else {
|
|
181
|
-
ok = false;
|
|
182
|
-
break;
|
|
183
|
-
}
|
|
184
|
-
}
|
|
185
|
-
if (ok && assembled.length === memories.length) {
|
|
186
|
-
embeddings = assembled;
|
|
187
|
-
}
|
|
188
|
-
}
|
|
189
|
-
const embedTelemetry = { embedMs, cacheHits, cacheMisses };
|
|
190
|
-
if (!embeddings || embeddings.length !== memories.length)
|
|
191
|
-
return { ordered: memories, embedTelemetry };
|
|
192
|
-
// Greedy nearest-neighbour chain.
|
|
158
|
+
const embeddings = hashes.map((hash) => vectors.get(hash));
|
|
159
|
+
if (embeddings.some((vec) => !vec))
|
|
160
|
+
return { ordered: memories, embedTelemetry: telemetry };
|
|
193
161
|
const used = new Array(memories.length).fill(false);
|
|
194
|
-
const ordered = [];
|
|
195
|
-
|
|
196
|
-
|
|
197
|
-
used[current] = true;
|
|
162
|
+
const ordered = [memories[0]];
|
|
163
|
+
used[0] = true;
|
|
164
|
+
let current = 0;
|
|
198
165
|
for (let step = 1; step < memories.length; step++) {
|
|
199
|
-
const currentEmb = embeddings[current];
|
|
200
166
|
let bestIdx = -1;
|
|
201
167
|
let bestSim = -Infinity;
|
|
202
168
|
for (let j = 0; j < memories.length; j++) {
|
|
203
169
|
if (used[j])
|
|
204
170
|
continue;
|
|
205
|
-
const sim = cosineSimilarity(
|
|
171
|
+
const sim = cosineSimilarity(embeddings[current], embeddings[j]);
|
|
206
172
|
if (sim > bestSim) {
|
|
207
173
|
bestSim = sim;
|
|
208
174
|
bestIdx = j;
|
|
@@ -214,49 +180,64 @@ async function clusterMemoriesBySimilarity(memories, config, stateDb, signal) {
|
|
|
214
180
|
used[bestIdx] = true;
|
|
215
181
|
current = bestIdx;
|
|
216
182
|
}
|
|
217
|
-
return { ordered, embedTelemetry };
|
|
183
|
+
return { ordered, embedTelemetry: telemetry };
|
|
218
184
|
}
|
|
219
|
-
// ── Chunk helpers ────────────────────────────────────────────────────────────
|
|
220
185
|
/**
|
|
221
|
-
*
|
|
222
|
-
*
|
|
223
|
-
*
|
|
224
|
-
* preserving stripped body) — the same domain used by the body-embedding
|
|
225
|
-
* cache. Empty set on any read/parse error — fail-safe to "annotate nothing"
|
|
226
|
-
* so the LLM still proposes.
|
|
186
|
+
* Anti-collapse (default on, `antiCollapse.enabled: false` opts out): a small
|
|
187
|
+
* deterministic sample of the pool is spread through the similarity order so
|
|
188
|
+
* consolidation is not purely similarity-driven.
|
|
227
189
|
*/
|
|
190
|
+
function injectRandomClusterMembers(memories, profile, warnings) {
|
|
191
|
+
const config = getImproveProcessConfig("consolidate", profile)?.antiCollapse ?? {};
|
|
192
|
+
if (config.enabled === false || memories.length <= 2)
|
|
193
|
+
return memories;
|
|
194
|
+
const fraction = config.randomClusterFraction ?? 0.05;
|
|
195
|
+
const randomCount = Math.max(1, Math.floor(memories.length * fraction));
|
|
196
|
+
const sample = [...memories]
|
|
197
|
+
.sort((a, b) => contentHash(a.name).localeCompare(contentHash(b.name)))
|
|
198
|
+
.slice(0, randomCount);
|
|
199
|
+
const sampled = new Set(sample.map((m) => m.name));
|
|
200
|
+
const interval = Math.max(2, Math.floor(memories.length / randomCount));
|
|
201
|
+
const out = [];
|
|
202
|
+
let next = 0;
|
|
203
|
+
for (let i = 0; i < memories.length; i++) {
|
|
204
|
+
const m = memories[i];
|
|
205
|
+
if (m && !sampled.has(m.name))
|
|
206
|
+
out.push(m);
|
|
207
|
+
if (i > 0 && i % interval === 0 && next < sample.length)
|
|
208
|
+
out.push(sample[next++]);
|
|
209
|
+
}
|
|
210
|
+
while (next < sample.length)
|
|
211
|
+
out.push(sample[next++]);
|
|
212
|
+
warnings.push(`Anti-collapse: injected ${randomCount} random (non-similarity-driven) cluster member(s) into consolidation pool (fraction=${fraction}).`);
|
|
213
|
+
return out;
|
|
214
|
+
}
|
|
215
|
+
/** Body hashes of pending consolidate proposals, so the prompt can mark memories already queued. */
|
|
228
216
|
function loadPendingConsolidateProposalHashes(stashDir) {
|
|
229
217
|
const hashes = new Set();
|
|
230
218
|
try {
|
|
231
|
-
const
|
|
232
|
-
|
|
219
|
+
for (const p of listProposalsReadOnly(stashDir, { status: "pending" })) {
|
|
220
|
+
if (p.source !== "consolidate")
|
|
221
|
+
continue;
|
|
233
222
|
try {
|
|
234
|
-
hashes.add(
|
|
223
|
+
hashes.add(contentHash(proposalContent(p), "body"));
|
|
235
224
|
}
|
|
236
225
|
catch {
|
|
237
|
-
//
|
|
226
|
+
// A malformed payload cannot dedup anyway.
|
|
238
227
|
}
|
|
239
228
|
}
|
|
240
229
|
}
|
|
241
230
|
catch {
|
|
242
|
-
//
|
|
231
|
+
// Annotate nothing; the model still proposes.
|
|
243
232
|
}
|
|
244
233
|
return hashes;
|
|
245
234
|
}
|
|
246
235
|
/**
|
|
247
|
-
*
|
|
248
|
-
*
|
|
249
|
-
* Pending-proposal dedup prevents repeated queue entries, but accepted
|
|
250
|
-
* proposals leave that set. Without a live-asset guard, the next run can copy
|
|
251
|
-
* the same memory body into a new knowledge slug indefinitely. Scan the target
|
|
252
|
-
* tree directly (rather than trusting the asynchronously refreshed index) so
|
|
253
|
-
* an already-written asset suppresses recurrence immediately.
|
|
236
|
+
* Body hashes of the live knowledge assets, read from disk (the index may lag
|
|
237
|
+
* a just-written asset), so an accepted promotion is not proposed again.
|
|
254
238
|
*/
|
|
255
239
|
export function loadExistingKnowledgeBodyHashes(targetRoot) {
|
|
256
240
|
const hashes = new Set();
|
|
257
|
-
const knowledgeRoot = path.join(targetRoot, "knowledge");
|
|
258
|
-
if (!fs.existsSync(knowledgeRoot))
|
|
259
|
-
return hashes;
|
|
260
241
|
const visit = (dir) => {
|
|
261
242
|
let entries;
|
|
262
243
|
try {
|
|
@@ -267,84 +248,31 @@ export function loadExistingKnowledgeBodyHashes(targetRoot) {
|
|
|
267
248
|
}
|
|
268
249
|
for (const entry of entries) {
|
|
269
250
|
const entryPath = path.join(dir, entry.name);
|
|
270
|
-
if (entry.isDirectory())
|
|
251
|
+
if (entry.isDirectory())
|
|
271
252
|
visit(entryPath);
|
|
272
|
-
}
|
|
273
253
|
else if (entry.isFile() && entry.name.endsWith(".md")) {
|
|
274
254
|
try {
|
|
275
|
-
hashes.add(
|
|
255
|
+
hashes.add(contentHash(fs.readFileSync(entryPath, "utf8"), "body"));
|
|
276
256
|
}
|
|
277
257
|
catch {
|
|
278
|
-
// An unreadable asset
|
|
258
|
+
// An unreadable asset is no duplicate evidence.
|
|
279
259
|
}
|
|
280
260
|
}
|
|
281
261
|
}
|
|
282
262
|
};
|
|
283
|
-
visit(
|
|
263
|
+
visit(path.join(targetRoot, "knowledge"));
|
|
284
264
|
return hashes;
|
|
285
265
|
}
|
|
286
|
-
/**
|
|
287
|
-
function
|
|
266
|
+
/** A provenance ref in its canonical display spelling. */
|
|
267
|
+
function canonicalXref(ref) {
|
|
288
268
|
try {
|
|
289
269
|
const p = parseRefInput(ref);
|
|
290
270
|
return displayRef({ type: p.type, name: p.name, bundleId: p.origin });
|
|
291
271
|
}
|
|
292
272
|
catch {
|
|
293
|
-
return
|
|
273
|
+
return ref;
|
|
294
274
|
}
|
|
295
275
|
}
|
|
296
|
-
function canonicalXref(ref) {
|
|
297
|
-
return canonicalStoredXref(ref) ?? ref;
|
|
298
|
-
}
|
|
299
|
-
/**
|
|
300
|
-
* The promoted asset's provenance xref set: existing body-frontmatter xrefs +
|
|
301
|
-
* the promoted source ref, deduped after canonicalization (WI-8.5b: emitted in
|
|
302
|
-
* the D-R5 new grammar via {@link canonicalXref}).
|
|
303
|
-
*/
|
|
304
|
-
function promoteProvenanceXrefs(existing, sourceRef) {
|
|
305
|
-
const priors = Array.isArray(existing) ? existing.map(String) : [];
|
|
306
|
-
return [...new Set([...priors, sourceRef].map(canonicalXref))];
|
|
307
|
-
}
|
|
308
|
-
// ── LLM resolution ──────────────────────────────────────────────────────────
|
|
309
|
-
/**
|
|
310
|
-
* Resolve the symbolic LLM runner for the consolidate pass.
|
|
311
|
-
*
|
|
312
|
-
* Priority order (mirrors extract / reflect / distill — see
|
|
313
|
-
* `resolveExtractRunConfig` in `src/commands/improve/extract.ts` and the
|
|
314
|
-
* canonical improve execution-cascade pattern):
|
|
315
|
-
*
|
|
316
|
-
* 1. `improve.strategies.<name>.processes.consolidate.engine`
|
|
317
|
-
* via the common execution planner. Lets the user pin
|
|
318
|
-
* a dedicated model (e.g. `ministral-3b`) for consolidation instead of
|
|
319
|
-
* whatever `defaults.llmEngine` happens to be.
|
|
320
|
-
* 2. the baseline default LLM engine.
|
|
321
|
-
*
|
|
322
|
-
* All consolidate execution crosses the same improve engine-resolution
|
|
323
|
-
* boundary as extract, reflect, and distill.
|
|
324
|
-
*/
|
|
325
|
-
function resolveConsolidateLlmRunner(config, activeProfile) {
|
|
326
|
-
return resolveImproveLlmExecution({
|
|
327
|
-
config,
|
|
328
|
-
profile: activeProfile,
|
|
329
|
-
process: getImproveProcessConfig("consolidate", activeProfile),
|
|
330
|
-
processName: "consolidate",
|
|
331
|
-
});
|
|
332
|
-
}
|
|
333
|
-
function consolidateRunnerFromOptions(opts, config) {
|
|
334
|
-
if (Object.hasOwn(opts, "llmRunner"))
|
|
335
|
-
return opts.llmRunner ?? undefined;
|
|
336
|
-
const resolved = resolveConsolidateLlmRunner(config, opts.improveProfile);
|
|
337
|
-
if (resolved)
|
|
338
|
-
opts.onNotices?.(resolved.notices);
|
|
339
|
-
return resolved?.runner;
|
|
340
|
-
}
|
|
341
|
-
/**
|
|
342
|
-
* Build a {@link ConsolidateResult} from partial overrides, filling the envelope
|
|
343
|
-
* defaults (schemaVersion / ok / shape + the zeroed counters). Collapses the
|
|
344
|
-
* ~7 near-identical result literals that previously appeared verbatim at every
|
|
345
|
-
* early-return site and the final return of `akmConsolidateInner`. Callers pass
|
|
346
|
-
* only the fields that differ from the all-zero, ok, non-preview baseline.
|
|
347
|
-
*/
|
|
348
276
|
export function makeConsolidateResult(overrides) {
|
|
349
277
|
return {
|
|
350
278
|
schemaVersion: 1,
|
|
@@ -361,7 +289,6 @@ export function makeConsolidateResult(overrides) {
|
|
|
361
289
|
...overrides,
|
|
362
290
|
};
|
|
363
291
|
}
|
|
364
|
-
// ── Main entry point ─────────────────────────────────────────────────────────
|
|
365
292
|
function resolveConsolidationWriteTarget(opts, config) {
|
|
366
293
|
if (opts.writeTarget) {
|
|
367
294
|
const root = path.resolve(opts.writeTarget.source.path);
|
|
@@ -374,140 +301,78 @@ function resolveConsolidationWriteTarget(opts, config) {
|
|
|
374
301
|
},
|
|
375
302
|
};
|
|
376
303
|
}
|
|
377
|
-
if (opts.target) {
|
|
378
|
-
const target = resolveWriteTarget(config, opts.target);
|
|
379
|
-
return { ...target, source: { ...target.source, path: path.resolve(target.source.path) } };
|
|
380
|
-
}
|
|
381
|
-
if (opts.stashDir) {
|
|
304
|
+
if (!opts.target && opts.stashDir) {
|
|
382
305
|
const root = path.resolve(opts.stashDir);
|
|
383
306
|
return {
|
|
384
307
|
source: { kind: "filesystem", name: "stash", path: root, adapterId: detectAdapterId(root) },
|
|
385
308
|
config: { type: "filesystem", name: "stash", path: root, writable: true },
|
|
386
309
|
};
|
|
387
310
|
}
|
|
388
|
-
const target = resolveWriteTarget(config);
|
|
311
|
+
const target = resolveWriteTarget(config, opts.target);
|
|
389
312
|
return { ...target, source: { ...target.source, path: path.resolve(target.source.path) } };
|
|
390
313
|
}
|
|
391
314
|
export async function akmConsolidate(opts = {}) {
|
|
392
315
|
const startMs = Date.now();
|
|
393
|
-
// Derive a stable PROV-DM token for this run. Callers (e.g. akmImprove)
|
|
394
|
-
// should pass opts.sourceRun to tie proposals back to the parent run;
|
|
395
|
-
// standalone `akm consolidate` gets a self-contained token.
|
|
396
|
-
const sourceRun = opts.sourceRun ?? `consolidate-${startMs}`;
|
|
397
316
|
const config = opts.config ?? loadConfig();
|
|
398
317
|
const writeTarget = resolveConsolidationWriteTarget(opts, config);
|
|
399
|
-
|
|
400
|
-
const activeProfile = opts.improveProfile ?? resolveImproveStrategy(undefined, config).config;
|
|
401
|
-
opts = { ...opts, improveProfile: activeProfile };
|
|
318
|
+
const profile = opts.improveProfile ?? resolveImproveStrategy(undefined, config).config;
|
|
402
319
|
const stashDir = writeTarget.source.path;
|
|
403
|
-
const
|
|
404
|
-
const
|
|
405
|
-
const
|
|
406
|
-
|
|
407
|
-
|
|
408
|
-
|
|
320
|
+
const notices = noticeSet(opts.onNotices);
|
|
321
|
+
const enabled = resolveProcessEnabled("consolidate", profile);
|
|
322
|
+
const runner = enabled ? stageRunner(opts, config, profile, "consolidate", notices.add) : undefined;
|
|
323
|
+
opts = {
|
|
324
|
+
...opts,
|
|
325
|
+
target: writeTarget.source.name,
|
|
326
|
+
writeTarget,
|
|
327
|
+
improveProfile: profile,
|
|
328
|
+
onNotices: notices.add,
|
|
329
|
+
sourceRun: opts.sourceRun ?? `consolidate-${startMs}`,
|
|
330
|
+
// Every later reader sees this one runner snapshot.
|
|
331
|
+
llmRunner: runner ?? null,
|
|
409
332
|
};
|
|
410
|
-
|
|
411
|
-
|
|
412
|
-
|
|
413
|
-
|
|
414
|
-
|
|
415
|
-
|
|
416
|
-
opts = { ...opts, llmRunner: frozenLlmRunner ?? null };
|
|
417
|
-
const withNotices = (result) => executionNotices.size > 0 ? { ...result, notices: Object.freeze([...executionNotices.values()]) } : result;
|
|
418
|
-
// WI-9.10: construct this run's RunContext from values already resolved
|
|
419
|
-
// above (sourceRun, config, stashDir) — no second config load, no new db
|
|
420
|
-
// handle. consolidate.ts has no `eventsCtx`/proposals-`ctx` option at all
|
|
421
|
-
// (WS-3a retired its only appendEvent usage; `emitProposal` here is always
|
|
422
|
-
// called with the default, seam-less ProposalsContext — see
|
|
423
|
-
// emitPromotionProposal below), so both get the safe empty-object default,
|
|
424
|
-
// behaviorally identical to `undefined` (EventsContext/ProposalsContext
|
|
425
|
-
// fields are all optional-chained by their consumers). LLM work uses the
|
|
426
|
-
// already-frozen symbolic runner through the shared dispatch seam.
|
|
427
|
-
const runContext = createRunContext({
|
|
428
|
-
stashDir,
|
|
429
|
-
config,
|
|
430
|
-
eventsCtx: {},
|
|
431
|
-
proposalsCtx: {},
|
|
432
|
-
getLlmRunner: () => opts.llmRunner ?? null,
|
|
433
|
-
sourceRun,
|
|
434
|
-
dryRun: opts.dryRun ?? false,
|
|
435
|
-
signal: opts.signal,
|
|
436
|
-
});
|
|
437
|
-
const warnings = [];
|
|
438
|
-
if (!consolidateEnabled) {
|
|
439
|
-
return withNotices(makeConsolidateResult({
|
|
440
|
-
// Sourced from runContext (identical value to `opts.dryRun ?? false`)
|
|
441
|
-
// so the constructed RunContext has a genuine downstream reference —
|
|
442
|
-
// consolidate's own content-read sites are out of this stage's stated
|
|
443
|
-
// item-2 scope (reflect + distill only; see the WI-9.10c report).
|
|
444
|
-
dryRun: runContext.dryRun,
|
|
445
|
-
target: opts.target ?? stashDir,
|
|
446
|
-
durationMs: Date.now() - startMs,
|
|
447
|
-
warnings,
|
|
448
|
-
}));
|
|
333
|
+
if (!enabled) {
|
|
334
|
+
const target = opts.target ?? stashDir;
|
|
335
|
+
return {
|
|
336
|
+
...makeConsolidateResult({ dryRun: opts.dryRun ?? false, target, durationMs: Date.now() - startMs }),
|
|
337
|
+
...notices.fields(),
|
|
338
|
+
};
|
|
449
339
|
}
|
|
450
|
-
//
|
|
451
|
-
|
|
452
|
-
// receive this handle; it is closed in the `finally` block below.
|
|
453
|
-
// Fail-open: any open error leaves it `undefined` and all cache paths skip.
|
|
454
|
-
let sharedStateDb;
|
|
340
|
+
// One state.db handle for the embedding cache; unavailable means no cache.
|
|
341
|
+
let stateDb;
|
|
455
342
|
if (config.embedding) {
|
|
456
343
|
try {
|
|
457
|
-
|
|
344
|
+
stateDb = openStateDatabase();
|
|
458
345
|
}
|
|
459
346
|
catch {
|
|
460
|
-
|
|
347
|
+
stateDb = undefined;
|
|
461
348
|
}
|
|
462
349
|
}
|
|
463
350
|
try {
|
|
464
|
-
return
|
|
351
|
+
return { ...(await consolidate(opts, config, stashDir, startMs, stateDb)), ...notices.fields() };
|
|
465
352
|
}
|
|
466
353
|
finally {
|
|
467
|
-
|
|
354
|
+
stateDb?.close();
|
|
468
355
|
}
|
|
469
356
|
}
|
|
470
|
-
|
|
471
|
-
|
|
472
|
-
|
|
473
|
-
|
|
474
|
-
|
|
475
|
-
|
|
476
|
-
|
|
477
|
-
|
|
478
|
-
|
|
479
|
-
|
|
480
|
-
|
|
481
|
-
acc.
|
|
482
|
-
// 2026-05-27 cross-chunk double-count fix: if `ref` already contributed
|
|
483
|
-
// to judgedNoAction in its own chunk (a different chunk proposed an op
|
|
484
|
-
// for it that is now being rejected here), promote it from the
|
|
485
|
-
// judgedNoAction bucket into the more specific skipReason bucket.
|
|
486
|
-
// Preserves the invariant: processed == actioned + judgedNoAction +
|
|
487
|
-
// Σ(skipReasons) + failedChunkMemories.
|
|
488
|
-
if (acc.judgedNoActionRefs.delete(ref))
|
|
489
|
-
acc.judgedNoAction--;
|
|
490
|
-
const existing = acc.skipReasonByRef.get(ref);
|
|
491
|
-
if (existing) {
|
|
492
|
-
// Already counted once for accounting. Append the extra skip to the
|
|
493
|
-
// ref's grouped entry for observability without adding a new array
|
|
494
|
-
// entry (which would break the accounting invariant).
|
|
495
|
-
existing.skips.push({ op, reason });
|
|
496
|
-
return;
|
|
497
|
-
}
|
|
498
|
-
const entry = { ref, skips: [{ op, reason }] };
|
|
499
|
-
acc.skipReasonByRef.set(ref, entry);
|
|
500
|
-
acc.skipReasons.push(entry);
|
|
501
|
-
};
|
|
502
|
-
return acc;
|
|
357
|
+
function pushSkipReason(acc, op, ref, reason) {
|
|
358
|
+
if (acc.judgedNoActionRefs.delete(ref))
|
|
359
|
+
acc.judgedNoAction--;
|
|
360
|
+
const existing = acc.skipReasonByRef.get(ref);
|
|
361
|
+
if (existing) {
|
|
362
|
+
// One entry per ref keeps the invariant; the extra reason is kept for observability.
|
|
363
|
+
existing.skips.push({ op, reason });
|
|
364
|
+
return;
|
|
365
|
+
}
|
|
366
|
+
const entry = { ref, skips: [{ op, reason }] };
|
|
367
|
+
acc.skipReasonByRef.set(ref, entry);
|
|
368
|
+
acc.skipReasons.push(entry);
|
|
503
369
|
}
|
|
504
370
|
function resolveConsolidationSourceOwner(opts, stashDir) {
|
|
505
371
|
const targetRoot = path.resolve(opts.writeTarget?.source.path ?? stashDir);
|
|
506
372
|
try {
|
|
507
373
|
const sources = resolveSourceEntries(stashDir, opts.config);
|
|
508
|
-
const installations = deriveInstallations(sources);
|
|
509
374
|
const targetIndex = sources.findIndex((source) => path.resolve(source.path) === targetRoot);
|
|
510
|
-
const target =
|
|
375
|
+
const target = deriveInstallations(sources)[targetIndex];
|
|
511
376
|
if (!target)
|
|
512
377
|
return undefined;
|
|
513
378
|
return {
|
|
@@ -523,20 +388,46 @@ function resolveConsolidationSourceOwner(opts, stashDir) {
|
|
|
523
388
|
return undefined;
|
|
524
389
|
}
|
|
525
390
|
}
|
|
391
|
+
const mtimeMsOf = (memory) => {
|
|
392
|
+
try {
|
|
393
|
+
return fs.statSync(memory.filePath).mtimeMs;
|
|
394
|
+
}
|
|
395
|
+
catch {
|
|
396
|
+
return 0;
|
|
397
|
+
}
|
|
398
|
+
};
|
|
526
399
|
/**
|
|
527
|
-
*
|
|
528
|
-
*
|
|
400
|
+
* The exact pool the live pass consumes, with no embedding, model call or
|
|
401
|
+
* write: on-disk eligible memories, minus those the ledger holds, narrowed
|
|
402
|
+
* incrementally, minus bodies already in `knowledge/`, capped to `limit`
|
|
403
|
+
* (oldest-modified first). Shared by preview and execution.
|
|
529
404
|
*/
|
|
530
405
|
export function inspectConsolidationPool(opts, stashDir, warnings, existingKnowledgeBodyHashes = new Set(), access) {
|
|
531
406
|
const readOnly = access?.readOnly === true;
|
|
532
|
-
|
|
533
|
-
let memories = loadMemoriesForSource(sourceOwner, warnings, readOnly);
|
|
407
|
+
let memories = loadMemoriesForSource(resolveConsolidationSourceOwner(opts, stashDir), warnings, readOnly);
|
|
534
408
|
const staleCount = memories.filter((memory) => !fs.existsSync(memory.filePath)).length;
|
|
535
409
|
if (staleCount > 0) {
|
|
536
410
|
warnings.push(`Pre-flight: filtered ${staleCount} stale DB entr${staleCount === 1 ? "y" : "ies"} (file absent on disk) from memory pool before chunking.`);
|
|
537
411
|
}
|
|
538
412
|
memories = memories.filter((memory) => fs.existsSync(memory.filePath));
|
|
539
413
|
const poolSize = memories.length;
|
|
414
|
+
// A memory judged within its revisit window comes back once it is edited.
|
|
415
|
+
const ledger = loadLedgerSnapshot({ proposalsCtx: opts.proposalsCtx, readOnly }, stashDir, ["consolidate"]);
|
|
416
|
+
if (ledger.size > 0) {
|
|
417
|
+
const nowIso = new Date().toISOString();
|
|
418
|
+
memories = memories.filter((memory) => {
|
|
419
|
+
const row = ledger.get(ledgerKey("consolidate", conceptIdFromTypeName("memory", memory.name)));
|
|
420
|
+
let changedAt;
|
|
421
|
+
try {
|
|
422
|
+
changedAt = fs.statSync(memory.filePath).mtime.toISOString();
|
|
423
|
+
}
|
|
424
|
+
catch {
|
|
425
|
+
changedAt = undefined;
|
|
426
|
+
}
|
|
427
|
+
return !row || !isLedgerBlocked(row, nowIso, changedAt);
|
|
428
|
+
});
|
|
429
|
+
}
|
|
430
|
+
const judgedUnchanged = poolSize - memories.length;
|
|
540
431
|
if (opts.incrementalSince && memories.length > 0) {
|
|
541
432
|
memories = narrowToIncrementalCandidates(memories, opts.incrementalSince, warnings, opts.neighborsPerChanged, readOnly);
|
|
542
433
|
}
|
|
@@ -544,859 +435,445 @@ export function inspectConsolidationPool(opts, stashDir, warnings, existingKnowl
|
|
|
544
435
|
if (opts.limit === undefined && memories.length > 150) {
|
|
545
436
|
warnings.push(`Consolidation: pool has ${memories.length} memories and no limit is set. Consider adding a limit to your consolidate config to prevent timeouts on slow LLM endpoints.`);
|
|
546
437
|
}
|
|
547
|
-
//
|
|
548
|
-
|
|
549
|
-
|
|
550
|
-
|
|
551
|
-
|
|
552
|
-
const { memories: prefiltered, prefilteredAlreadyPromoted } = prefilterAlreadyPromotedMemories(memories, existingKnowledgeBodyHashes);
|
|
553
|
-
memories = prefiltered;
|
|
554
|
-
if (opts.limit !== undefined && memories.length > opts.limit) {
|
|
555
|
-
const mtimeOf = (memory) => {
|
|
438
|
+
// Before the limit, so the cap picks from memories the run can act on.
|
|
439
|
+
let prefilteredAlreadyPromoted = 0;
|
|
440
|
+
if (existingKnowledgeBodyHashes.size > 0) {
|
|
441
|
+
memories = memories.filter((memory) => {
|
|
442
|
+
let raw;
|
|
556
443
|
try {
|
|
557
|
-
|
|
444
|
+
raw = fs.readFileSync(memory.filePath, "utf8");
|
|
558
445
|
}
|
|
559
446
|
catch {
|
|
560
|
-
return
|
|
447
|
+
return true;
|
|
561
448
|
}
|
|
562
|
-
|
|
563
|
-
|
|
564
|
-
|
|
449
|
+
const duplicate = existingKnowledgeBodyHashes.has(contentHash(raw, "body"));
|
|
450
|
+
if (duplicate)
|
|
451
|
+
prefilteredAlreadyPromoted++;
|
|
452
|
+
return !duplicate;
|
|
453
|
+
});
|
|
454
|
+
}
|
|
455
|
+
if (opts.limit !== undefined && memories.length > opts.limit) {
|
|
456
|
+
const mtimes = new Map(memories.map((memory) => [memory.filePath, mtimeMsOf(memory)]));
|
|
457
|
+
memories = [...memories].sort((a, b) => (mtimes.get(a.filePath) ?? 0) - (mtimes.get(b.filePath) ?? 0));
|
|
565
458
|
warnings.push(`Consolidation: pool capped at ${opts.limit} of ${memories.length} memories (limit option, oldest-modified first).`);
|
|
566
459
|
memories = memories.slice(0, opts.limit);
|
|
567
460
|
}
|
|
568
|
-
return {
|
|
569
|
-
|
|
570
|
-
|
|
571
|
-
|
|
572
|
-
|
|
573
|
-
|
|
574
|
-
|
|
575
|
-
|
|
576
|
-
* Duplicate`, used to hash `cacheHash(parseFrontmatter(memoryContent).content
|
|
577
|
-
* .trim())` — double-stripping frontmatter and diverging from this pre-filter
|
|
578
|
-
* for a memory body starting with its own `---` block. It now hashes
|
|
579
|
-
* `cacheHash(memoryContent)` directly, matching this single-strip domain.)
|
|
580
|
-
* An unreadable memory is kept (fail-safe: let the later passes surface the
|
|
581
|
-
* read error).
|
|
582
|
-
*/
|
|
583
|
-
function prefilterAlreadyPromotedMemories(memories, existingKnowledgeBodyHashes) {
|
|
584
|
-
if (existingKnowledgeBodyHashes.size === 0)
|
|
585
|
-
return { memories, prefilteredAlreadyPromoted: 0 };
|
|
586
|
-
const kept = [];
|
|
587
|
-
let prefilteredAlreadyPromoted = 0;
|
|
588
|
-
for (const memory of memories) {
|
|
589
|
-
let raw;
|
|
590
|
-
try {
|
|
591
|
-
raw = fs.readFileSync(memory.filePath, "utf8");
|
|
592
|
-
}
|
|
593
|
-
catch {
|
|
594
|
-
kept.push(memory);
|
|
595
|
-
continue;
|
|
596
|
-
}
|
|
597
|
-
if (existingKnowledgeBodyHashes.has(cacheHash(raw))) {
|
|
598
|
-
prefilteredAlreadyPromoted++;
|
|
599
|
-
}
|
|
600
|
-
else {
|
|
601
|
-
kept.push(memory);
|
|
602
|
-
}
|
|
603
|
-
}
|
|
604
|
-
return { memories: kept, prefilteredAlreadyPromoted };
|
|
605
|
-
}
|
|
606
|
-
/**
|
|
607
|
-
* Pass 1 — narrow the memory pool before any LLM work: drop stale DB entries,
|
|
608
|
-
* apply incremental-since narrowing, pre-filter memories already promoted
|
|
609
|
-
* verbatim into `knowledge/` (R5 (b) — discovered previously only after the
|
|
610
|
-
* LLM chunk call, paying for the judgement on ~84% of the pool just to skip
|
|
611
|
-
* it), and cap to `opts.limit` (oldest-modified first, drawn from the
|
|
612
|
-
* pre-filtered pool — R2-1). Returns an early envelope when the pool empties
|
|
613
|
-
* at any stage; otherwise returns the narrowed pool and the state the
|
|
614
|
-
* plan/apply passes consume.
|
|
615
|
-
*/
|
|
616
|
-
async function narrowConsolidationPool(opts, stashDir, startMs, warnings, existingKnowledgeBodyHashes) {
|
|
617
|
-
const snapshot = inspectConsolidationPool(opts, stashDir, warnings, existingKnowledgeBodyHashes);
|
|
618
|
-
const { memories, prefilteredAlreadyPromoted } = snapshot;
|
|
619
|
-
if (prefilteredAlreadyPromoted > 0) {
|
|
620
|
-
warnings.push(`Consolidation: pre-filtered ${prefilteredAlreadyPromoted} memor${prefilteredAlreadyPromoted === 1 ? "y" : "ies"} whose body already exists verbatim in knowledge/ before chunking.`);
|
|
621
|
-
}
|
|
622
|
-
// (The former WS-3b Step 0a homeostatic demotion pass was removed — R4:
|
|
623
|
-
// it was default-off and self-undoing (the next salience recompute
|
|
624
|
-
// unconditionally overwrote the demoted values). Continuous decay now lives
|
|
625
|
-
// in computeSalience's recency term, whose floor decays on a long half-life.)
|
|
626
|
-
if (memories.length === 0) {
|
|
627
|
-
return {
|
|
628
|
-
done: true,
|
|
629
|
-
result: makeConsolidateResult({
|
|
630
|
-
dryRun: opts.dryRun ?? false,
|
|
631
|
-
target: opts.target ?? stashDir,
|
|
632
|
-
warnings,
|
|
633
|
-
durationMs: Date.now() - startMs,
|
|
634
|
-
prefilteredAlreadyPromoted,
|
|
635
|
-
}),
|
|
636
|
-
};
|
|
637
|
-
}
|
|
638
|
-
return { done: false, memories, dedupPoolSize: snapshot.dedupPoolSize, prefilteredAlreadyPromoted };
|
|
639
|
-
}
|
|
640
|
-
/**
|
|
641
|
-
* Pass 2 — turn the narrowed pool into an executable plan. Sizes chunks to the
|
|
642
|
-
* model context window, clusters by embedding similarity, injects the
|
|
643
|
-
* anti-collapse random fraction, applies the cold-start budget cap, runs the
|
|
644
|
-
* per-chunk LLM calls (with retry + failure-rate abort), and reconciles the
|
|
645
|
-
* per-chunk op arrays via {@link mergePlans}. Populates `accounting` in place.
|
|
646
|
-
* Behavior-identical to the former inlined plan-generation block.
|
|
647
|
-
*/
|
|
648
|
-
/**
|
|
649
|
-
* Per-chunk judgedNoAction accounting: count memories the LLM saw inside a chunk
|
|
650
|
-
* but proposed no op for. Membership is by `memory:<name>` ref against the
|
|
651
|
-
* targets of each op (primary + secondaries for merge; ref otherwise). 2026-05-26:
|
|
652
|
-
* pre-fix this was a 78/119 (66%) silent drop in the cron run — no warning,
|
|
653
|
-
* event, or counter. See tuning investigation §Q2. Moved verbatim.
|
|
654
|
-
*/
|
|
655
|
-
function recordChunkJudgedNoAction(chunk, ops, accounting) {
|
|
656
|
-
const targetRefs = new Set();
|
|
657
|
-
for (const op of ops) {
|
|
658
|
-
if (op.op === "merge") {
|
|
659
|
-
targetRefs.add(op.primary);
|
|
660
|
-
for (const s of op.secondaries)
|
|
661
|
-
targetRefs.add(s);
|
|
662
|
-
}
|
|
663
|
-
else {
|
|
664
|
-
targetRefs.add(op.ref);
|
|
665
|
-
}
|
|
666
|
-
}
|
|
667
|
-
let chunkNoAction = 0;
|
|
668
|
-
for (const m of chunk) {
|
|
669
|
-
const memRef = conceptIdFromTypeName("memory", m.name);
|
|
670
|
-
if (!targetRefs.has(memRef)) {
|
|
671
|
-
chunkNoAction++;
|
|
672
|
-
accounting.judgedNoActionRefs.add(memRef);
|
|
673
|
-
}
|
|
674
|
-
}
|
|
675
|
-
accounting.judgedNoAction += chunkNoAction;
|
|
461
|
+
return {
|
|
462
|
+
poolSize,
|
|
463
|
+
candidatePoolSize: memories.length,
|
|
464
|
+
dedupPoolSize,
|
|
465
|
+
memories,
|
|
466
|
+
prefilteredAlreadyPromoted,
|
|
467
|
+
judgedUnchanged,
|
|
468
|
+
};
|
|
676
469
|
}
|
|
470
|
+
const ABORT_MIN_CHUNKS = 4;
|
|
471
|
+
const ABORT_FAILURE_RATE = 0.5;
|
|
677
472
|
/**
|
|
678
|
-
*
|
|
679
|
-
* chunks
|
|
680
|
-
* (
|
|
681
|
-
*
|
|
682
|
-
* abort-rate policy, all-hot early-exit, and the 2026-05-26 accounting invariant
|
|
683
|
-
* (`processed == actioned + judgedNoAction + Σ(skipReasons) + failedChunkMemories`)
|
|
684
|
-
* are byte-identical, and every counter-increment point is unmoved.
|
|
473
|
+
* The chunk loop: stop cleanly on the budget signal, abort once ≥50% of at
|
|
474
|
+
* least 4 chunks failed (the model is likely down), skip an all-hot chunk
|
|
475
|
+
* without a call (the only thing the model could do with it is refused), and
|
|
476
|
+
* count every memory into exactly one accounting bucket.
|
|
685
477
|
*/
|
|
686
478
|
async function judgeConsolidationChunks(args) {
|
|
687
|
-
const { chunks, opts, config,
|
|
688
|
-
const
|
|
689
|
-
|
|
690
|
-
|
|
691
|
-
|
|
692
|
-
|
|
693
|
-
|
|
694
|
-
|
|
695
|
-
|
|
696
|
-
|
|
697
|
-
|
|
698
|
-
|
|
699
|
-
|
|
700
|
-
|
|
701
|
-
|
|
702
|
-
const ABORT_FAILURE_RATE = 0.5;
|
|
479
|
+
const { chunks, opts, config, warnings, acc } = args;
|
|
480
|
+
const llmRunner = opts.llmRunner ?? undefined;
|
|
481
|
+
const memRef = (m) => conceptIdFromTypeName("memory", m.name);
|
|
482
|
+
const failChunk = (message, chunk) => {
|
|
483
|
+
warn(message);
|
|
484
|
+
warnings.push(message);
|
|
485
|
+
acc.totalChunksFailed++;
|
|
486
|
+
acc.failedChunkMemories += chunk.length;
|
|
487
|
+
};
|
|
488
|
+
const skipRemaining = (from) => {
|
|
489
|
+
for (let i = from; i < chunks.length; i++)
|
|
490
|
+
acc.failedChunkMemories += chunks[i].length;
|
|
491
|
+
};
|
|
492
|
+
const planned = [];
|
|
493
|
+
let processed = 0;
|
|
703
494
|
for (let chunkIdx = 0; chunkIdx < chunks.length; chunkIdx++) {
|
|
704
|
-
|
|
705
|
-
// caller's budget has been exhausted. Commits work done so far.
|
|
495
|
+
const label = `chunk ${chunkIdx + 1}`;
|
|
706
496
|
if (opts.signal?.aborted) {
|
|
707
|
-
const
|
|
708
|
-
const msg = `[consolidate] budget signal aborted before chunk ${chunkIdx + 1}/${chunks.length}; ${skipped} chunk(s) not processed (partial_timeout — work done so far committed).`;
|
|
497
|
+
const msg = `[consolidate] budget signal aborted before chunk ${chunkIdx + 1}/${chunks.length}; ${chunks.length - chunkIdx} chunk(s) not processed (partial_timeout — work done so far committed).`;
|
|
709
498
|
warn(msg);
|
|
710
499
|
warnings.push(msg);
|
|
711
|
-
|
|
712
|
-
for (let i = chunkIdx; i < chunks.length; i++) {
|
|
713
|
-
accounting.failedChunkMemories += chunks[i].length;
|
|
714
|
-
}
|
|
500
|
+
skipRemaining(chunkIdx);
|
|
715
501
|
break;
|
|
716
502
|
}
|
|
717
|
-
|
|
718
|
-
|
|
719
|
-
const failureRate = accounting.totalChunksFailed / totalChunksProcessed;
|
|
503
|
+
if (processed >= ABORT_MIN_CHUNKS) {
|
|
504
|
+
const failureRate = acc.totalChunksFailed / processed;
|
|
720
505
|
if (failureRate >= ABORT_FAILURE_RATE) {
|
|
721
|
-
const
|
|
722
|
-
|
|
723
|
-
|
|
724
|
-
|
|
725
|
-
// Account for memories in chunks we never attempted: they are
|
|
726
|
-
// neither judgedNoAction (no plan parsed) nor skipReason (no op
|
|
727
|
-
// rejected). Without this, the accounting invariant fails by
|
|
728
|
-
// `Σ(unattempted_chunk.length)` whenever the abort fires.
|
|
729
|
-
for (let i = chunkIdx; i < chunks.length; i++) {
|
|
730
|
-
accounting.failedChunkMemories += chunks[i].length;
|
|
731
|
-
}
|
|
506
|
+
const msg = `Consolidation aborted — failure rate ${(failureRate * 100).toFixed(0)}% over ${processed} chunks (>= ${ABORT_FAILURE_RATE * 100}% threshold). LLM may be unavailable. ${chunks.length - chunkIdx} chunk(s) skipped.`;
|
|
507
|
+
warn(msg);
|
|
508
|
+
warnings.push(msg);
|
|
509
|
+
skipRemaining(chunkIdx);
|
|
732
510
|
break;
|
|
733
511
|
}
|
|
734
512
|
}
|
|
735
513
|
const chunk = chunks[chunkIdx];
|
|
736
|
-
// All-hot chunk early-exit. The per-prompt hot-list block (see
|
|
737
|
-
// buildChunkPrompt) only *discourages* delete proposals on a mixed chunk;
|
|
738
|
-
// when EVERY memory in the chunk is captureMode: hot, the only ops the LLM
|
|
739
|
-
// could ever propose are deletes — all of which the downstream guard
|
|
740
|
-
// refuses unconditionally. Calling the model is therefore pure token waste.
|
|
741
|
-
// Skip the request entirely and bucket every memory as judgedNoAction (we
|
|
742
|
-
// judged "no action" without spending an LLM call), preserving the
|
|
743
|
-
// accounting invariant `processed == actioned + judgedNoAction +
|
|
744
|
-
// Σ(skipReasons) + failedChunkMemories`. Not counted toward the
|
|
745
|
-
// LLM-failure-rate abort policy — no request was attempted.
|
|
746
514
|
if (chunk.length > 0 && chunk.every((m) => isHotCapturedMemory(m.filePath))) {
|
|
747
|
-
for (const m of chunk)
|
|
748
|
-
|
|
749
|
-
|
|
515
|
+
for (const m of chunk) {
|
|
516
|
+
acc.judgedNoActionRefs.add(memRef(m));
|
|
517
|
+
acc.judgedRefs.add(memRef(m));
|
|
518
|
+
}
|
|
519
|
+
acc.judgedNoAction += chunk.length;
|
|
750
520
|
warn(`[consolidate] chunk ${chunkIdx + 1}/${chunks.length}: all ${chunk.length} memories are captureMode: hot — skipping LLM (judged no-action).`);
|
|
751
521
|
continue;
|
|
752
522
|
}
|
|
753
523
|
warn(`[consolidate] chunk ${chunkIdx + 1}/${chunks.length} (${chunk.length} memories) …`);
|
|
754
|
-
|
|
755
|
-
|
|
756
|
-
|
|
757
|
-
// apart from their fallback error string). responseSchema lift (PR 1,
|
|
758
|
-
// asset-writers-investigation §5): providers with `supportsJsonSchema: true`
|
|
759
|
-
// enforce the shape upstream; others fall through to
|
|
760
|
-
// `parseEmbeddedJsonResponse` on the response side.
|
|
761
|
-
const callChunkLlm = async (fallbackError) => {
|
|
762
|
-
// The gate runs with enabled:true (always open), so this guard is
|
|
763
|
-
// exactly the envelope the gated fn used to return first thing.
|
|
764
|
-
if (!llmRunner)
|
|
765
|
-
return { ok: false, error: "No LLM configured for consolidation" };
|
|
766
|
-
return callStructured({
|
|
767
|
-
feature: "memory_consolidation",
|
|
768
|
-
akmConfig: config,
|
|
769
|
-
enabled: true,
|
|
770
|
-
runner: llmRunner,
|
|
771
|
-
...(lease ? { lease } : {}),
|
|
772
|
-
messages: [
|
|
773
|
-
{ role: "system", content: CONSOLIDATE_SYSTEM_PROMPT },
|
|
774
|
-
{ role: "user", content: userPrompt },
|
|
775
|
-
],
|
|
776
|
-
request: {
|
|
777
|
-
responseSchema: CONSOLIDATE_PLAN_JSON_SCHEMA,
|
|
778
|
-
enableThinking: false,
|
|
779
|
-
timeoutMs: llmRunner.timeoutMs,
|
|
780
|
-
signal: opts.signal,
|
|
781
|
-
},
|
|
782
|
-
parse: (raw) => ({ ok: true, content: raw ?? "" }),
|
|
783
|
-
// A transport throw was caught INSIDE the gated fn and returned as an
|
|
784
|
-
// {ok:false} envelope (never reaching the gate's fallback); onError
|
|
785
|
-
// reproduces that. The fallback fires only on wrapper timeout.
|
|
786
|
-
onError: (_cls, e) => ({ ok: false, error: String(e) }),
|
|
787
|
-
fallback: { ok: false, error: fallbackError },
|
|
788
|
-
...(opts.onNotices ? { onNotices: opts.onNotices } : {}),
|
|
789
|
-
});
|
|
790
|
-
};
|
|
791
|
-
// callChunkLlm already retries once internally (llm/client.ts's
|
|
792
|
-
// chatCompletion, jittered 200-800ms backoff) — a second, outer retry
|
|
793
|
-
// here stacked an uncoordinated fixed 2s backoff on top of it. Removed;
|
|
794
|
-
// only mark the chunk failed once the single retry the client already
|
|
795
|
-
// performs has been exhausted.
|
|
796
|
-
const raw = await callChunkLlm(`chunk ${chunkIdx + 1} failed`);
|
|
797
|
-
if (!raw.ok) {
|
|
798
|
-
warn(raw.error ?? `chunk ${chunkIdx + 1} failed`);
|
|
799
|
-
warnings.push(raw.error ?? `chunk ${chunkIdx + 1} failed`);
|
|
800
|
-
totalChunksProcessed++;
|
|
801
|
-
accounting.totalChunksFailed++;
|
|
802
|
-
// Account for the chunk's memories under the failed-chunk bucket.
|
|
803
|
-
// judgedNoAction does NOT run on this path (it's after the success
|
|
804
|
-
// guards) so without this the accounting invariant breaks on every
|
|
805
|
-
// chunk-level transport/parse failure.
|
|
806
|
-
accounting.failedChunkMemories += chunk.length;
|
|
524
|
+
processed++;
|
|
525
|
+
if (!llmRunner) {
|
|
526
|
+
failChunk("No LLM configured for consolidation", chunk);
|
|
807
527
|
continue;
|
|
808
528
|
}
|
|
809
|
-
//
|
|
810
|
-
|
|
811
|
-
|
|
812
|
-
|
|
813
|
-
|
|
814
|
-
|
|
529
|
+
// The transport already retries once; a failed chunk is not retried here.
|
|
530
|
+
const outcome = await callStage({
|
|
531
|
+
feature: "memory_consolidation",
|
|
532
|
+
runner: llmRunner,
|
|
533
|
+
system: consolidateSystemPrompt,
|
|
534
|
+
prompt: buildChunkPrompt(args.sourceName, chunk, chunkIdx, chunks.length, args.bodyTruncation, args.pendingProposalBodyHashes),
|
|
535
|
+
gate: { config, enabled: true },
|
|
536
|
+
request: {
|
|
537
|
+
responseSchema: CONSOLIDATE_PLAN_JSON_SCHEMA,
|
|
538
|
+
enableThinking: false,
|
|
539
|
+
timeoutMs: llmRunner.timeoutMs,
|
|
540
|
+
signal: opts.signal,
|
|
541
|
+
},
|
|
542
|
+
...(opts.onNotices ? { onNotices: opts.onNotices } : {}),
|
|
543
|
+
});
|
|
544
|
+
if (!outcome.ok) {
|
|
545
|
+
failChunk(outcome.reason === "error" && outcome.error ? outcome.error : `${label} failed`, chunk);
|
|
546
|
+
continue;
|
|
815
547
|
}
|
|
816
|
-
|
|
548
|
+
warnVerbose(`[akm:consolidate] ${label} raw response (first 500 chars): ${outcome.raw.slice(0, 500)}`);
|
|
549
|
+
const parsed = parseEmbeddedJsonResponse(outcome.raw);
|
|
817
550
|
if (!parsed || !Array.isArray(parsed.operations)) {
|
|
818
|
-
const hint = raw.
|
|
819
|
-
|
|
820
|
-
|
|
821
|
-
|
|
822
|
-
|
|
823
|
-
|
|
824
|
-
accounting.totalChunksFailed++;
|
|
825
|
-
accounting.failedChunkMemories += chunk.length;
|
|
551
|
+
const hint = outcome.raw.trim() === "" ? " (empty response — if using a thinking model, disable thinking mode)" : "";
|
|
552
|
+
const msg = `Chunk ${chunkIdx + 1}: invalid plan from AI — skipping.${hint}`;
|
|
553
|
+
warn(msg);
|
|
554
|
+
warnings.push(msg);
|
|
555
|
+
acc.totalChunksFailed++;
|
|
556
|
+
acc.failedChunkMemories += chunk.length;
|
|
826
557
|
continue;
|
|
827
558
|
}
|
|
828
|
-
totalChunksProcessed++; // success
|
|
829
559
|
const ops = [];
|
|
830
560
|
for (const op of parsed.operations) {
|
|
831
|
-
if (isValidOp(op))
|
|
561
|
+
if (isValidOp(op))
|
|
832
562
|
ops.push(op);
|
|
833
|
-
|
|
834
|
-
else {
|
|
563
|
+
else
|
|
835
564
|
warnings.push(`Chunk ${chunkIdx + 1}: skipping invalid operation: ${JSON.stringify(op)}`);
|
|
836
|
-
}
|
|
837
565
|
}
|
|
838
|
-
|
|
839
|
-
|
|
840
|
-
|
|
841
|
-
|
|
842
|
-
|
|
566
|
+
for (const w of Array.isArray(parsed.warnings) ? parsed.warnings : [])
|
|
567
|
+
if (typeof w === "string")
|
|
568
|
+
warnings.push(w);
|
|
569
|
+
// Memories the model saw but proposed nothing for.
|
|
570
|
+
const targeted = new Set(ops.map((op) => op.ref));
|
|
571
|
+
for (const m of chunk) {
|
|
572
|
+
acc.judgedRefs.add(memRef(m));
|
|
573
|
+
if (targeted.has(memRef(m)))
|
|
574
|
+
continue;
|
|
575
|
+
acc.judgedNoAction++;
|
|
576
|
+
acc.judgedNoActionRefs.add(memRef(m));
|
|
843
577
|
}
|
|
844
|
-
|
|
845
|
-
chunkOpsArrays.push(ops);
|
|
578
|
+
planned.push(ops);
|
|
846
579
|
}
|
|
847
|
-
return
|
|
580
|
+
return planned;
|
|
848
581
|
}
|
|
849
|
-
|
|
850
|
-
|
|
851
|
-
|
|
852
|
-
|
|
853
|
-
|
|
854
|
-
|
|
582
|
+
/**
|
|
583
|
+
* The model's plan for the narrowed pool: chunk size from the context window,
|
|
584
|
+
* an up-front cap when the remaining budget cannot cover every chunk (oldest
|
|
585
|
+
* first, the rest deferred), similarity clustering, anti-collapse, then the
|
|
586
|
+
* chunk loop.
|
|
587
|
+
*/
|
|
588
|
+
async function planConsolidation(opts, config, stashDir, memories, warnings, stateDb, acc) {
|
|
855
589
|
const llmRunner = opts.llmRunner ?? undefined;
|
|
856
|
-
//
|
|
857
|
-
// window so that the full prompt (system prompt + chunk user prompt) never
|
|
858
|
-
// exceeds the model's n_ctx limit. When no context length is configured we
|
|
859
|
-
// fall back to DEFAULT_CONTEXT_LENGTH_TOKENS (8 000) which is conservative
|
|
860
|
-
// enough for most 8K–16K local models.
|
|
861
|
-
//
|
|
862
|
-
// bodyTruncation caps the body excerpt included per memory in the prompt.
|
|
863
|
-
// Reducing it further than 500 chars degrades consolidation quality, so we
|
|
864
|
-
// keep it fixed and let computeSafeChunkSize vary the number of memories
|
|
865
|
-
// per chunk instead.
|
|
590
|
+
// 500 body chars per memory keep the judgement useful; chunk size varies instead.
|
|
866
591
|
const bodyTruncation = 500;
|
|
867
|
-
const
|
|
868
|
-
const chunkSize = computeSafeChunkSize(modelContextLength, bodyTruncation, opts.maxChunkSize);
|
|
869
|
-
// -- Phase A: plan generation -----------------------------------------------
|
|
592
|
+
const chunkSize = computeSafeChunkSize(llmRunner?.connection.contextLength ?? DEFAULT_CONTEXT_LENGTH_TOKENS, bodyTruncation, opts.maxChunkSize);
|
|
870
593
|
const sourceName = opts.target ?? stashDir;
|
|
871
|
-
let
|
|
872
|
-
|
|
873
|
-
|
|
874
|
-
|
|
875
|
-
|
|
876
|
-
|
|
877
|
-
|
|
878
|
-
|
|
879
|
-
|
|
880
|
-
|
|
881
|
-
|
|
882
|
-
try {
|
|
883
|
-
mtimeMs = fs.statSync(entry.filePath).mtimeMs;
|
|
884
|
-
}
|
|
885
|
-
catch {
|
|
886
|
-
// Missing files sort first and are filtered by the existing guards.
|
|
887
|
-
}
|
|
888
|
-
return { entry, mtimeMs };
|
|
889
|
-
})
|
|
890
|
-
.sort((a, b) => a.mtimeMs - b.mtimeMs || a.entry.name.localeCompare(b.entry.name))
|
|
891
|
-
.map(({ entry }) => entry)
|
|
892
|
-
.slice(0, cap);
|
|
893
|
-
const msg = `[consolidate] cold-start budget: reducing pool from ${memories.length} to ${budgetedMemories.length} memories (${safeChunks} safe chunks; remainder deferred).`;
|
|
894
|
-
warn(msg);
|
|
895
|
-
warnings.push(msg);
|
|
896
|
-
}
|
|
897
|
-
}
|
|
898
|
-
}
|
|
899
|
-
// WS-5: capture llmPoolSize after every pre-LLM cap.
|
|
900
|
-
const llmPoolSize = budgetedMemories.length;
|
|
901
|
-
const dispatchingChunks = [];
|
|
902
|
-
for (let i = 0; i < budgetedMemories.length; i += chunkSize) {
|
|
903
|
-
const chunk = budgetedMemories.slice(i, i + chunkSize);
|
|
904
|
-
if (chunk.length > 0 && !chunk.every((memory) => isHotCapturedMemory(memory.filePath))) {
|
|
905
|
-
dispatchingChunks.push(chunk);
|
|
906
|
-
}
|
|
907
|
-
}
|
|
908
|
-
const dispatchLease = llmRunner && dispatchingChunks.length > 0 ? await preflightStructuredLlmRunner(llmRunner) : undefined;
|
|
909
|
-
try {
|
|
910
|
-
// C-1 / #380: Pre-cluster memories by embedding similarity before chunking.
|
|
911
|
-
// This ensures that semantically similar memories land in the same LLM
|
|
912
|
-
// context window, allowing the model to detect and merge duplicates that
|
|
913
|
-
// would otherwise be split across chunks and survive indefinitely.
|
|
914
|
-
// mem0 arXiv:2504.19413, A-MEM arXiv:2502.12110.
|
|
915
|
-
// Fails open: if embeddings are unavailable or fail, original order is used.
|
|
916
|
-
const { ordered: clusteredMemories, embedTelemetry } = await clusterMemoriesBySimilarity(budgetedMemories, config, sharedStateDb, opts.signal);
|
|
917
|
-
// WS-3b Anti-collapse step 8c: inject random (non-similar) clusters.
|
|
918
|
-
// A small fraction (default 5%) of the pool is shuffled into random positions
|
|
919
|
-
// so the pipeline isn't PURELY similarity-driven. This prevents rich-get-richer
|
|
920
|
-
// entrenchment where only the most-retrieved assets ever get consolidated.
|
|
921
|
-
// DEFAULT ON since R5 — opt out via antiCollapse.enabled: false.
|
|
922
|
-
let finalClusteredMemories = clusteredMemories;
|
|
923
|
-
{
|
|
924
|
-
const antiCollapseForCluster = getImproveProcessConfig("consolidate", opts.improveProfile)?.antiCollapse ??
|
|
925
|
-
{};
|
|
926
|
-
if (antiCollapseForCluster.enabled !== false && clusteredMemories.length > 2) {
|
|
927
|
-
const fraction = antiCollapseForCluster.randomClusterFraction ?? DEFAULT_RANDOM_CLUSTER_FRACTION;
|
|
928
|
-
const randomCount = Math.max(1, Math.floor(clusteredMemories.length * fraction));
|
|
929
|
-
// Pick `randomCount` positions to inject random (un-clustered) members.
|
|
930
|
-
// Use a seeded-ish shuffle: sort by hash of the name so it's deterministic
|
|
931
|
-
// per run but not strictly similarity-driven.
|
|
932
|
-
const shuffled = [...clusteredMemories].sort((a, b) => {
|
|
933
|
-
// Deterministic shuffle: compare sha256-ish (use name hash as proxy).
|
|
934
|
-
const ha = a.name.split("").reduce((acc, c) => ((acc << 5) - acc + c.charCodeAt(0)) | 0, 0);
|
|
935
|
-
const hb = b.name.split("").reduce((acc, c) => ((acc << 5) - acc + c.charCodeAt(0)) | 0, 0);
|
|
936
|
-
return ha - hb;
|
|
937
|
-
});
|
|
938
|
-
const randomSlice = shuffled.slice(0, randomCount);
|
|
939
|
-
const randomSet = new Set(randomSlice.map((m) => m.name));
|
|
940
|
-
// Insert random members at intervals through the clustered sequence.
|
|
941
|
-
const withRandom = [];
|
|
942
|
-
const interval = Math.max(2, Math.floor(clusteredMemories.length / randomCount));
|
|
943
|
-
let randomIdx = 0;
|
|
944
|
-
for (let i = 0; i < clusteredMemories.length; i++) {
|
|
945
|
-
const m = clusteredMemories[i];
|
|
946
|
-
if (m && !randomSet.has(m.name))
|
|
947
|
-
withRandom.push(m);
|
|
948
|
-
if (i > 0 && i % interval === 0 && randomIdx < randomSlice.length) {
|
|
949
|
-
const r = randomSlice[randomIdx++];
|
|
950
|
-
if (r)
|
|
951
|
-
withRandom.push(r);
|
|
952
|
-
}
|
|
953
|
-
}
|
|
954
|
-
// Append any remaining random members not yet inserted.
|
|
955
|
-
while (randomIdx < randomSlice.length) {
|
|
956
|
-
const r = randomSlice[randomIdx++];
|
|
957
|
-
if (r)
|
|
958
|
-
withRandom.push(r);
|
|
959
|
-
}
|
|
960
|
-
finalClusteredMemories = withRandom;
|
|
961
|
-
warnings.push(`Anti-collapse: injected ${randomCount} random (non-similarity-driven) cluster member(s) into consolidation pool (fraction=${fraction}).`);
|
|
962
|
-
}
|
|
963
|
-
}
|
|
964
|
-
const chunks = [];
|
|
965
|
-
for (let i = 0; i < finalClusteredMemories.length; i += chunkSize) {
|
|
966
|
-
chunks.push(finalClusteredMemories.slice(i, i + chunkSize));
|
|
594
|
+
let budgeted = memories;
|
|
595
|
+
const budgetMs = opts.signal?.remainingBudgetMs;
|
|
596
|
+
if (opts.signal && budgetMs !== undefined) {
|
|
597
|
+
const safeChunks = Math.max(0, Math.floor((Math.max(0, budgetMs) / 1000 / (opts.p90ChunkSecondsDefault ?? 30)) * 0.6));
|
|
598
|
+
if (safeChunks * chunkSize < memories.length) {
|
|
599
|
+
budgeted = [...memories]
|
|
600
|
+
.sort((a, b) => mtimeMsOf(a) - mtimeMsOf(b) || a.name.localeCompare(b.name))
|
|
601
|
+
.slice(0, safeChunks * chunkSize);
|
|
602
|
+
const msg = `[consolidate] cold-start budget: reducing pool from ${memories.length} to ${budgeted.length} memories (${safeChunks} safe chunks; remainder deferred).`;
|
|
603
|
+
warn(msg);
|
|
604
|
+
warnings.push(msg);
|
|
967
605
|
}
|
|
968
|
-
// 2026-05-27 prompt-context fix: precompute body-hashes of pending
|
|
969
|
-
// consolidate proposals once, so the per-chunk prompt can annotate
|
|
970
|
-
// memories whose body would just produce a deterministic
|
|
971
|
-
// `dedup_pending_proposal` skip. Cuts ~110 wasted LLM proposals per
|
|
972
|
-
// 4h on this user's stack. See
|
|
973
|
-
// /tmp/akm-health-investigations/tuning-reasons-investigation.md §Q3.
|
|
974
|
-
const pendingProposalBodyHashes = loadPendingConsolidateProposalHashes(stashDir);
|
|
975
|
-
warn(`[consolidate] ${budgetedMemories.length} memories / ${chunks.length} chunk(s) / chunk_size=${chunkSize}` +
|
|
976
|
-
` / pending-proposal hashes: ${pendingProposalBodyHashes.size}`);
|
|
977
|
-
const chunkOpsArrays = await judgeConsolidationChunks({
|
|
978
|
-
chunks,
|
|
979
|
-
opts,
|
|
980
|
-
config,
|
|
981
|
-
llmRunner,
|
|
982
|
-
lease: dispatchLease,
|
|
983
|
-
sourceName,
|
|
984
|
-
bodyTruncation,
|
|
985
|
-
pendingProposalBodyHashes,
|
|
986
|
-
warnings,
|
|
987
|
-
accounting,
|
|
988
|
-
});
|
|
989
|
-
// Build the known-refs set from the already-filtered memory pool so
|
|
990
|
-
// mergePlans() can reject LLM-hallucinated primary refs before execution.
|
|
991
|
-
const knownRefs = new Set(budgetedMemories.map((m) => conceptIdFromTypeName("memory", m.name)));
|
|
992
|
-
const { ops: allOps, warnings: mergeWarnings } = mergePlans(chunkOpsArrays, knownRefs);
|
|
993
|
-
warnings.push(...mergeWarnings);
|
|
994
|
-
return {
|
|
995
|
-
allOps,
|
|
996
|
-
totalChunks: chunks.length,
|
|
997
|
-
llmPoolSize,
|
|
998
|
-
deferredMemories: memories.length - budgetedMemories.length,
|
|
999
|
-
embedTelemetry,
|
|
1000
|
-
sourceName,
|
|
1001
|
-
};
|
|
1002
|
-
}
|
|
1003
|
-
finally {
|
|
1004
|
-
if (dispatchLease)
|
|
1005
|
-
disposeLoweredExecutionDispatchLease(dispatchLease);
|
|
1006
606
|
}
|
|
607
|
+
const slice = (list) => Array.from({ length: Math.ceil(list.length / chunkSize) }, (_, i) => list.slice(i * chunkSize, (i + 1) * chunkSize));
|
|
608
|
+
const willDispatch = slice(budgeted).some((chunk) => chunk.length > 0 && !chunk.every((memory) => isHotCapturedMemory(memory.filePath)));
|
|
609
|
+
if (llmRunner && willDispatch)
|
|
610
|
+
assertRunnerCredentials(llmRunner);
|
|
611
|
+
const { ordered, embedTelemetry } = await clusterMemoriesBySimilarity(budgeted, config, stateDb, opts.signal);
|
|
612
|
+
const chunks = slice(injectRandomClusterMembers(ordered, opts.improveProfile, warnings));
|
|
613
|
+
const pendingProposalBodyHashes = loadPendingConsolidateProposalHashes(stashDir);
|
|
614
|
+
warn(`[consolidate] ${budgeted.length} memories / ${chunks.length} chunk(s) / chunk_size=${chunkSize}` +
|
|
615
|
+
` / pending-proposal hashes: ${pendingProposalBodyHashes.size}`);
|
|
616
|
+
const planned = await judgeConsolidationChunks({
|
|
617
|
+
chunks,
|
|
618
|
+
opts,
|
|
619
|
+
config,
|
|
620
|
+
sourceName,
|
|
621
|
+
bodyTruncation,
|
|
622
|
+
pendingProposalBodyHashes,
|
|
623
|
+
warnings,
|
|
624
|
+
acc,
|
|
625
|
+
});
|
|
626
|
+
return {
|
|
627
|
+
allOps: mergePlans(planned),
|
|
628
|
+
totalChunks: chunks.length,
|
|
629
|
+
llmPoolSize: budgeted.length,
|
|
630
|
+
deferredMemories: memories.length - budgeted.length,
|
|
631
|
+
embedTelemetry,
|
|
632
|
+
sourceName,
|
|
633
|
+
};
|
|
1007
634
|
}
|
|
1008
|
-
async function
|
|
1009
|
-
|
|
1010
|
-
// the post-LLM promote-dedup check (shouldSkipPromotionBodyDuplicate) below
|
|
1011
|
-
// — knowledge/ can hold thousands of files, so walking it twice per run
|
|
1012
|
-
// would double that cost for no benefit. When the caller already walked
|
|
1013
|
-
// knowledge/ for the pool preview (runConsolidationPass, R2-1/R3-1), reuse
|
|
1014
|
-
// that set instead of walking it again here.
|
|
635
|
+
async function consolidate(opts, config, stashDir, startMs, stateDb) {
|
|
636
|
+
const warnings = [];
|
|
1015
637
|
const existingKnowledgeBodyHashes = opts.existingKnowledgeBodyHashes ?? loadExistingKnowledgeBodyHashes(stashDir);
|
|
1016
|
-
|
|
1017
|
-
const
|
|
1018
|
-
|
|
1019
|
-
|
|
1020
|
-
|
|
1021
|
-
|
|
1022
|
-
|
|
1023
|
-
|
|
1024
|
-
|
|
1025
|
-
|
|
638
|
+
const pool = inspectConsolidationPool(opts, stashDir, warnings, existingKnowledgeBodyHashes);
|
|
639
|
+
const { memories, prefilteredAlreadyPromoted } = pool;
|
|
640
|
+
const plural = (n) => `memor${n === 1 ? "y" : "ies"}`;
|
|
641
|
+
if (pool.judgedUnchanged > 0) {
|
|
642
|
+
warnings.push(`Consolidation: skipped ${pool.judgedUnchanged} ${plural(pool.judgedUnchanged)} judged within the revisit window and unchanged since.`);
|
|
643
|
+
}
|
|
644
|
+
if (prefilteredAlreadyPromoted > 0) {
|
|
645
|
+
warnings.push(`Consolidation: pre-filtered ${prefilteredAlreadyPromoted} ${plural(prefilteredAlreadyPromoted)} whose body already exists verbatim in knowledge/ before chunking.`);
|
|
646
|
+
}
|
|
647
|
+
const target = opts.target ?? stashDir;
|
|
648
|
+
if (memories.length === 0) {
|
|
1026
649
|
return makeConsolidateResult({
|
|
1027
|
-
dryRun:
|
|
1028
|
-
|
|
1029
|
-
target: sourceName,
|
|
1030
|
-
processed: llmPoolSize,
|
|
1031
|
-
failedChunks: accounting.totalChunksFailed,
|
|
1032
|
-
totalChunks,
|
|
1033
|
-
judgedNoAction: accounting.judgedNoAction,
|
|
1034
|
-
skipReasons: accounting.skipReasons,
|
|
1035
|
-
// No merge has executed on the preview path — the per-secondary tally is
|
|
1036
|
-
// provably still 0 here (it only increments in the op-execution loop).
|
|
1037
|
-
mergedSecondaries: 0,
|
|
1038
|
-
failedChunkMemories: accounting.failedChunkMemories,
|
|
1039
|
-
deferredMemories,
|
|
1040
|
-
planned: allOps,
|
|
650
|
+
dryRun: opts.dryRun ?? false,
|
|
651
|
+
target,
|
|
1041
652
|
warnings,
|
|
1042
653
|
durationMs: Date.now() - startMs,
|
|
1043
654
|
prefilteredAlreadyPromoted,
|
|
1044
655
|
});
|
|
1045
656
|
}
|
|
1046
|
-
|
|
1047
|
-
|
|
1048
|
-
|
|
1049
|
-
|
|
1050
|
-
|
|
1051
|
-
|
|
1052
|
-
|
|
657
|
+
const acc = {
|
|
658
|
+
judgedNoAction: 0,
|
|
659
|
+
failedChunkMemories: 0,
|
|
660
|
+
totalChunksFailed: 0,
|
|
661
|
+
skipReasons: [],
|
|
662
|
+
skipReasonByRef: new Map(),
|
|
663
|
+
judgedNoActionRefs: new Set(),
|
|
664
|
+
judgedRefs: new Set(),
|
|
665
|
+
};
|
|
666
|
+
const plan = await planConsolidation(opts, config, stashDir, memories, warnings, stateDb, acc);
|
|
667
|
+
// Evaluated at return time: a promotion skip can move a ref out of judgedNoAction.
|
|
668
|
+
const summary = () => ({
|
|
669
|
+
target: plan.sourceName,
|
|
670
|
+
processed: plan.llmPoolSize,
|
|
671
|
+
failedChunks: acc.totalChunksFailed,
|
|
672
|
+
totalChunks: plan.totalChunks,
|
|
673
|
+
judgedNoAction: acc.judgedNoAction,
|
|
674
|
+
skipReasons: acc.skipReasons,
|
|
675
|
+
mergedSecondaries: 0,
|
|
676
|
+
failedChunkMemories: acc.failedChunkMemories,
|
|
677
|
+
deferredMemories: plan.deferredMemories,
|
|
678
|
+
planned: plan.allOps,
|
|
679
|
+
warnings,
|
|
680
|
+
prefilteredAlreadyPromoted,
|
|
681
|
+
durationMs: Date.now() - startMs,
|
|
682
|
+
});
|
|
683
|
+
if (opts.dryRun)
|
|
684
|
+
return makeConsolidateResult({ ...summary(), dryRun: true, previewOnly: true });
|
|
685
|
+
warn(`[consolidate] plan: ${plan.allOps.length} operation(s)`);
|
|
686
|
+
const ctx = {
|
|
1053
687
|
config,
|
|
1054
688
|
stashDir,
|
|
1055
689
|
sourceRun: opts.sourceRun ?? `consolidate-${startMs}`,
|
|
1056
690
|
proposalsCtx: opts.proposalsCtx,
|
|
1057
691
|
target: opts.writeTarget,
|
|
1058
|
-
memoryByRef,
|
|
1059
|
-
promoted,
|
|
692
|
+
memoryByRef: new Map(memories.map((memory) => [conceptIdFromTypeName("memory", memory.name), memory])),
|
|
693
|
+
promoted: [],
|
|
1060
694
|
promotedSourceRefs: new Set(),
|
|
1061
695
|
existingKnowledgeBodyHashes,
|
|
1062
|
-
promotionFailures,
|
|
696
|
+
promotionFailures: { count: 0 },
|
|
1063
697
|
warnings,
|
|
1064
|
-
pushSkipReason:
|
|
1065
|
-
llmRunner: opts.llmRunner ?? null,
|
|
698
|
+
pushSkipReason: (op, ref, reason) => pushSkipReason(acc, op, ref, reason),
|
|
1066
699
|
};
|
|
1067
|
-
for (const op of allOps)
|
|
1068
|
-
|
|
1069
|
-
|
|
1070
|
-
|
|
700
|
+
for (const op of plan.allOps)
|
|
701
|
+
await emitPromotionProposal(op, ctx);
|
|
702
|
+
// Every other judged memory waits out its revisit window (or its next edit);
|
|
703
|
+
// a promotion that failed to persist is retried next run.
|
|
704
|
+
recordLedgerAttempt({ proposalsCtx: opts.proposalsCtx }, [...acc.judgedRefs]
|
|
705
|
+
.filter((ref) => !ctx.promotedSourceRefs.has(ref) &&
|
|
706
|
+
!acc.skipReasonByRef.get(ref)?.skips.some((skip) => skip.reason === "promote_create_failed"))
|
|
707
|
+
.map((ref) => ({ stashDir, ref, source: "consolidate", outcome: "judged_no_action" })));
|
|
1071
708
|
return makeConsolidateResult({
|
|
1072
|
-
|
|
1073
|
-
|
|
1074
|
-
|
|
1075
|
-
totalChunks,
|
|
1076
|
-
judgedNoAction: accounting.judgedNoAction,
|
|
1077
|
-
skipReasons: accounting.skipReasons,
|
|
1078
|
-
mergedSecondaries: 0,
|
|
1079
|
-
failedChunkMemories: accounting.failedChunkMemories,
|
|
1080
|
-
deferredMemories,
|
|
1081
|
-
promoted,
|
|
1082
|
-
failedPromotions: promotionFailures.count,
|
|
1083
|
-
planned: allOps,
|
|
1084
|
-
warnings,
|
|
1085
|
-
durationMs: Date.now() - startMs,
|
|
1086
|
-
prefilteredAlreadyPromoted,
|
|
709
|
+
...summary(),
|
|
710
|
+
promoted: ctx.promoted,
|
|
711
|
+
failedPromotions: ctx.promotionFailures.count,
|
|
1087
712
|
perfTelemetry: {
|
|
1088
|
-
dedupPoolSize,
|
|
1089
|
-
llmPoolSize,
|
|
1090
|
-
embedMs: embedTelemetry.embedMs,
|
|
1091
|
-
embedCacheHits: embedTelemetry.cacheHits,
|
|
1092
|
-
embedCacheMisses: embedTelemetry.cacheMisses,
|
|
713
|
+
dedupPoolSize: pool.dedupPoolSize,
|
|
714
|
+
llmPoolSize: plan.llmPoolSize,
|
|
715
|
+
embedMs: plan.embedTelemetry.embedMs,
|
|
716
|
+
embedCacheHits: plan.embedTelemetry.cacheHits,
|
|
717
|
+
embedCacheMisses: plan.embedTelemetry.cacheMisses,
|
|
1093
718
|
},
|
|
1094
719
|
});
|
|
1095
720
|
}
|
|
1096
|
-
/**
|
|
1097
|
-
function
|
|
1098
|
-
|
|
1099
|
-
|
|
1100
|
-
|
|
1101
|
-
|
|
1102
|
-
|
|
721
|
+
/** The conceptId a ref maps to, or undefined for an invalid ref. */
|
|
722
|
+
function conceptIdForRef(ref) {
|
|
723
|
+
try {
|
|
724
|
+
const p = parseRefInput(ref);
|
|
725
|
+
return conceptIdFromTypeName(p.type, p.name);
|
|
726
|
+
}
|
|
727
|
+
catch {
|
|
728
|
+
return undefined;
|
|
1103
729
|
}
|
|
1104
|
-
const contentDupProposal = listProposals(ctx.stashDir, { status: "pending" })
|
|
1105
|
-
.filter((proposal) => proposal.source === "consolidate")
|
|
1106
|
-
.find((proposal) => cacheHash(proposalContent(proposal)) === bodyHash);
|
|
1107
|
-
if (!contentDupProposal)
|
|
1108
|
-
return false;
|
|
1109
|
-
ctx.warnings.push(`Skipping promote: identical body already pending as proposal ${contentDupProposal.id} (ref: ${contentDupProposal.ref}); skipping duplicate for ${op.ref} → ${knowledgeRef}`);
|
|
1110
|
-
ctx.pushSkipReason("promote", op.ref, "dedup_pending_proposal");
|
|
1111
|
-
return true;
|
|
1112
730
|
}
|
|
1113
|
-
/**
|
|
1114
|
-
|
|
731
|
+
/** A slug with dates, counters and word order folded away, for spotting variants. */
|
|
732
|
+
function normalizeSlugForDedup(ref) {
|
|
733
|
+
const monthRe = /(?:jan|feb|mar|apr|may|jun|jul|aug|sep|oct|nov|dec)/i;
|
|
734
|
+
return parseRefInput(ref)
|
|
735
|
+
.name.toLowerCase()
|
|
736
|
+
.split("-")
|
|
737
|
+
.filter((tok) => tok.length > 0 && !/^\d+$/.test(tok) && !monthRe.test(tok))
|
|
738
|
+
.sort()
|
|
739
|
+
.join("-");
|
|
740
|
+
}
|
|
741
|
+
const PROMOTE_BODY_MIN_CHARS = 100;
|
|
742
|
+
/**
|
|
743
|
+
* Queue one promotion as a proposal. Refused (with a skip reason) when the
|
|
744
|
+
* memory is unknown, already promoted this run, already pending or present
|
|
745
|
+
* as knowledge (by concept, body hash or slug variant), unreadable, fails
|
|
746
|
+
* sanitization, is superseded, has a body too small to be knowledge, or has
|
|
747
|
+
* no valid description.
|
|
748
|
+
* @internal Exported for promotion-path integration tests.
|
|
749
|
+
*/
|
|
1115
750
|
export async function emitPromotionProposal(op, ctx) {
|
|
1116
|
-
const {
|
|
1117
|
-
const entry = memoryByRef.get(op.ref);
|
|
751
|
+
const { stashDir, target, warnings, pushSkipReason } = ctx;
|
|
752
|
+
const entry = ctx.memoryByRef.get(op.ref);
|
|
1118
753
|
if (!entry) {
|
|
754
|
+
// A phantom ref was never counted as processed, so it gets no skip reason.
|
|
1119
755
|
warnings.push(`Promote: ${op.ref} not found in loaded memories — skipping.`);
|
|
1120
|
-
// Phantom ref: not in processed, so no skipReason (same rationale as
|
|
1121
|
-
// delete_ref_missing above).
|
|
1122
756
|
return;
|
|
1123
757
|
}
|
|
1124
|
-
|
|
1125
|
-
|
|
1126
|
-
|
|
1127
|
-
|
|
1128
|
-
if (promotedSourceRefs.has(op.ref)) {
|
|
1129
|
-
|
|
1130
|
-
pushSkipReason("promote", op.ref, "promote_already_promoted_this_run");
|
|
1131
|
-
return;
|
|
758
|
+
const skip = (reason, message) => {
|
|
759
|
+
warnings.push(message);
|
|
760
|
+
pushSkipReason("promote", op.ref, reason);
|
|
761
|
+
};
|
|
762
|
+
if (ctx.promotedSourceRefs.has(op.ref)) {
|
|
763
|
+
return skip("promote_already_promoted_this_run", `Skipping promote: ${op.ref} already promoted in this run`);
|
|
1132
764
|
}
|
|
1133
|
-
const
|
|
765
|
+
const slug = (op.knowledgeRef.split("/").filter(Boolean).at(-1) ??
|
|
1134
766
|
entry.name.split("/").filter(Boolean).at(-1) ??
|
|
1135
|
-
"promoted-memory"
|
|
1136
|
-
const slug = proposedName
|
|
767
|
+
"promoted-memory")
|
|
1137
768
|
.replace(/[^a-z0-9-]/gi, "-")
|
|
1138
769
|
.replace(/-+/g, "-")
|
|
1139
770
|
.replace(/^-|-$/g, "")
|
|
1140
771
|
.toLowerCase();
|
|
1141
772
|
const knowledgeRef = conceptIdFromTypeName("knowledge", slug);
|
|
1142
|
-
parseRefInput(knowledgeRef);
|
|
1143
|
-
if (knowledgeRef !== op.knowledgeRef)
|
|
773
|
+
const parsedKnowledgeRef = parseRefInput(knowledgeRef);
|
|
774
|
+
if (knowledgeRef !== op.knowledgeRef)
|
|
1144
775
|
warnings.push(`Normalized generated ref "${op.knowledgeRef}" → "${knowledgeRef}"`);
|
|
776
|
+
const pending = listProposals(stashDir, { status: "pending" });
|
|
777
|
+
const wantConcept = conceptIdForRef(knowledgeRef);
|
|
778
|
+
if (wantConcept !== undefined && pending.some((p) => conceptIdForRef(p.ref) === wantConcept)) {
|
|
779
|
+
return skip("promote_pending_proposal_exists", `Skipping promote: pending proposal already exists for ${knowledgeRef}`);
|
|
1145
780
|
}
|
|
1146
|
-
|
|
1147
|
-
|
|
1148
|
-
if (hasPendingProposalForConcept(stashDir, knowledgeRef)) {
|
|
1149
|
-
warnings.push(`Skipping promote: pending proposal already exists for ${knowledgeRef}`);
|
|
1150
|
-
pushSkipReason("promote", op.ref, "promote_pending_proposal_exists");
|
|
1151
|
-
return;
|
|
1152
|
-
}
|
|
1153
|
-
// Idempotency: check if knowledge asset already exists
|
|
1154
|
-
const parsedKnowledgeRef = parseRefInput(knowledgeRef);
|
|
1155
|
-
const destPath = path.join(target.source.path, "knowledge", `${parsedKnowledgeRef.name}.md`);
|
|
1156
|
-
if (fs.existsSync(destPath)) {
|
|
1157
|
-
warnings.push(`Skipping promote: ${knowledgeRef} already exists in source`);
|
|
1158
|
-
pushSkipReason("promote", op.ref, "promote_already_exists");
|
|
1159
|
-
return;
|
|
781
|
+
if (fs.existsSync(path.join(target.source.path, "knowledge", `${parsedKnowledgeRef.name}.md`))) {
|
|
782
|
+
return skip("promote_already_exists", `Skipping promote: ${knowledgeRef} already exists in source`);
|
|
1160
783
|
}
|
|
1161
|
-
let memoryContent
|
|
784
|
+
let memoryContent;
|
|
1162
785
|
try {
|
|
1163
786
|
memoryContent = fs.readFileSync(entry.filePath, "utf8");
|
|
1164
787
|
}
|
|
1165
788
|
catch (e) {
|
|
1166
|
-
|
|
1167
|
-
pushSkipReason("promote", op.ref, "promote_read_failed");
|
|
1168
|
-
return;
|
|
789
|
+
return skip("promote_read_failed", `Promote: could not read ${op.ref}: ${String(e)}`);
|
|
1169
790
|
}
|
|
1170
|
-
|
|
1171
|
-
|
|
1172
|
-
|
|
1173
|
-
warnings.push(`Promote: rejected ${op.ref} — source memory failed sanitization (${promoteSanitized.reason}).`);
|
|
1174
|
-
pushSkipReason("promote", op.ref, "promote_sanitization_failed");
|
|
1175
|
-
return;
|
|
791
|
+
const sanitized = sanitizeMergedContent(memoryContent);
|
|
792
|
+
if (!sanitized.ok) {
|
|
793
|
+
return skip("promote_sanitization_failed", `Promote: rejected ${op.ref} — source memory failed sanitization (${sanitized.reason}).`);
|
|
1176
794
|
}
|
|
1177
|
-
memoryContent =
|
|
1178
|
-
|
|
1179
|
-
|
|
1180
|
-
// (`hasSupersededStatus`) so tests can exercise it directly.
|
|
1181
|
-
if (hasSupersededStatus(promoteSanitized.result.frontmatter)) {
|
|
1182
|
-
warnings.push(`Promote: refused for ${op.ref} → ${knowledgeRef} — source memory has status:superseded; superseded memories are not promotable knowledge.`);
|
|
1183
|
-
pushSkipReason("promote", op.ref, "promote_superseded");
|
|
1184
|
-
return;
|
|
795
|
+
memoryContent = sanitized.result.content;
|
|
796
|
+
if (hasSupersededStatus(sanitized.result.frontmatter)) {
|
|
797
|
+
return skip("promote_superseded", `Promote: refused for ${op.ref} → ${knowledgeRef} — source memory has status:superseded; superseded memories are not promotable knowledge.`);
|
|
1185
798
|
}
|
|
1186
|
-
// Parse the source memory up-front so the body/frontmatter checks below
|
|
1187
|
-
// share the same parsed view.
|
|
1188
799
|
const parsedMemory = parseFrontmatter(memoryContent);
|
|
1189
|
-
// Reject sources whose body is too small to make useful knowledge.
|
|
1190
|
-
// Observed failure: memory files whose body is literally a tags string
|
|
1191
|
-
// ("discord,notification,send-notification") get promoted to knowledge
|
|
1192
|
-
// proposals that no reviewer would accept. Threshold is conservative —
|
|
1193
|
-
// 100 chars catches single-line tag dumps without rejecting genuinely
|
|
1194
|
-
// terse but valid notes.
|
|
1195
|
-
const PROMOTE_BODY_MIN_CHARS = 100;
|
|
1196
800
|
const sourceBody = parsedMemory.content.trim();
|
|
1197
801
|
if (sourceBody.length < PROMOTE_BODY_MIN_CHARS) {
|
|
1198
|
-
|
|
1199
|
-
|
|
1200
|
-
|
|
802
|
+
return skip("promote_source_too_small", `Promote: rejected ${op.ref} → ${knowledgeRef} — source memory body is too small (${sourceBody.length} chars; need ≥${PROMOTE_BODY_MIN_CHARS}) to make useful knowledge.`);
|
|
803
|
+
}
|
|
804
|
+
// The body is the load-bearing content: twins that differ only in
|
|
805
|
+
// bookkeeping frontmatter, or an earlier run's differently-slugged proposal,
|
|
806
|
+
// are the same promotion.
|
|
807
|
+
const bodyHash = contentHash(memoryContent, "body");
|
|
808
|
+
if (ctx.existingKnowledgeBodyHashes.has(bodyHash)) {
|
|
809
|
+
return skip("dedup_existing_knowledge", `Skipping promote: identical body already exists in knowledge; skipping duplicate for ${op.ref} → ${knowledgeRef}`);
|
|
810
|
+
}
|
|
811
|
+
const pendingConsolidate = listProposals(stashDir, { status: "pending" }).filter((p) => p.source === "consolidate");
|
|
812
|
+
const sameBody = pendingConsolidate.find((p) => contentHash(proposalContent(p), "body") === bodyHash);
|
|
813
|
+
if (sameBody) {
|
|
814
|
+
return skip("dedup_pending_proposal", `Skipping promote: identical body already pending as proposal ${sameBody.id} (ref: ${sameBody.ref}); skipping duplicate for ${op.ref} → ${knowledgeRef}`);
|
|
1201
815
|
}
|
|
1202
|
-
// Cross-run + within-run content dedup: if an identical body already
|
|
1203
|
-
// exists in ANY pending consolidate proposal (regardless of target ref),
|
|
1204
|
-
// skip. This prevents duplicate proposals when:
|
|
1205
|
-
// (a) Multiple source memories have identical bodies but differ only
|
|
1206
|
-
// in noise frontmatter (`inferenceProcessed: true` twin alongside
|
|
1207
|
-
// the original; differing `updated:` timestamps; etc.) — the body
|
|
1208
|
-
// is the load-bearing content, so dedup must hash on body only.
|
|
1209
|
-
// (b) A prior run created a proposal for the same body under a
|
|
1210
|
-
// different knowledgeRef slug.
|
|
1211
|
-
// Use cacheHash (case-preserving stripped body) to match the canonical
|
|
1212
|
-
// hash domain used by the body-embedding cache and pending-proposal set.
|
|
1213
|
-
// H1: hash `memoryContent` (one frontmatter strip, inside cacheHash)
|
|
1214
|
-
// rather than the already-stripped `sourceBody` — hashing sourceBody here
|
|
1215
|
-
// double-stripped (parseFrontmatter ran once above to produce sourceBody,
|
|
1216
|
-
// then cacheHash's internal stripFrontmatterBody ran again), diverging from
|
|
1217
|
-
// the single-strip domain `loadExistingKnowledgeBodyHashes` and the
|
|
1218
|
-
// pre-filter use whenever a body starts with its own `---` block.
|
|
1219
|
-
const bodyHash = cacheHash(memoryContent);
|
|
1220
|
-
if (shouldSkipPromotionBodyDuplicate({ bodyHash, op, knowledgeRef, ctx }))
|
|
1221
|
-
return;
|
|
1222
816
|
try {
|
|
1223
|
-
// Use LLM-provided description; fall back to memory's own description
|
|
1224
|
-
// (post-sanitization frontmatter is authoritative).
|
|
1225
817
|
const description = (typeof op.description === "string" && op.description.trim()
|
|
1226
818
|
? op.description.trim()
|
|
1227
819
|
: parsedMemory.data?.description?.trim()) ?? "";
|
|
1228
|
-
// Validate the resolved frontmatter before emitting a proposal.
|
|
1229
|
-
// Required field: non-empty description. Reject obvious truncation
|
|
1230
|
-
// markers (description ends with `,`/`;`/`:`/`...`/hanging connector)
|
|
1231
|
-
// so the queue never sees half-formed metadata that the reviewer
|
|
1232
|
-
// would only reject.
|
|
1233
820
|
const fmCheck = validateProposalFrontmatter({ description });
|
|
1234
821
|
if (!fmCheck.ok) {
|
|
1235
|
-
|
|
1236
|
-
pushSkipReason("promote", op.ref, "promote_invalid_frontmatter");
|
|
1237
|
-
return;
|
|
822
|
+
return skip("promote_invalid_frontmatter", `Promote: rejected ${op.ref} → ${knowledgeRef} — ${fmCheck.reason}.`);
|
|
1238
823
|
}
|
|
1239
|
-
//
|
|
1240
|
-
|
|
1241
|
-
|
|
1242
|
-
// `payload.frontmatter`), and a memory's native frontmatter has
|
|
1243
|
-
// `captureMode`/`beliefState`/etc. but never `description` — without
|
|
1244
|
-
// this merge, 60+ pending proposals were blocked at accept-time with
|
|
1245
|
-
// MISSING_FRONTMATTER_DESCRIPTION even though the envelope had it.
|
|
1246
|
-
// (The body-frontmatter assumption baked into the 2026-05-20 comment
|
|
1247
|
-
// below was wrong: body fm and envelope fm only converge when the
|
|
1248
|
-
// writer explicitly merges them, which it now does.)
|
|
1249
|
-
const mergedBodyFm = {
|
|
824
|
+
// The description goes into the body frontmatter, which accept-time validation reads.
|
|
825
|
+
const xrefs = Array.isArray(parsedMemory.data?.xrefs) ? parsedMemory.data.xrefs.map(String) : [];
|
|
826
|
+
const mergedFrontmatter = {
|
|
1250
827
|
...(parsedMemory.data ?? {}),
|
|
1251
828
|
description,
|
|
1252
|
-
xrefs:
|
|
829
|
+
xrefs: [...new Set([...xrefs, op.ref].map(canonicalXref))],
|
|
1253
830
|
};
|
|
1254
|
-
const
|
|
1255
|
-
const
|
|
1256
|
-
|
|
1257
|
-
|
|
1258
|
-
// dedup inside `mergePlans` handles duplicates against existing
|
|
1259
|
-
// stash assets — see commit history for the deletion of the
|
|
1260
|
-
// unbounded embedding + cross-type slug branches.
|
|
1261
|
-
const dedup = await checkPreEmitDedup({
|
|
1262
|
-
candidateRef: knowledgeRef,
|
|
1263
|
-
candidateText: `${description}. ${memoryContent}`,
|
|
1264
|
-
stashDir,
|
|
1265
|
-
config,
|
|
1266
|
-
});
|
|
1267
|
-
if (dedup.duplicate) {
|
|
1268
|
-
warnings.push(`Promote: skipped ${op.ref} → ${knowledgeRef} — ${dedup.reason}.`);
|
|
1269
|
-
pushSkipReason("promote", op.ref, "promote_dedup_window");
|
|
1270
|
-
return;
|
|
831
|
+
const normalized = normalizeSlugForDedup(knowledgeRef);
|
|
832
|
+
const variant = pendingConsolidate.find((p) => normalizeSlugForDedup(p.ref) === normalized);
|
|
833
|
+
if (variant) {
|
|
834
|
+
return skip("promote_dedup_window", `Promote: skipped ${op.ref} → ${knowledgeRef} — slug-variant of pending proposal ${variant.id} (${variant.ref}).`);
|
|
1271
835
|
}
|
|
1272
|
-
const
|
|
836
|
+
const proposal = mintProposal(stashDir, ctx.proposalsCtx, {
|
|
1273
837
|
ref: knowledgeRef,
|
|
1274
838
|
target: { source: target.source.name, root: target.source.path },
|
|
1275
839
|
source: "consolidate",
|
|
1276
|
-
sourceRun,
|
|
1277
|
-
// §23.6 fingerprint model-id term (WI-6.4).
|
|
1278
|
-
...(ctx.llmRunner?.connection.model ? { modelId: ctx.llmRunner.connection.model } : {}),
|
|
840
|
+
sourceRun: ctx.sourceRun,
|
|
1279
841
|
payload: {
|
|
1280
|
-
content:
|
|
842
|
+
content: assembleAssetFromString(serializeFrontmatter(mergedFrontmatter), parsedMemory.content),
|
|
1281
843
|
frontmatter: { description, xrefs: [canonicalXref(op.ref)] },
|
|
1282
844
|
},
|
|
1283
845
|
...(typeof op.confidence === "number" ? { confidence: op.confidence } : {}),
|
|
846
|
+
// The ledger keys the attempt by the source memory.
|
|
847
|
+
attemptedRefs: [op.ref],
|
|
1284
848
|
});
|
|
1285
|
-
|
|
1286
|
-
|
|
1287
|
-
pushSkipReason("promote", op.ref, `promote_proposal_${proposalResult.reason}`);
|
|
1288
|
-
}
|
|
1289
|
-
else {
|
|
1290
|
-
promoted.push(proposalResult.id);
|
|
1291
|
-
promotedSourceRefs.add(op.ref);
|
|
1292
|
-
}
|
|
849
|
+
ctx.promoted.push(proposal.id);
|
|
850
|
+
ctx.promotedSourceRefs.add(op.ref);
|
|
1293
851
|
}
|
|
1294
852
|
catch (e) {
|
|
1295
853
|
ctx.promotionFailures.count++;
|
|
1296
|
-
|
|
1297
|
-
pushSkipReason("promote", op.ref, "promote_create_failed");
|
|
854
|
+
skip("promote_create_failed", `Promote: createProposal failed for ${op.ref}: ${String(e)}`);
|
|
1298
855
|
}
|
|
1299
856
|
}
|
|
1300
|
-
// ── Helpers ─────────────────────────────────────────────────────────────────
|
|
1301
857
|
/**
|
|
1302
|
-
*
|
|
1303
|
-
*
|
|
1304
|
-
* - numeric counter suffixes (`-2`, `-3`)
|
|
1305
|
-
* - trailing -patterns / -2026-05-03 styles
|
|
1306
|
-
* - word reorderings via alphabetical sort of the remaining tokens.
|
|
1307
|
-
*
|
|
1308
|
-
* Two slugs that normalise to the same string are considered the same asset
|
|
1309
|
-
* for dedup purposes even if they don't share an exact ref.
|
|
1310
|
-
*/
|
|
1311
|
-
/** The conceptId a proposal ref maps to, or undefined for an invalid ref. */
|
|
1312
|
-
function conceptIdForRef(ref) {
|
|
1313
|
-
try {
|
|
1314
|
-
const p = parseRefInput(ref);
|
|
1315
|
-
return conceptIdFromTypeName(p.type, p.name);
|
|
1316
|
-
}
|
|
1317
|
-
catch {
|
|
1318
|
-
return undefined;
|
|
1319
|
-
}
|
|
1320
|
-
}
|
|
1321
|
-
/** Is a pending proposal already queued for `conceptRef`'s concept? */
|
|
1322
|
-
function hasPendingProposalForConcept(stashDir, conceptRef) {
|
|
1323
|
-
const want = conceptIdForRef(conceptRef);
|
|
1324
|
-
return (want !== undefined && listProposals(stashDir, { status: "pending" }).some((p) => conceptIdForRef(p.ref) === want));
|
|
1325
|
-
}
|
|
1326
|
-
function normalizeSlugForDedup(ref) {
|
|
1327
|
-
const slug = parseRefInput(ref).name;
|
|
1328
|
-
const monthRe = /(?:jan|feb|mar|apr|may|jun|jul|aug|sep|oct|nov|dec)/i;
|
|
1329
|
-
const tokens = slug
|
|
1330
|
-
.toLowerCase()
|
|
1331
|
-
.split("-")
|
|
1332
|
-
.filter((tok) => tok.length > 0)
|
|
1333
|
-
// Strip purely-numeric tokens (years, dates, counter suffixes like -2 / -3).
|
|
1334
|
-
// Numbers carry no semantic information for our dedup purposes — every
|
|
1335
|
-
// observed defective slug variant differs only in dates or counters.
|
|
1336
|
-
.filter((tok) => !/^\d+$/.test(tok))
|
|
1337
|
-
.filter((tok) => !monthRe.test(tok));
|
|
1338
|
-
// Sort to absorb word reorderings.
|
|
1339
|
-
tokens.sort();
|
|
1340
|
-
return tokens.join("-");
|
|
1341
|
-
}
|
|
1342
|
-
/**
|
|
1343
|
-
* Pre-emit dedup check: compare the candidate ref against pending consolidate
|
|
1344
|
-
* proposals only. Returns a reason string if a slug-variant match is found,
|
|
1345
|
-
* else null.
|
|
1346
|
-
*
|
|
1347
|
-
* Historical context (REMOVED 2026-05-20): this function previously also ran
|
|
1348
|
-
* (a) a normalised-slug match against existing knowledge AND memory entries
|
|
1349
|
-
* in the DB, and
|
|
1350
|
-
* (b) an embedding cosine-similarity check (>= 0.85) against ALL knowledge
|
|
1351
|
-
* and non-derived memory entries.
|
|
1352
|
-
* Both branches had ZERO observed fires across 30 sampled runs in the
|
|
1353
|
-
* post-fix window. The 29 actual dedup catches all came from the SEPARATE
|
|
1354
|
-
* content-hash dedup inside `mergePlans` (the older SHA-256 helper). The
|
|
1355
|
-
* embedding branch in particular had unbounded cost per promote (embedded
|
|
1356
|
-
* every knowledge + non-derived memory entry, every time) with no observed
|
|
1357
|
-
* benefit. Empirical signal → deleted.
|
|
1358
|
-
*
|
|
1359
|
-
* What remains: a check against pending consolidate proposals in the SAME
|
|
1360
|
-
* improve run. This catches duplicates queued back-to-back within a single
|
|
1361
|
-
* improve invocation — a different concern from the cross-run content-hash
|
|
1362
|
-
* dedup, and cheap (no embeddings, no DB query).
|
|
1363
|
-
*/
|
|
1364
|
-
async function checkPreEmitDedup(opts) {
|
|
1365
|
-
const normCandidate = normalizeSlugForDedup(opts.candidateRef);
|
|
1366
|
-
// Pending consolidate proposals (slug match) — within the same improve run.
|
|
1367
|
-
const pendingConsolidate = listProposals(opts.stashDir, { status: "pending" }).filter((p) => p.source === "consolidate");
|
|
1368
|
-
for (const p of pendingConsolidate) {
|
|
1369
|
-
if (normalizeSlugForDedup(p.ref) === normCandidate) {
|
|
1370
|
-
return { duplicate: true, reason: `slug-variant of pending proposal ${p.id} (${p.ref})` };
|
|
1371
|
-
}
|
|
1372
|
-
}
|
|
1373
|
-
return { duplicate: false };
|
|
1374
|
-
}
|
|
1375
|
-
/**
|
|
1376
|
-
* Incremental candidate set: {changed} ∪ {top-k persisted-vector neighbours of
|
|
1377
|
-
* each changed memory}, intersected with the loaded pool. Returns [] when
|
|
1378
|
-
* nothing changed (caller emits a no-op envelope), the full pool when
|
|
1379
|
-
* everything changed or the index can't answer (fail-open to preserve merge
|
|
1380
|
-
* correctness). `since` is an ISO timestamp.
|
|
858
|
+
* {changed} ∪ {top-k indexed neighbours of each changed memory}, within the
|
|
859
|
+
* pool: nothing changed → []; everything changed or no index → the full pool.
|
|
1381
860
|
*/
|
|
1382
861
|
export function narrowToIncrementalCandidates(memories, since, warnings, neighborsPerChanged = 5, readOnly = false) {
|
|
1383
|
-
// Lenient
|
|
1384
|
-
// string comparison below then selects nothing (see core/time.ts doc).
|
|
862
|
+
// Lenient: a garbage `since` passes through and selects nothing.
|
|
1385
863
|
const sinceIso = parseSinceToIsoLenient(since);
|
|
1386
|
-
const
|
|
864
|
+
const changed = memories.filter((m) => {
|
|
1387
865
|
try {
|
|
1388
866
|
return fs.statSync(m.filePath).mtime.toISOString() > sinceIso;
|
|
1389
867
|
}
|
|
1390
868
|
catch {
|
|
1391
869
|
return true; // never silently drop a memory we cannot stat
|
|
1392
870
|
}
|
|
1393
|
-
};
|
|
1394
|
-
const changed = memories.filter(isChanged);
|
|
871
|
+
});
|
|
1395
872
|
if (changed.length === 0)
|
|
1396
873
|
return [];
|
|
1397
874
|
if (changed.length === memories.length)
|
|
1398
875
|
return memories;
|
|
1399
|
-
const
|
|
876
|
+
const inPool = new Set(memories.map((m) => m.name));
|
|
1400
877
|
const keep = new Set(changed.map((m) => m.name));
|
|
1401
878
|
let db;
|
|
1402
879
|
try {
|
|
@@ -1410,12 +887,9 @@ export function narrowToIncrementalCandidates(memories, since, warnings, neighbo
|
|
|
1410
887
|
for (const hit of getNeighborsByEntryId(db, id, neighborsPerChanged + 1)) {
|
|
1411
888
|
if (hit.id === id)
|
|
1412
889
|
continue;
|
|
1413
|
-
const
|
|
1414
|
-
if (
|
|
1415
|
-
|
|
1416
|
-
const name = entry.entry.name;
|
|
1417
|
-
if (byName.has(name))
|
|
1418
|
-
keep.add(name); // only neighbours present in the loaded pool
|
|
890
|
+
const name = getEntryById(db, hit.id)?.entry.name;
|
|
891
|
+
if (name && inPool.has(name))
|
|
892
|
+
keep.add(name);
|
|
1419
893
|
}
|
|
1420
894
|
}
|
|
1421
895
|
}
|
|
@@ -1431,24 +905,17 @@ export function narrowToIncrementalCandidates(memories, since, warnings, neighbo
|
|
|
1431
905
|
warnings.push(`Incremental consolidation: ${changed.length} changed + neighbours → ${candidates.length}/${memories.length} memories considered (since ${since}${sinceIso !== since ? ` = ${sinceIso}` : ""}).`);
|
|
1432
906
|
return candidates;
|
|
1433
907
|
}
|
|
908
|
+
/** The target bundle's eligible memories from the index, else walked from disk. */
|
|
1434
909
|
function loadMemoriesForSource(source, warnings, readOnly) {
|
|
1435
|
-
// Load from DB first
|
|
1436
910
|
let memories = [];
|
|
1437
911
|
let db;
|
|
1438
912
|
try {
|
|
1439
913
|
db = readOnly ? openReadonlyExistingDatabase(undefined, { isolatedSnapshot: true }) : openExistingDatabase();
|
|
1440
914
|
if (!db)
|
|
1441
915
|
throw new Error("index unavailable");
|
|
1442
|
-
|
|
1443
|
-
|
|
1444
|
-
.filter((
|
|
1445
|
-
.filter((e) => isConsolidationEligibleMemoryName(e.entry.name))
|
|
1446
|
-
// Skip stale DB entries whose file was deleted by a prior run but not yet
|
|
1447
|
-
// re-indexed. Without this guard the deleted file's ref appears in chunks
|
|
1448
|
-
// sent to the LLM, which then proposes a second delete → delete_failed
|
|
1449
|
-
// because the file is already gone. Re-indexing runs on a cron cadence so
|
|
1450
|
-
// several successful deletes can accumulate before the DB catches up.
|
|
1451
|
-
.filter((e) => fs.existsSync(e.filePath))
|
|
916
|
+
memories = getAllEntries(db, "memory")
|
|
917
|
+
.filter((e) => source !== undefined && e.bundleId === source.bundleId)
|
|
918
|
+
.filter((e) => isConsolidationEligibleMemoryName(e.entry.name) && fs.existsSync(e.filePath))
|
|
1452
919
|
.map((e) => ({
|
|
1453
920
|
name: e.entry.name,
|
|
1454
921
|
filePath: e.filePath,
|
|
@@ -1464,34 +931,30 @@ function loadMemoriesForSource(source, warnings, readOnly) {
|
|
|
1464
931
|
if (db)
|
|
1465
932
|
closeDatabase(db);
|
|
1466
933
|
}
|
|
1467
|
-
if (memories.length
|
|
1468
|
-
|
|
1469
|
-
|
|
1470
|
-
|
|
1471
|
-
|
|
1472
|
-
|
|
1473
|
-
|
|
1474
|
-
|
|
1475
|
-
|
|
1476
|
-
|
|
1477
|
-
if (
|
|
1478
|
-
if (source.excludedSourceRoots.has(path.resolve(filePath)))
|
|
1479
|
-
continue;
|
|
934
|
+
if (memories.length > 0 || !source)
|
|
935
|
+
return memories;
|
|
936
|
+
const memoriesDir = path.join(source.sourceRoot, "memories");
|
|
937
|
+
if (fs.existsSync(memoriesDir)) {
|
|
938
|
+
const pending = [memoriesDir];
|
|
939
|
+
while (pending.length > 0) {
|
|
940
|
+
const current = pending.pop();
|
|
941
|
+
for (const entry of fs.readdirSync(current, { withFileTypes: true })) {
|
|
942
|
+
const filePath = path.join(current, entry.name);
|
|
943
|
+
if (entry.isDirectory()) {
|
|
944
|
+
if (!source.excludedSourceRoots.has(path.resolve(filePath)))
|
|
1480
945
|
pending.push(filePath);
|
|
1481
|
-
|
|
1482
|
-
|
|
1483
|
-
|
|
1484
|
-
|
|
1485
|
-
|
|
1486
|
-
|
|
1487
|
-
|
|
1488
|
-
memories.push({ name, filePath, description: "", tags: [], stashDir: fsStashDir });
|
|
946
|
+
continue;
|
|
947
|
+
}
|
|
948
|
+
if (!entry.isFile() || !entry.name.endsWith(".md"))
|
|
949
|
+
continue;
|
|
950
|
+
const name = path.relative(memoriesDir, filePath).replace(/\.md$/, "").split(path.sep).join("/");
|
|
951
|
+
if (isConsolidationEligibleMemoryName(name)) {
|
|
952
|
+
memories.push({ name, filePath, description: "", tags: [], stashDir: source.sourceRoot });
|
|
1489
953
|
}
|
|
1490
954
|
}
|
|
1491
955
|
}
|
|
1492
|
-
if (memories.length > 0) {
|
|
1493
|
-
warnings.push("DB not found or empty — loaded memories directly from filesystem.");
|
|
1494
|
-
}
|
|
1495
956
|
}
|
|
957
|
+
if (memories.length > 0)
|
|
958
|
+
warnings.push("DB not found or empty — loaded memories directly from filesystem.");
|
|
1496
959
|
return memories;
|
|
1497
960
|
}
|