akm-cli 0.9.0-rc.0 → 0.9.0-rc.13
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +1283 -22
- package/README.md +62 -37
- package/SECURITY.md +46 -31
- package/dist/akm +162 -38
- package/dist/akm-migrate +44 -0
- package/dist/assets/backends/schtasks-template.xml +2 -1
- package/dist/assets/hints/cli-hints-full.md +268 -118
- package/dist/assets/hints/cli-hints-short.md +87 -24
- package/dist/assets/{profiles → improve-strategies}/catchup.json +3 -1
- package/dist/assets/{profiles → improve-strategies}/consolidate.json +3 -1
- package/dist/assets/{profiles → improve-strategies}/default.json +6 -7
- package/dist/assets/improve-strategies/frequent.json +15 -0
- package/dist/assets/{profiles → improve-strategies}/graph-refresh.json +4 -2
- package/dist/assets/{profiles → improve-strategies}/memory-focus.json +4 -1
- package/dist/assets/{profiles → improve-strategies}/proactive-maintenance.json +5 -5
- package/dist/assets/{profiles → improve-strategies}/quick.json +4 -2
- package/dist/assets/improve-strategies/reflect-distill.json +30 -0
- package/dist/assets/{profiles → improve-strategies}/thorough.json +1 -1
- package/dist/assets/prompts/consolidate-system.md +5 -5
- package/dist/assets/prompts/extract-session.md +2 -6
- package/dist/assets/prompts/memory-infer-user.md +2 -3
- package/dist/assets/prompts/reflect-llm-framed-contract.md +11 -0
- package/dist/assets/prompts/reflect-llm-schema-contract.md +3 -0
- package/dist/assets/prompts/reflect-output-repair.md +3 -0
- package/dist/assets/prompts/workflow-unit-preamble.md +26 -0
- package/dist/assets/stash-skeleton/README.md +38 -10
- package/dist/assets/stash-skeleton/facts/conventions/assets/agent.md +8 -0
- package/dist/assets/stash-skeleton/facts/conventions/assets/command.md +8 -0
- package/dist/assets/stash-skeleton/facts/conventions/assets/fact.md +14 -1
- package/dist/assets/stash-skeleton/facts/conventions/assets/knowledge.md +13 -1
- package/dist/assets/stash-skeleton/facts/conventions/assets/lesson.md +9 -1
- package/dist/assets/stash-skeleton/facts/conventions/assets/memory.md +11 -0
- package/dist/assets/stash-skeleton/facts/conventions/assets/script.md +9 -0
- package/dist/assets/stash-skeleton/facts/conventions/assets/skill.md +9 -0
- package/dist/assets/stash-skeleton/facts/conventions/assets/workflow.md +8 -0
- package/dist/assets/stash-skeleton/facts/conventions/backlinks.md +100 -0
- package/dist/assets/stash-skeleton/facts/conventions/domains.md +64 -0
- package/dist/assets/stash-skeleton/facts/conventions/organization.md +136 -0
- package/dist/assets/tasks/core/extract.yml +3 -2
- package/dist/assets/tasks/core/improve.yml +2 -1
- package/dist/assets/tasks/core/index-refresh.yml +1 -0
- package/dist/assets/tasks/core/sync.yml +1 -0
- package/dist/assets/tasks/core/version-check.yml +2 -1
- package/dist/assets/tasks/improve/akm-graph-refresh-weekly.yml +5 -0
- package/dist/assets/tasks/improve/akm-improve-catchup.yml +8 -0
- package/dist/assets/tasks/improve/akm-improve-consolidate.yml +5 -0
- package/dist/assets/tasks/improve/akm-improve-frequent.yml +5 -0
- package/dist/assets/tasks/improve/akm-improve-nightly.yml +5 -0
- package/dist/assets/templates/html/health.html +5 -4
- package/dist/assets/workflows/workflow-template.md +31 -15
- package/dist/cli/invocation.js +279 -0
- package/dist/cli/parse-args.js +5 -90
- package/dist/cli/retired-commands.js +78 -0
- package/dist/cli/shared.js +158 -48
- package/dist/cli-node.mjs +2 -1
- package/dist/cli.js +747 -293
- package/dist/commands/agent/agent-dispatch.js +19 -18
- package/dist/commands/agent/agent-support.js +0 -24
- package/dist/commands/agent/contribute-cli.js +43 -97
- package/dist/commands/completions.js +80 -23
- package/dist/commands/config-cli.js +44 -281
- package/dist/commands/env/env-binding.js +99 -0
- package/dist/commands/env/env-cli.js +84 -224
- package/dist/commands/env/env.js +12 -163
- package/dist/commands/env/marker-path.js +6 -0
- package/dist/commands/env/secret-cli.js +45 -61
- package/dist/commands/env/secret.js +32 -62
- package/dist/commands/feedback-cli.js +179 -85
- package/dist/commands/health/accept-rate.js +58 -0
- package/dist/commands/health/advisories.js +7 -8
- package/dist/commands/health/checks.js +279 -94
- package/dist/commands/health/html-report.js +197 -578
- package/dist/commands/health/improve-metrics.js +277 -246
- package/dist/commands/health/llm-usage.js +19 -19
- package/dist/commands/health/md-report.js +16 -7
- package/dist/commands/health/metrics.js +67 -32
- package/dist/commands/health/renderers.js +47 -0
- package/dist/commands/health/report-view-model.js +508 -0
- package/dist/commands/health/stash-exposure.js +1 -1
- package/dist/commands/health/surfaces.js +16 -56
- package/dist/commands/health/task-runs.js +3 -67
- package/dist/{migrate-storage-node.mjs → commands/health/types-checks.js} +1 -5
- package/dist/commands/health/types-improve.js +29 -0
- package/dist/{output/text/save.js → commands/health/types-metrics.js} +1 -2
- package/dist/commands/health/types-result.js +7 -0
- package/dist/commands/health/types-runs.js +4 -0
- package/dist/commands/health/types-session-log.js +4 -0
- package/dist/commands/health/types-windows.js +4 -0
- package/dist/commands/health/types.js +26 -21
- package/dist/commands/health/windows.js +2 -3
- package/dist/commands/health.js +296 -167
- package/dist/commands/improve/anti-collapse.js +5 -5
- package/dist/commands/improve/autonomy-gate.js +68 -0
- package/dist/commands/improve/collapse-detector.js +65 -52
- package/dist/commands/improve/consolidate/chunking.js +9 -7
- package/dist/commands/improve/consolidate/eligibility.js +1 -23
- package/dist/commands/improve/consolidate/merge.js +4 -0
- package/dist/commands/improve/consolidate.js +454 -1354
- package/dist/commands/improve/content-hash.js +39 -0
- package/dist/commands/improve/distill/content-repair.js +4 -10
- package/dist/commands/improve/distill/promote-memory.js +89 -64
- package/dist/commands/improve/distill/quality-gate.js +118 -42
- package/dist/commands/improve/distill-guards.js +1 -1
- package/dist/commands/improve/distill-promotion-policy.js +33 -888
- package/dist/commands/improve/distill.js +607 -363
- package/dist/commands/improve/eligibility.js +165 -79
- package/dist/commands/improve/extract-cli.js +35 -126
- package/dist/commands/improve/extract-prompt.js +6 -35
- package/dist/commands/improve/extract.js +640 -391
- package/dist/commands/improve/feedback-valence.js +2 -12
- package/dist/commands/improve/improve-cli.js +134 -135
- package/dist/commands/improve/improve-result-file.js +30 -50
- package/dist/commands/improve/improve-run-types.js +4 -0
- package/dist/commands/improve/improve-strategies.js +135 -0
- package/dist/commands/improve/improve.js +904 -701
- package/dist/commands/improve/locks.js +64 -111
- package/dist/commands/improve/loop-stages.js +1110 -923
- package/dist/commands/improve/memory/derived-ref.js +124 -0
- package/dist/commands/improve/memory/memory-belief.js +79 -7
- package/dist/commands/improve/memory/memory-contradiction-detect.js +49 -52
- package/dist/commands/improve/memory/memory-improve.js +25 -37
- package/dist/commands/improve/outcome-loop.js +25 -88
- package/dist/commands/improve/preparation.js +1034 -813
- package/dist/commands/improve/proactive-maintenance.js +34 -9
- package/dist/commands/improve/proposal-envelope.js +31 -0
- package/dist/commands/improve/reflect.js +983 -794
- package/dist/commands/improve/run-context.js +119 -0
- package/dist/commands/improve/salience.js +24 -127
- package/dist/commands/improve/session-asset.js +7 -3
- package/dist/commands/improve/shared.js +14 -34
- package/dist/commands/improve/source-identity.js +28 -0
- package/dist/commands/improve/triage.js +20 -17
- package/dist/commands/lint/base-linter.js +340 -313
- package/dist/commands/lint/env-key-rules.js +31 -47
- package/dist/commands/lint/index.js +185 -30
- package/dist/commands/{events.js → log.js} +28 -38
- package/dist/commands/migrate-cli.js +54 -0
- package/dist/commands/migration-tool.js +55 -0
- package/dist/commands/observability-cli.js +70 -208
- package/dist/commands/proposal/diff-format.js +50 -0
- package/dist/commands/proposal/drain-policies.js +0 -6
- package/dist/commands/proposal/drain.js +91 -40
- package/dist/commands/proposal/proposal-cli.js +134 -132
- package/dist/commands/proposal/proposal-types.js +56 -0
- package/dist/commands/proposal/proposal.js +83 -65
- package/dist/commands/proposal/propose-cli.js +88 -0
- package/dist/commands/proposal/propose.js +105 -88
- package/dist/commands/proposal/repository.js +1303 -278
- package/dist/commands/proposal/validators/proposal-quality-validators.js +16 -6
- package/dist/commands/proposal/validators/proposal-validators.js +61 -12
- package/dist/commands/proposal/validators/proposals.js +6 -8
- package/dist/commands/read/curate.js +78 -73
- package/dist/commands/read/knowledge.js +510 -13
- package/dist/commands/read/registry-search.js +2 -2
- package/dist/commands/read/remember-cli.js +84 -15
- package/dist/commands/read/search-cli.js +203 -96
- package/dist/commands/read/search.js +126 -94
- package/dist/commands/read/show.js +226 -250
- package/dist/commands/registry-cli.js +34 -60
- package/dist/commands/remember.js +18 -57
- package/dist/commands/sources/add-cli.js +104 -49
- package/dist/commands/sources/bundle-cli.js +166 -0
- package/dist/commands/sources/bundle-config-ops.js +63 -0
- package/dist/commands/sources/info.js +27 -15
- package/dist/commands/sources/init.js +30 -40
- package/dist/commands/sources/installed-stashes.js +469 -172
- package/dist/commands/sources/migration-help.js +7 -4
- package/dist/commands/sources/schema-repair.js +10 -9
- package/dist/commands/sources/self-update.js +182 -121
- package/dist/commands/sources/source-add.js +169 -178
- package/dist/commands/sources/source-clone.js +144 -41
- package/dist/commands/sources/source-manage.js +94 -59
- package/dist/commands/sources/sources-cli.js +64 -205
- package/dist/commands/sources/stash-cli.js +91 -54
- package/dist/commands/sources/stash-skeleton.js +1 -1
- package/dist/commands/tasks/tasks-cli.js +106 -104
- package/dist/commands/tasks/tasks.js +445 -262
- package/dist/commands/workflow-cli.js +232 -121
- package/dist/core/action-contributors.js +1 -1
- package/dist/core/activation-policy.js +49 -0
- package/dist/core/adapter/adapters/agent-skills-adapter.js +181 -0
- package/dist/core/adapter/adapters/akm-adapter.js +528 -0
- package/dist/core/adapter/adapters/akm-lint.js +392 -0
- package/dist/core/adapter/adapters/akm-metadata.js +387 -0
- package/dist/core/adapter/adapters/akm-task-adapter.js +149 -0
- package/dist/core/adapter/adapters/akm-workflow-adapter.js +180 -0
- package/dist/core/adapter/adapters/claude-adapter.js +61 -0
- package/dist/core/adapter/adapters/dotenv-adapter.js +187 -0
- package/dist/core/adapter/adapters/generic-files-adapter.js +119 -0
- package/dist/core/adapter/adapters/index.js +80 -0
- package/dist/core/adapter/adapters/llm-wiki-adapter.js +419 -0
- package/dist/core/adapter/adapters/okf-adapter.js +391 -0
- package/dist/core/adapter/adapters/opencode-adapter.js +68 -0
- package/dist/core/adapter/adapters/shared.js +286 -0
- package/dist/core/adapter/adapters/tool-dir-shared.js +217 -0
- package/dist/core/adapter/adapters/website-snapshot-adapter.js +155 -0
- package/dist/core/adapter/bundle-adapter.js +4 -0
- package/dist/core/adapter/detect-adapter.js +17 -0
- package/dist/core/adapter/recognize-match.js +44 -0
- package/dist/core/adapter/registry.js +56 -0
- package/dist/core/adapter/types.js +4 -0
- package/dist/core/asset/akm-markdown.js +30 -0
- package/dist/core/asset/asset-placement.js +243 -0
- package/dist/core/asset/asset-ref.js +110 -79
- package/dist/core/asset/asset-serialize.js +20 -0
- package/dist/core/asset/frontmatter.js +28 -12
- package/dist/core/asset/markdown.js +40 -51
- package/dist/core/asset/resolve-ref.js +274 -0
- package/dist/core/asset/stash-meta.js +2 -2
- package/dist/core/bundle-id.js +51 -0
- package/dist/core/common.js +281 -86
- package/dist/core/config/config-io.js +42 -128
- package/dist/core/config/config-schema.js +233 -834
- package/dist/core/config/config-sources.js +162 -39
- package/dist/core/config/config-types.js +16 -11
- package/dist/core/config/config-version.js +29 -0
- package/dist/core/config/config-walker.js +126 -37
- package/dist/core/config/config.js +154 -331
- package/dist/core/config/deep-merge.js +41 -0
- package/dist/core/config/engine-semantics.js +28 -0
- package/dist/core/config/experimental.js +21 -0
- package/dist/core/config/schema/embedding.js +38 -0
- package/dist/core/config/schema/engines.js +116 -0
- package/dist/core/config/schema/experimental.js +47 -0
- package/dist/core/config/schema/feedback.js +31 -0
- package/dist/core/config/schema/improve-processes.js +389 -0
- package/dist/core/config/schema/improve.js +94 -0
- package/dist/core/config/schema/index-config.js +176 -0
- package/dist/core/config/schema/output.js +18 -0
- package/dist/core/config/schema/primitives.js +94 -0
- package/dist/core/config/schema/search.js +30 -0
- package/dist/core/config/schema/setup.js +18 -0
- package/dist/core/config/schema/sources-bundles.js +169 -0
- package/dist/core/config/schema/workflow.js +29 -0
- package/dist/core/env-secret-ref.js +155 -20
- package/dist/core/errors.js +17 -15
- package/dist/core/events-types.js +4 -0
- package/dist/core/events.js +46 -128
- package/dist/core/extra-params.js +62 -0
- package/dist/core/file-change.js +17 -0
- package/dist/core/file-lock.js +202 -57
- package/dist/core/fs-txn.js +392 -0
- package/dist/core/git-message.js +59 -0
- package/dist/core/improve-result.js +167 -0
- package/dist/core/json-schema.js +142 -0
- package/dist/core/lesson-lint.js +1 -17
- package/dist/core/logs-db.js +1 -1
- package/dist/core/maintenance-barrier.js +135 -0
- package/dist/core/migration-operation.js +44 -0
- package/dist/core/mutation-target.js +78 -0
- package/dist/core/paths.js +22 -25
- package/dist/core/platform.js +10 -0
- package/dist/core/recognition-util.js +128 -0
- package/dist/core/redaction.js +392 -0
- package/dist/core/standards/resolve-standards-context.js +36 -65
- package/dist/core/standards/resolve-stash-standards.js +2 -2
- package/dist/core/standards/resolve-type-conventions.js +5 -5
- package/dist/core/state/migrations.js +242 -11
- package/dist/core/state-db.js +98 -10
- package/dist/core/structured.js +1 -1
- package/dist/core/subprocess.js +303 -0
- package/dist/core/text-truncation.js +9 -5
- package/dist/core/time.js +20 -0
- package/dist/core/type-presentation.js +130 -0
- package/dist/core/warn.js +0 -3
- package/dist/core/write-source.js +834 -118
- package/dist/indexer/bundle-identity-guard.js +92 -0
- package/dist/indexer/db/graph-db.js +1 -25
- package/dist/indexer/db/llm-cache.js +1 -1
- package/dist/indexer/ensure-index.js +30 -9
- package/dist/indexer/graph/graph-boost.js +9 -30
- package/dist/indexer/graph/graph-extraction.js +41 -27
- package/dist/indexer/graph/graph-types.js +4 -0
- package/dist/indexer/index-writer-lock.js +93 -49
- package/dist/indexer/index-written-assets.js +100 -53
- package/dist/indexer/indexer.js +746 -329
- package/dist/indexer/init.js +18 -25
- package/dist/indexer/installations.js +142 -0
- package/dist/indexer/passes/dir-staleness.js +18 -10
- package/dist/indexer/passes/memory-inference.js +25 -15
- package/dist/indexer/passes/metadata.js +412 -243
- package/dist/indexer/scan/doc-to-entry.js +160 -0
- package/dist/indexer/scan/drain-dir.js +134 -0
- package/dist/indexer/search/db-search.js +292 -108
- package/dist/indexer/search/fts-query.js +64 -0
- package/dist/indexer/search/ranking-contributors.js +145 -25
- package/dist/indexer/search/ranking-types.js +4 -0
- package/dist/indexer/search/ranking.js +28 -71
- package/dist/indexer/search/search-attribution.js +67 -0
- package/dist/indexer/search/search-fields.js +18 -3
- package/dist/indexer/search/search-hit-enrichers.js +30 -40
- package/dist/indexer/search/search-source.js +157 -111
- package/dist/indexer/search/semantic-status.js +4 -1
- package/dist/indexer/usage/usage-events.js +10 -30
- package/dist/indexer/walk/file-context.js +3 -45
- package/dist/indexer/walk/matchers.js +42 -34
- package/dist/indexer/walk/path-resolver.js +11 -5
- package/dist/indexer/walk/walker.js +42 -14
- package/dist/integrations/agent/builder-shared.js +7 -0
- package/dist/integrations/agent/builders.js +5 -56
- package/dist/integrations/agent/config.js +3 -143
- package/dist/integrations/agent/detect.js +17 -2
- package/dist/integrations/agent/engine-resolution.js +231 -0
- package/dist/integrations/agent/index.js +1 -2
- package/dist/integrations/agent/model-aliases.js +16 -2
- package/dist/integrations/agent/profiles.js +36 -62
- package/dist/integrations/agent/prompts.js +46 -18
- package/dist/integrations/agent/runner-dispatch.js +93 -4
- package/dist/integrations/agent/runner.js +76 -208
- package/dist/integrations/agent/spawn.js +88 -196
- package/dist/integrations/harnesses/aider/agent-builder.js +114 -0
- package/dist/integrations/harnesses/aider/index.js +48 -0
- package/dist/integrations/harnesses/aider/result-extractor.js +53 -0
- package/dist/integrations/harnesses/amazonq/agent-builder.js +147 -0
- package/dist/integrations/harnesses/amazonq/index.js +45 -0
- package/dist/integrations/harnesses/amazonq/result-extractor.js +48 -0
- package/dist/integrations/harnesses/claude/agent-builder.js +46 -8
- package/dist/integrations/harnesses/claude/config-import.js +1 -3
- package/dist/integrations/harnesses/claude/index.js +24 -35
- package/dist/integrations/harnesses/claude/result-extractor.js +52 -0
- package/dist/integrations/harnesses/claude/session-log.js +27 -75
- package/dist/integrations/harnesses/codex/agent-builder.js +138 -0
- package/dist/integrations/harnesses/codex/index.js +52 -0
- package/dist/integrations/harnesses/codex/result-extractor.js +73 -0
- package/dist/integrations/harnesses/copilot/agent-builder.js +122 -0
- package/dist/integrations/harnesses/copilot/index.js +48 -0
- package/dist/integrations/harnesses/copilot/result-extractor.js +151 -0
- package/dist/integrations/harnesses/gemini/agent-builder.js +120 -0
- package/dist/integrations/harnesses/gemini/index.js +48 -0
- package/dist/integrations/harnesses/gemini/result-extractor.js +121 -0
- package/dist/integrations/harnesses/ids.js +24 -0
- package/dist/integrations/harnesses/index.js +54 -34
- package/dist/integrations/harnesses/opencode/agent-builder.js +23 -5
- package/dist/integrations/harnesses/opencode/config-import.js +1 -3
- package/dist/integrations/harnesses/opencode/index.js +14 -32
- package/dist/integrations/harnesses/opencode/session-log.js +67 -125
- package/dist/integrations/harnesses/opencode-sdk/harness.js +51 -0
- package/dist/integrations/harnesses/opencode-sdk/sdk-runner.js +681 -108
- package/dist/integrations/harnesses/openhands/agent-builder.js +128 -0
- package/dist/integrations/harnesses/openhands/index.js +48 -0
- package/dist/integrations/harnesses/openhands/result-extractor.js +103 -0
- package/dist/integrations/harnesses/pi/agent-builder.js +97 -0
- package/dist/integrations/harnesses/pi/index.js +45 -0
- package/dist/integrations/harnesses/pi/result-extractor.js +135 -0
- package/dist/integrations/harnesses/shared.js +17 -0
- package/dist/integrations/harnesses/types.js +43 -32
- package/dist/integrations/lockfile.js +211 -24
- package/dist/integrations/session-logs/index.js +36 -39
- package/dist/integrations/session-logs/provider-base.js +113 -0
- package/dist/llm/client.js +182 -110
- package/dist/llm/embedders/deterministic.js +2 -2
- package/dist/llm/embedders/remote.js +21 -9
- package/dist/llm/feature-gate.js +17 -57
- package/dist/llm/graph-extract.js +12 -13
- package/dist/llm/index-passes.js +8 -42
- package/dist/llm/memory-infer.js +144 -1
- package/dist/llm/metadata-enhance.js +45 -30
- package/dist/llm/structured-call.js +16 -8
- package/dist/llm/usage-persist.js +30 -5
- package/dist/llm/usage-telemetry.js +59 -6
- package/dist/output/cli-hints.js +1 -2
- package/dist/output/command-registry.js +27 -0
- package/dist/output/context.js +22 -7
- package/dist/output/format-exempt.js +80 -0
- package/dist/output/generic-render.js +251 -0
- package/dist/output/html-render.js +11 -16
- package/dist/output/render-registry.js +57 -0
- package/dist/output/renderers.js +14 -279
- package/dist/output/shapes/curate.js +10 -1
- package/dist/output/shapes/events.js +12 -7
- package/dist/output/shapes/helpers.js +58 -84
- package/dist/output/shapes/passthrough.js +11 -39
- package/dist/output/shapes/proposal/producer.js +15 -7
- package/dist/output/shapes/registry.js +12 -6
- package/dist/output/shapes.js +0 -9
- package/dist/output/text/{init.js → bundle-create.js} +3 -1
- package/dist/output/text/bundle-show.js +7 -0
- package/dist/output/text/command-format.js +562 -0
- package/dist/output/text/env.js +1 -3
- package/dist/output/text/events.js +8 -7
- package/dist/output/text/helpers.js +15 -1164
- package/dist/output/text/proposal/producer.js +4 -2
- package/dist/output/text/proposal-format.js +202 -0
- package/dist/output/text/registry-commands.js +1 -2
- package/dist/output/text/registry.js +12 -6
- package/dist/output/text/show-directives.js +117 -0
- package/dist/output/text/show-format.js +103 -0
- package/dist/output/text/sync.js +5 -0
- package/dist/output/text/workflow-format.js +332 -0
- package/dist/output/text/workflow.js +3 -2
- package/dist/output/text.js +10 -19
- package/dist/registry/factory.js +4 -6
- package/dist/registry/origin-resolve.js +16 -27
- package/dist/registry/providers/skills-sh.js +3 -3
- package/dist/registry/providers/static-index.js +15 -25
- package/dist/registry/resolve.js +43 -94
- package/dist/registry/semver.js +43 -0
- package/dist/runtime.js +81 -12
- package/dist/scripts/akm-migrate.js +35529 -0
- package/dist/setup/detect.js +5 -7
- package/dist/setup/detected-engines.js +136 -0
- package/dist/setup/engine-config.js +100 -0
- package/dist/setup/registry-stash-loader.js +3 -3
- package/dist/setup/semantic-assets.js +12 -9
- package/dist/setup/setup.js +444 -208
- package/dist/setup/steps/connection-shared.js +120 -0
- package/dist/setup/steps/connection.js +108 -305
- package/dist/setup/steps/platforms.js +13 -12
- package/dist/setup/steps/semantic.js +15 -3
- package/dist/setup/steps/sources.js +21 -15
- package/dist/setup/steps/stashdir.js +6 -4
- package/dist/setup/steps/tasks.js +236 -119
- package/dist/setup/steps.js +3 -2
- package/dist/sources/freshness.js +39 -0
- package/dist/sources/provider-factory.js +11 -17
- package/dist/sources/providers/filesystem.js +2 -3
- package/dist/sources/providers/git-install.js +278 -34
- package/dist/sources/providers/git-provider.js +54 -56
- package/dist/sources/providers/git-stash.js +420 -91
- package/dist/sources/providers/git.js +2 -2
- package/dist/sources/providers/npm.js +16 -19
- package/dist/sources/providers/provider-utils.js +47 -22
- package/dist/sources/providers/sync-from-ref.js +3 -9
- package/dist/sources/providers/website.js +2 -2
- package/dist/sources/resolve.js +11 -10
- package/dist/sources/snapshot-fetchers/types.js +4 -0
- package/dist/sources/{website-ingest.js → snapshot-fetchers/website-ingest.js} +110 -41
- package/dist/storage/database.js +60 -4
- package/dist/storage/engines/sqlite-migrations.js +156 -5
- package/dist/storage/locations.js +1 -2
- package/dist/storage/repositories/canaries-repository.js +1 -1
- package/dist/storage/repositories/events-repository.js +51 -11
- package/dist/storage/repositories/improve-runs-repository.js +6 -32
- package/dist/storage/repositories/index-connection.js +79 -0
- package/dist/storage/repositories/index-db.js +4 -3
- package/dist/storage/repositories/index-entries-repository.js +863 -0
- package/dist/{indexer/db/entry-mapper.js → storage/repositories/index-entry-mapper.js} +19 -2
- package/dist/storage/repositories/index-entry-types.js +4 -0
- package/dist/storage/repositories/index-fts-repository.js +167 -0
- package/dist/storage/repositories/index-llm-cache-repository.js +108 -0
- package/dist/storage/repositories/index-meta-repository.js +49 -0
- package/dist/{indexer/db/schema.js → storage/repositories/index-schema.js} +226 -100
- package/dist/storage/repositories/index-sql.js +12 -0
- package/dist/storage/repositories/index-utility-repository.js +356 -0
- package/dist/storage/repositories/index-vec-repository.js +250 -0
- package/dist/storage/repositories/outcome-repository.js +119 -0
- package/dist/storage/repositories/proposals-repository.js +317 -75
- package/dist/storage/repositories/registry-cache.js +1 -1
- package/dist/storage/repositories/salience-repository.js +172 -0
- package/dist/storage/repositories/task-history-repository.js +110 -3
- package/dist/storage/repositories/workflow-runs-repository.js +240 -19
- package/dist/tasks/backends/cron.js +169 -46
- package/dist/tasks/backends/exec-utils.js +76 -3
- package/dist/tasks/backends/index.js +6 -9
- package/dist/tasks/backends/launchd.js +292 -55
- package/dist/tasks/backends/schtasks.js +557 -70
- package/dist/tasks/backends/types.js +4 -0
- package/dist/tasks/command-executable.js +93 -0
- package/dist/tasks/embedded.js +56 -38
- package/dist/tasks/parser.js +156 -64
- package/dist/tasks/resolve-akm-bin.js +144 -51
- package/dist/tasks/runner.js +377 -209
- package/dist/tasks/schedule.js +108 -19
- package/dist/tasks/scheduler-invocation.js +296 -0
- package/dist/tasks/schema.js +1 -1
- package/dist/tasks/task-id.js +35 -0
- package/dist/tasks/validator.js +30 -16
- package/dist/text-import-hook.mjs +1 -1
- package/dist/workflows/authoring/authoring.js +104 -43
- package/dist/workflows/authoring/scope-key.js +1 -1
- package/dist/workflows/cli.js +0 -16
- package/dist/workflows/concurrency-policy.js +15 -0
- package/dist/workflows/exec/brief.js +450 -0
- package/dist/workflows/exec/frozen-judge.js +47 -0
- package/dist/workflows/exec/native-executor.js +1038 -0
- package/dist/workflows/exec/param-secrets.js +115 -0
- package/dist/workflows/exec/report.js +1460 -0
- package/dist/workflows/exec/run-workflow.js +602 -0
- package/dist/workflows/exec/scheduler.js +71 -0
- package/dist/workflows/exec/step-work.js +1190 -0
- package/dist/workflows/exec/unit-writer.js +23 -0
- package/dist/workflows/exec/workflow-engine-gate.js +67 -0
- package/dist/workflows/exec/worktree.js +171 -0
- package/dist/workflows/ir/compile.js +246 -0
- package/dist/workflows/ir/freeze.js +233 -0
- package/dist/workflows/ir/params.js +54 -0
- package/dist/workflows/ir/plan-hash.js +68 -0
- package/dist/workflows/ir/schema.js +540 -0
- package/dist/workflows/parser.js +878 -304
- package/dist/workflows/program/expressions.js +181 -0
- package/dist/workflows/program/schema.js +51 -0
- package/dist/workflows/renderer.js +100 -45
- package/dist/workflows/resource-limits.js +22 -0
- package/dist/workflows/runtime/agent-identity.js +59 -14
- package/dist/workflows/runtime/checkin.js +1 -1
- package/dist/workflows/runtime/plan-classifier.js +131 -0
- package/dist/workflows/runtime/runs.js +376 -119
- package/dist/workflows/runtime/unit-checkin.js +45 -0
- package/dist/workflows/runtime/unit-phases.js +20 -0
- package/dist/workflows/runtime/workflow-asset-loader.js +241 -40
- package/dist/workflows/schema.js +1 -11
- package/dist/workflows/validate-summary.js +2 -3
- package/dist/workflows/validator.js +52 -30
- package/docs/README.md +42 -78
- package/docs/migration/README.md +8 -0
- package/docs/migration/release-notes/0.6.0.md +1 -1
- package/docs/migration/release-notes/0.7.0.md +9 -8
- package/docs/migration/release-notes/0.9.0.md +158 -14
- package/docs/migration/v0.7-to-v0.8.md +46 -47
- package/docs/migration/v0.8-to-v0.9.md +844 -0
- package/docs/reference/README.md +12 -0
- package/docs/reference/data-and-telemetry.md +333 -0
- package/package.json +21 -17
- package/schemas/akm-asset-envelope.json +93 -0
- package/schemas/akm-config.json +4636 -0
- package/schemas/akm-task.json +87 -0
- package/schemas/akm-workflow.json +373 -0
- package/dist/akm-migrate-storage +0 -38
- package/dist/assets/help/help-accept.md +0 -12
- package/dist/assets/help/help-improve.md +0 -84
- package/dist/assets/help/help-proposals.md +0 -17
- package/dist/assets/help/help-propose.md +0 -17
- package/dist/assets/help/help-reject.md +0 -11
- package/dist/assets/profiles/frequent.json +0 -13
- package/dist/assets/profiles/recombine-only.json +0 -21
- package/dist/assets/profiles/reflect-distill.json +0 -30
- package/dist/assets/profiles/synthesize.json +0 -15
- package/dist/assets/prompts/procedural-system.md +0 -44
- package/dist/assets/prompts/recombine-system.md +0 -40
- package/dist/assets/prompts/staleness-detect-system.md +0 -6
- package/dist/assets/tasks/core/backup.yml +0 -4
- package/dist/assets/tasks/graph-refresh-weekly.yml +0 -10
- package/dist/assets/templates/html/default.html +0 -78
- package/dist/assets/templates/html/vendor/echarts.min.js +0 -45
- package/dist/assets/wiki/index-template.md +0 -12
- package/dist/assets/wiki/ingest-workflow-template.md +0 -83
- package/dist/assets/wiki/log-template.md +0 -8
- package/dist/assets/wiki/schema-template.md +0 -61
- package/dist/cli/config-migrate.js +0 -150
- package/dist/cli/config-validate.js +0 -39
- package/dist/commands/graph/graph-cli.js +0 -124
- package/dist/commands/graph/graph.js +0 -487
- package/dist/commands/improve/calibration.js +0 -161
- package/dist/commands/improve/dedup.js +0 -482
- package/dist/commands/improve/extract-watch.js +0 -140
- package/dist/commands/improve/hot-probation.js +0 -45
- package/dist/commands/improve/improve-auto-accept.js +0 -276
- package/dist/commands/improve/improve-profiles.js +0 -168
- package/dist/commands/improve/procedural.js +0 -398
- package/dist/commands/improve/recombine.js +0 -818
- package/dist/commands/improve/schema-similarity-gate.js +0 -168
- package/dist/commands/lint/agent-linter.js +0 -44
- package/dist/commands/lint/command-linter.js +0 -44
- package/dist/commands/lint/default-linter.js +0 -16
- package/dist/commands/lint/fact-linter.js +0 -39
- package/dist/commands/lint/knowledge-linter.js +0 -16
- package/dist/commands/lint/memory-linter.js +0 -61
- package/dist/commands/lint/registry.js +0 -41
- package/dist/commands/lint/skill-linter.js +0 -45
- package/dist/commands/lint/task-linter.js +0 -50
- package/dist/commands/lint/workflow-linter.js +0 -81
- package/dist/commands/proposal/legacy-import.js +0 -115
- package/dist/commands/sources/history.js +0 -196
- package/dist/commands/tasks/default-tasks.js +0 -186
- package/dist/commands/wiki-cli.js +0 -292
- package/dist/core/asset/asset-registry.js +0 -76
- package/dist/core/asset/asset-spec.js +0 -259
- package/dist/core/config/config-migration.js +0 -602
- package/dist/core/deep-merge.js +0 -38
- package/dist/core/eval/rank-metrics.js +0 -113
- package/dist/core/ripgrep/install.js +0 -163
- package/dist/core/ripgrep/resolve.js +0 -81
- package/dist/indexer/db/db.js +0 -1413
- package/dist/indexer/manifest.js +0 -170
- package/dist/indexer/passes/metadata-contributors.js +0 -31
- package/dist/indexer/usage/unmigrated-vaults-guard.js +0 -94
- package/dist/integrations/harnesses/opencode-sdk/index.js +0 -49
- package/dist/llm/call-ai.js +0 -62
- package/dist/llm/memory-infer-impl.js +0 -138
- package/dist/output/shapes/distill.js +0 -14
- package/dist/output/shapes/history.js +0 -11
- package/dist/output/text/distill.js +0 -6
- package/dist/output/text/enable-disable.js +0 -8
- package/dist/output/text/history.js +0 -6
- package/dist/output/text/wiki.js +0 -16
- package/dist/registry/build-index.js +0 -386
- package/dist/scripts/migrate-storage.js +0 -19108
- package/dist/scripts/migrations/import-fs-improve-runs-to-db.js +0 -9411
- package/dist/scripts/migrations/v16-to-v17.js +0 -141
- package/dist/setup/legacy-config.js +0 -106
- package/dist/storage/repositories/consolidation-repository.js +0 -38
- package/dist/storage/repositories/recombine-repository.js +0 -213
- package/dist/wiki/wiki-templates.js +0 -15
- package/dist/wiki/wiki.js +0 -1012
- package/dist/workflows/db.js +0 -215
- package/docs/data-and-telemetry.md +0 -226
- /package/dist/sources/{wiki-fetchers → snapshot-fetchers}/registry.js +0 -0
- /package/dist/sources/{wiki-fetchers → snapshot-fetchers}/youtube.js +0 -0
|
@@ -4,58 +4,47 @@
|
|
|
4
4
|
import { createHash } from "node:crypto";
|
|
5
5
|
import fs from "node:fs";
|
|
6
6
|
import path from "node:path";
|
|
7
|
-
import readline from "node:readline";
|
|
8
7
|
import consolidateSystemPrompt from "../../assets/prompts/consolidate-system.md" with { type: "text" };
|
|
9
|
-
import {
|
|
8
|
+
import { detectAdapterId } from "../../core/adapter/detect-adapter.js";
|
|
10
9
|
import { assembleAssetFromString, serializeFrontmatter } from "../../core/asset/asset-serialize.js";
|
|
11
10
|
import { parseFrontmatter } from "../../core/asset/frontmatter.js";
|
|
12
|
-
import {
|
|
13
|
-
import {
|
|
14
|
-
import { ConfigError } from "../../core/errors.js";
|
|
15
|
-
// Note: appendEvent import removed (WS-3a: archive TTL machinery retired)
|
|
11
|
+
import { conceptIdFromTypeName, displayRef, parseRefInput } from "../../core/asset/resolve-ref.js";
|
|
12
|
+
import { getImproveProcessConfig, loadConfig } from "../../core/config/config.js";
|
|
16
13
|
import { parseEmbeddedJsonResponse } from "../../core/parse.js";
|
|
17
|
-
import {
|
|
18
|
-
import {
|
|
19
|
-
import {
|
|
20
|
-
import {
|
|
21
|
-
import {
|
|
22
|
-
import {
|
|
23
|
-
import {
|
|
24
|
-
import { shouldSkipHotProbationInLlm } from "./hot-probation.js";
|
|
25
|
-
import { writeContradictEdge } from "./memory/memory-belief.js";
|
|
26
|
-
// Re-export the moved helpers so existing test imports continue to resolve.
|
|
27
|
-
export { hasSupersededStatus, validateProposalFrontmatter };
|
|
28
|
-
import { openStateDatabase, withStateDb } from "../../core/state-db.js";
|
|
29
|
-
import { warn } from "../../core/warn.js";
|
|
30
|
-
import { commitWriteTargetBoundary, deleteAssetFromSource, resolveWriteTarget, writeAssetToSource, } from "../../core/write-source.js";
|
|
31
|
-
import { closeDatabase, findEntryIdByRef, getAllEntries, getEntryById, getNeighborsByEntryId, openExistingDatabase, } from "../../indexer/db/db.js";
|
|
32
|
-
import { resolveImproveProcessRunnerFromProfile, runnerIsLlm } from "../../integrations/agent/runner.js";
|
|
33
|
-
import { chatCompletion } from "../../llm/client.js";
|
|
14
|
+
import { resolveStandardsContext } from "../../core/standards/resolve-standards-context.js";
|
|
15
|
+
import { openStateDatabase } from "../../core/state-db.js";
|
|
16
|
+
import { parseSinceToIsoLenient } from "../../core/time.js";
|
|
17
|
+
import { warn, warnVerbose } from "../../core/warn.js";
|
|
18
|
+
import { resolveWriteTarget } from "../../core/write-source.js";
|
|
19
|
+
import { getDefaultLlmConfig } from "../../integrations/agent/engine-resolution.js";
|
|
20
|
+
import { materializeLlmRunnerConnection, resolveImproveProcessRunner } from "../../integrations/agent/runner.js";
|
|
34
21
|
import { cosineSimilarity, embedBatch, resolveEmbeddingModelId } from "../../llm/embedder.js";
|
|
35
|
-
import {
|
|
36
|
-
import { getConsolidationJudgedMap, upsertConsolidationJudged, } from "../../storage/repositories/consolidation-repository.js";
|
|
22
|
+
import { callStructured } from "../../llm/structured-call.js";
|
|
37
23
|
import { getBodyEmbeddings, upsertBodyEmbeddings } from "../../storage/repositories/embeddings-repository.js";
|
|
24
|
+
import { closeDatabase, openExistingDatabase } from "../../storage/repositories/index-connection.js";
|
|
25
|
+
import { findEntryIdByRef, getAllEntries, getEntryById } from "../../storage/repositories/index-entries-repository.js";
|
|
26
|
+
import { getNeighborsByEntryId } from "../../storage/repositories/index-vec-repository.js";
|
|
27
|
+
import { isProposalSkipped, listProposals, proposalContent } from "../proposal/repository.js";
|
|
28
|
+
import { hasSupersededStatus, validateProposalFrontmatter } from "../proposal/validators/proposal-quality-validators.js";
|
|
29
|
+
import { cacheHash } from "./content-hash.js";
|
|
30
|
+
import { resolveImproveStrategy, resolveProcessEnabled } from "./improve-strategies.js";
|
|
31
|
+
import { emitProposal } from "./proposal-envelope.js";
|
|
32
|
+
import { createRunContext } from "./run-context.js";
|
|
38
33
|
// Chunk sizing + per-chunk prompt assembly live in ./consolidate/chunking.
|
|
39
|
-
// Imported for internal use by the orchestrator and re-exported for importers.
|
|
40
34
|
import { buildChunkPrompt, computeSafeChunkSize, DEFAULT_CONTEXT_LENGTH_TOKENS } from "./consolidate/chunking.js";
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
// ./consolidate/sanitize. Imported for internal use + re-exported for importers.
|
|
44
|
-
import { normalizeUpdatedField, sanitizeMergedContent } from "./consolidate/sanitize.js";
|
|
45
|
-
export { normalizeUpdatedField, sanitizeMergedContent, stripOuterCodeFence } from "./consolidate/sanitize.js";
|
|
46
|
-
// Eligibility / safety predicates live in ./consolidate/eligibility. Imported
|
|
47
|
-
// for internal guard use; the two public predicates are re-exported.
|
|
48
|
-
import { consolidateGuardStatus, isConsolidationEligibleMemoryName, isHotCapturedMemory, } from "./consolidate/eligibility.js";
|
|
49
|
-
export { isConsolidationEligibleMemoryName, isHotCapturedMemory } from "./consolidate/eligibility.js";
|
|
35
|
+
// Eligibility / safety predicates live in ./consolidate/eligibility.
|
|
36
|
+
import { isConsolidationEligibleMemoryName, isHotCapturedMemory } from "./consolidate/eligibility.js";
|
|
50
37
|
// Plan parsing / merging (pure op-reconciliation algebra) lives in
|
|
51
|
-
// ./consolidate/merge.
|
|
38
|
+
// ./consolidate/merge.
|
|
52
39
|
import { isValidOp, mergePlans } from "./consolidate/merge.js";
|
|
53
|
-
|
|
40
|
+
// LLM-output sanitization (pure string/frontmatter transforms) lives in
|
|
41
|
+
// ./consolidate/sanitize.
|
|
42
|
+
import { sanitizeMergedContent } from "./consolidate/sanitize.js";
|
|
54
43
|
// ── Prompts ─────────────────────────────────────────────────────────────────
|
|
55
44
|
const CONSOLIDATE_SYSTEM_PROMPT = consolidateSystemPrompt;
|
|
56
45
|
/**
|
|
57
46
|
* JSON Schema for structured consolidate plans (PR 1 of the asset-writers
|
|
58
|
-
* decision — see knowledge
|
|
47
|
+
* decision — see knowledge/projects/akm/asset-writers-investigation/00-synthesis).
|
|
59
48
|
* Mirrors the {ops[], warnings?[]} shape currently described in
|
|
60
49
|
* CONSOLIDATE_SYSTEM_PROMPT. Providers with `supportsJsonSchema: true` enforce
|
|
61
50
|
* the shape upstream so the chunk-level "invalid plan from AI — skipping"
|
|
@@ -86,6 +75,7 @@ export const CONSOLIDATE_PLAN_JSON_SCHEMA = {
|
|
|
86
75
|
secondaries: {
|
|
87
76
|
type: "array",
|
|
88
77
|
minItems: 1,
|
|
78
|
+
maxItems: 1,
|
|
89
79
|
items: { type: "string", minLength: 1 },
|
|
90
80
|
},
|
|
91
81
|
mergeStrategy: { type: "string", minLength: 1 },
|
|
@@ -138,7 +128,7 @@ export const CONSOLIDATE_PLAN_JSON_SCHEMA = {
|
|
|
138
128
|
},
|
|
139
129
|
},
|
|
140
130
|
};
|
|
141
|
-
async function clusterMemoriesBySimilarity(memories, config, stateDb) {
|
|
131
|
+
async function clusterMemoriesBySimilarity(memories, config, stateDb, signal) {
|
|
142
132
|
const noTelemetry = { embedMs: 0, cacheHits: 0, cacheMisses: 0 };
|
|
143
133
|
if (memories.length < 3 || !config.embedding)
|
|
144
134
|
return { ordered: memories, embedTelemetry: noTelemetry };
|
|
@@ -189,7 +179,7 @@ async function clusterMemoriesBySimilarity(memories, config, stateDb) {
|
|
|
189
179
|
if (missTexts.length > 0) {
|
|
190
180
|
const embedStart = Date.now();
|
|
191
181
|
try {
|
|
192
|
-
missVecs = await embedBatch(missTexts, config.embedding);
|
|
182
|
+
missVecs = await embedBatch(missTexts, config.embedding, signal);
|
|
193
183
|
}
|
|
194
184
|
catch {
|
|
195
185
|
// Fail open: embedding failures degrade gracefully to original order.
|
|
@@ -275,8 +265,8 @@ async function clusterMemoriesBySimilarity(memories, config, stateDb) {
|
|
|
275
265
|
* the per-chunk prompt can annotate memories whose body would just produce
|
|
276
266
|
* a deterministic `dedup_pending_proposal` skip. Uses `cacheHash` (case-
|
|
277
267
|
* preserving stripped body) — the same domain used by the body-embedding
|
|
278
|
-
* cache
|
|
279
|
-
*
|
|
268
|
+
* cache. Empty set on any read/parse error — fail-safe to "annotate nothing"
|
|
269
|
+
* so the LLM still proposes.
|
|
280
270
|
*/
|
|
281
271
|
function loadPendingConsolidateProposalHashes(stashDir) {
|
|
282
272
|
const hashes = new Set();
|
|
@@ -284,7 +274,7 @@ function loadPendingConsolidateProposalHashes(stashDir) {
|
|
|
284
274
|
const pending = listProposals(stashDir, { status: "pending" }).filter((p) => p.source === "consolidate");
|
|
285
275
|
for (const p of pending) {
|
|
286
276
|
try {
|
|
287
|
-
hashes.add(cacheHash(p
|
|
277
|
+
hashes.add(cacheHash(proposalContent(p)));
|
|
288
278
|
}
|
|
289
279
|
catch {
|
|
290
280
|
// skip malformed payloads — they can't dedup anyway
|
|
@@ -296,210 +286,41 @@ function loadPendingConsolidateProposalHashes(stashDir) {
|
|
|
296
286
|
}
|
|
297
287
|
return hashes;
|
|
298
288
|
}
|
|
299
|
-
|
|
300
|
-
|
|
301
|
-
}
|
|
302
|
-
function getBackupDir(stashDir, timestamp) {
|
|
303
|
-
return path.join(stashDir, ".akm", "consolidate-backup", timestamp);
|
|
304
|
-
}
|
|
305
|
-
function removeStaleJournal(stashDir, journal, warnings) {
|
|
306
|
-
const journalPath = getJournalPath(stashDir);
|
|
307
|
-
try {
|
|
308
|
-
fs.unlinkSync(journalPath);
|
|
309
|
-
}
|
|
310
|
-
catch {
|
|
311
|
-
warnings.push(`Failed to remove stale consolidate journal at ${journalPath}.`);
|
|
312
|
-
}
|
|
313
|
-
const backupTimestamp = typeof journal.backupTimestamp === "string" && journal.backupTimestamp.trim().length > 0
|
|
314
|
-
? journal.backupTimestamp.trim()
|
|
315
|
-
: typeof journal.startedAt === "string" && journal.startedAt.trim().length > 0
|
|
316
|
-
? journal.startedAt.replace(/[:.]/g, "-")
|
|
317
|
-
: "";
|
|
318
|
-
if (!backupTimestamp)
|
|
319
|
-
return;
|
|
320
|
-
const backupDir = getBackupDir(stashDir, backupTimestamp);
|
|
321
|
-
if (!fs.existsSync(backupDir))
|
|
322
|
-
return;
|
|
323
|
-
try {
|
|
324
|
-
fs.rmSync(backupDir, { recursive: true, force: true });
|
|
325
|
-
}
|
|
326
|
-
catch {
|
|
327
|
-
warnings.push(`Failed to remove stale consolidate backup at ${backupDir}.`);
|
|
328
|
-
}
|
|
329
|
-
warnings.push(`Cleared stale consolidate backup at ${backupDir}.`);
|
|
330
|
-
}
|
|
331
|
-
function checkForIncompleteJournal(stashDir, recoveryMode, warnings) {
|
|
332
|
-
const journalPath = getJournalPath(stashDir);
|
|
333
|
-
if (!fs.existsSync(journalPath))
|
|
334
|
-
return;
|
|
335
|
-
let journal;
|
|
336
|
-
try {
|
|
337
|
-
journal = JSON.parse(fs.readFileSync(journalPath, "utf8"));
|
|
338
|
-
}
|
|
339
|
-
catch {
|
|
340
|
-
if (recoveryMode === "clean") {
|
|
341
|
-
try {
|
|
342
|
-
fs.unlinkSync(journalPath);
|
|
343
|
-
warnings.push(`Removed unreadable consolidate journal at ${journalPath}.`);
|
|
344
|
-
}
|
|
345
|
-
catch {
|
|
346
|
-
warnings.push(`Failed to remove unreadable consolidate journal at ${journalPath}.`);
|
|
347
|
-
}
|
|
348
|
-
return;
|
|
349
|
-
}
|
|
350
|
-
throw new ConfigError(`Incomplete consolidation state detected: unreadable journal at ${journalPath}. Re-run with --consolidate-recovery clean to remove stale journal artifacts, or remove the file manually.`, "INVALID_CONFIG_FILE");
|
|
351
|
-
}
|
|
352
|
-
const operationCount = Array.isArray(journal.operations) ? journal.operations.length : 0;
|
|
353
|
-
const completedCount = Array.isArray(journal.completed) ? journal.completed.length : 0;
|
|
354
|
-
if (completedCount >= operationCount)
|
|
355
|
-
return;
|
|
356
|
-
if (recoveryMode === "clean") {
|
|
357
|
-
removeStaleJournal(stashDir, journal, warnings);
|
|
358
|
-
warnings.push(`Removed stale consolidation journal at ${journalPath} (${completedCount}/${operationCount} operations completed).`);
|
|
359
|
-
return;
|
|
360
|
-
}
|
|
361
|
-
const backupHint = typeof journal.backupTimestamp === "string" && journal.backupTimestamp.trim().length > 0
|
|
362
|
-
? ` Backup dir: ${getBackupDir(stashDir, journal.backupTimestamp.trim())}.`
|
|
363
|
-
: "";
|
|
364
|
-
throw new ConfigError(`Incomplete consolidation run detected at ${journalPath} (${completedCount}/${operationCount} operations completed). Re-run with --consolidate-recovery clean to remove stale journal artifacts.${backupHint}`, "INVALID_CONFIG_FILE");
|
|
365
|
-
}
|
|
366
|
-
function writeJournal(stashDir, ops, backupTimestamp) {
|
|
367
|
-
const journalPath = getJournalPath(stashDir);
|
|
368
|
-
fs.mkdirSync(path.dirname(journalPath), { recursive: true });
|
|
369
|
-
const journal = {
|
|
370
|
-
startedAt: new Date().toISOString(),
|
|
371
|
-
operations: ops,
|
|
372
|
-
completed: [],
|
|
373
|
-
backupTimestamp,
|
|
374
|
-
};
|
|
375
|
-
fs.writeFileSync(journalPath, JSON.stringify(journal, null, 2), "utf8");
|
|
376
|
-
}
|
|
377
|
-
function markJournalCompleted(stashDir, opRef) {
|
|
378
|
-
const journalPath = getJournalPath(stashDir);
|
|
379
|
-
if (!fs.existsSync(journalPath))
|
|
380
|
-
return;
|
|
381
|
-
try {
|
|
382
|
-
const journal = JSON.parse(fs.readFileSync(journalPath, "utf8"));
|
|
383
|
-
journal.completed.push(opRef);
|
|
384
|
-
fs.writeFileSync(journalPath, JSON.stringify(journal, null, 2), "utf8");
|
|
385
|
-
}
|
|
386
|
-
catch {
|
|
387
|
-
// best-effort
|
|
388
|
-
}
|
|
389
|
-
}
|
|
390
|
-
function cleanupJournal(stashDir, timestamp) {
|
|
391
|
-
const journalPath = getJournalPath(stashDir);
|
|
392
|
-
try {
|
|
393
|
-
fs.unlinkSync(journalPath);
|
|
394
|
-
}
|
|
395
|
-
catch {
|
|
396
|
-
// ignore
|
|
397
|
-
}
|
|
398
|
-
const backupDir = getBackupDir(stashDir, timestamp);
|
|
399
|
-
try {
|
|
400
|
-
fs.rmSync(backupDir, { recursive: true, force: true });
|
|
401
|
-
}
|
|
402
|
-
catch {
|
|
403
|
-
// ignore
|
|
404
|
-
}
|
|
405
|
-
}
|
|
406
|
-
function backupFile(filePath, backupDir, name) {
|
|
289
|
+
/** Parse a stored provenance ref and emit its canonical D-R5 display spelling. */
|
|
290
|
+
function canonicalStoredXref(ref) {
|
|
407
291
|
try {
|
|
408
|
-
|
|
409
|
-
|
|
292
|
+
const p = parseRefInput(ref);
|
|
293
|
+
return displayRef({ type: p.type, name: p.name, bundleId: p.origin });
|
|
410
294
|
}
|
|
411
295
|
catch {
|
|
412
|
-
|
|
296
|
+
return undefined;
|
|
413
297
|
}
|
|
414
298
|
}
|
|
415
|
-
|
|
416
|
-
|
|
417
|
-
* Inject `generation` and `source_refs` into merged content.
|
|
418
|
-
* generation = max(sourceGenerations) + 1.
|
|
419
|
-
* source_refs = UNION of the provided provenance refs (participants + their
|
|
420
|
-
* cited sources) with anything already present in the merged frontmatter —
|
|
421
|
-
* R5 §4.2: the old set-if-absent behavior dropped second-generation
|
|
422
|
-
* provenance whenever the LLM emitted its own (partial) source_refs.
|
|
423
|
-
* Fails open — returns original content if frontmatter can't be parsed.
|
|
424
|
-
*/
|
|
425
|
-
function injectGenerationFrontmatter(mergedContent, sourceGenerations, provenanceRefs) {
|
|
426
|
-
try {
|
|
427
|
-
const parsed = parseFrontmatter(mergedContent);
|
|
428
|
-
const existingFm = parsed.data;
|
|
429
|
-
const existingRefs = Array.isArray(existingFm.source_refs) ? existingFm.source_refs.map(String) : [];
|
|
430
|
-
const updatedFm = {
|
|
431
|
-
...existingFm,
|
|
432
|
-
generation: computeMergedGeneration(sourceGenerations),
|
|
433
|
-
source_refs: [...new Set([...existingRefs, ...provenanceRefs])],
|
|
434
|
-
};
|
|
435
|
-
return assembleAssetFromString(serializeFrontmatter(updatedFm), parsed.content);
|
|
436
|
-
}
|
|
437
|
-
catch {
|
|
438
|
-
return mergedContent; // fail open
|
|
439
|
-
}
|
|
299
|
+
function canonicalXref(ref) {
|
|
300
|
+
return canonicalStoredXref(ref) ?? ref;
|
|
440
301
|
}
|
|
441
|
-
// ── Archive helper (P1-B: soft-invalidation) ─────────────────────────────────
|
|
442
302
|
/**
|
|
443
|
-
*
|
|
444
|
-
*
|
|
445
|
-
*
|
|
446
|
-
*
|
|
447
|
-
* Archive filename: `<iso-ts>-<opIndex>-<basename>.md`
|
|
448
|
-
* New frontmatter fields: status, superseded_at, superseded_by (optional),
|
|
449
|
-
* superseded_reason.
|
|
303
|
+
* The promoted asset's provenance xref set: existing body-frontmatter xrefs +
|
|
304
|
+
* the promoted source ref, deduped after canonicalization (WI-8.5b: emitted in
|
|
305
|
+
* the D-R5 new grammar via {@link canonicalXref}).
|
|
450
306
|
*/
|
|
451
|
-
function
|
|
452
|
-
const
|
|
453
|
-
|
|
454
|
-
let raw;
|
|
455
|
-
try {
|
|
456
|
-
raw = fs.readFileSync(filePath, "utf8");
|
|
457
|
-
}
|
|
458
|
-
catch {
|
|
459
|
-
if (warnings)
|
|
460
|
-
warnings.push(`archiveMemory: could not read ${ref} for archiving — skipping archive write`);
|
|
461
|
-
return;
|
|
462
|
-
}
|
|
463
|
-
let content = raw;
|
|
464
|
-
try {
|
|
465
|
-
const parsed = parseFrontmatter(raw);
|
|
466
|
-
const newFm = {
|
|
467
|
-
...parsed.data,
|
|
468
|
-
status: "superseded",
|
|
469
|
-
superseded_at: new Date().toISOString(),
|
|
470
|
-
...(supersededBy ? { superseded_by: supersededBy } : {}),
|
|
471
|
-
superseded_reason: reason,
|
|
472
|
-
};
|
|
473
|
-
content = assembleAssetFromString(serializeFrontmatter(newFm), parsed.content);
|
|
474
|
-
}
|
|
475
|
-
catch {
|
|
476
|
-
if (warnings)
|
|
477
|
-
warnings.push(`archiveMemory: could not parse frontmatter for ${ref} — archiving raw`);
|
|
478
|
-
}
|
|
479
|
-
const ts = timestampForFilename();
|
|
480
|
-
const safeName = path.basename(filePath, ".md");
|
|
481
|
-
const archivePath = path.join(archiveDir, `${ts}-${opIndex}-${safeName}.md`);
|
|
482
|
-
try {
|
|
483
|
-
fs.writeFileSync(archivePath, content, "utf8");
|
|
484
|
-
}
|
|
485
|
-
catch (e) {
|
|
486
|
-
if (warnings)
|
|
487
|
-
warnings.push(`archiveMemory: write failed for ${ref}: ${String(e)}`);
|
|
488
|
-
}
|
|
307
|
+
function promoteProvenanceXrefs(existing, sourceRef) {
|
|
308
|
+
const priors = Array.isArray(existing) ? existing.map(String) : [];
|
|
309
|
+
return [...new Set([...priors, sourceRef].map(canonicalXref))];
|
|
489
310
|
}
|
|
490
311
|
// ── LLM resolution ──────────────────────────────────────────────────────────
|
|
491
312
|
/**
|
|
492
313
|
* Resolve the LLM connection for the consolidate pass.
|
|
493
314
|
*
|
|
494
315
|
* Priority order (mirrors extract / reflect / distill — see
|
|
495
|
-
* `src/commands/extract.ts
|
|
496
|
-
* `
|
|
316
|
+
* `resolveExtractRunConfig` in `src/commands/improve/extract.ts` and the
|
|
317
|
+
* canonical `resolveImproveProcessRunner` pattern):
|
|
497
318
|
*
|
|
498
|
-
* 1. `
|
|
499
|
-
* via {@link
|
|
319
|
+
* 1. `improve.strategies.<name>.processes.consolidate.engine`
|
|
320
|
+
* via {@link resolveImproveProcessRunner}. Lets the user pin
|
|
500
321
|
* a dedicated model (e.g. `ministral-3b`) for consolidation instead of
|
|
501
|
-
* whatever `defaults.
|
|
502
|
-
* 2. `getDefaultLlmConfig(config)` — the baseline default LLM
|
|
322
|
+
* whatever `defaults.llmEngine` happens to be.
|
|
323
|
+
* 2. `getDefaultLlmConfig(config)` — the baseline default LLM engine.
|
|
503
324
|
*
|
|
504
325
|
* Regression guard (2026-05-26): before this resolver, `akmConsolidate`
|
|
505
326
|
* called `getDefaultLlmConfig` directly and silently ignored a configured
|
|
@@ -509,44 +330,11 @@ function archiveMemory(filePath, stashDir, ref, reason, opIndex, supersededBy, w
|
|
|
509
330
|
* `/tmp/akm-health-investigations/consolidation-no-op.md`.
|
|
510
331
|
*/
|
|
511
332
|
function resolveConsolidateLlmConfig(config, activeProfile) {
|
|
512
|
-
const
|
|
513
|
-
|
|
514
|
-
|
|
515
|
-
return runnerSpec.connection;
|
|
516
|
-
}
|
|
517
|
-
// Non-LLM runner modes (agent/sdk) don't apply to consolidate's HTTP path;
|
|
518
|
-
// fall back to the default LLM profile rather than disabling the pass.
|
|
333
|
+
const runnerSpec = resolveImproveProcessRunner(activeProfile, "consolidate", config);
|
|
334
|
+
if (runnerSpec)
|
|
335
|
+
return materializeLlmRunnerConnection(runnerSpec);
|
|
519
336
|
return getDefaultLlmConfig(config);
|
|
520
337
|
}
|
|
521
|
-
// ── Judged-state cache (#581) ────────────────────────────────────────────────
|
|
522
|
-
/**
|
|
523
|
-
* Stable content hash for a memory file used by the judged-state cache (#581)
|
|
524
|
-
* and the body-embedding cache (WS-3a). Uses `cacheHash` from dedup.ts
|
|
525
|
-
* (sha256 of the case-preserving stripped body) plus the sorted `tags` list,
|
|
526
|
-
* so semantic-metadata drift re-enters the judge while cosmetic frontmatter
|
|
527
|
-
* touches (`updated:`, `inferenceProcessed:`) still hash identically and never
|
|
528
|
-
* force a needless re-judge. Returns `undefined` on any read/parse error so
|
|
529
|
-
* callers fail open (treat the memory as un-cached → it stays in the LLM pool).
|
|
530
|
-
*/
|
|
531
|
-
function computeMemoryContentHash(filePath) {
|
|
532
|
-
try {
|
|
533
|
-
const raw = fs.readFileSync(filePath, "utf8");
|
|
534
|
-
let tagSuffix = "";
|
|
535
|
-
try {
|
|
536
|
-
const { data } = parseFrontmatter(raw);
|
|
537
|
-
const tags = Array.isArray(data?.tags) ? data.tags.map(String).sort() : [];
|
|
538
|
-
if (tags.length > 0)
|
|
539
|
-
tagSuffix = `\n\u0000tags:${tags.join(",")}`;
|
|
540
|
-
}
|
|
541
|
-
catch {
|
|
542
|
-
// Unparseable frontmatter → body-only hash (prior behaviour).
|
|
543
|
-
}
|
|
544
|
-
return cacheHash(raw + tagSuffix);
|
|
545
|
-
}
|
|
546
|
-
catch {
|
|
547
|
-
return undefined;
|
|
548
|
-
}
|
|
549
|
-
}
|
|
550
338
|
/**
|
|
551
339
|
* Build a {@link ConsolidateResult} from partial overrides, filling the envelope
|
|
552
340
|
* defaults (schemaVersion / ok / shape + the zeroed counters). Collapses the
|
|
@@ -571,6 +359,32 @@ export function makeConsolidateResult(overrides) {
|
|
|
571
359
|
};
|
|
572
360
|
}
|
|
573
361
|
// ── Main entry point ─────────────────────────────────────────────────────────
|
|
362
|
+
function resolveConsolidationWriteTarget(opts, config) {
|
|
363
|
+
if (opts.writeTarget) {
|
|
364
|
+
const root = path.resolve(opts.writeTarget.source.path);
|
|
365
|
+
return {
|
|
366
|
+
...opts.writeTarget,
|
|
367
|
+
source: {
|
|
368
|
+
...opts.writeTarget.source,
|
|
369
|
+
path: root,
|
|
370
|
+
adapterId: opts.writeTarget.source.adapterId ?? detectAdapterId(root),
|
|
371
|
+
},
|
|
372
|
+
};
|
|
373
|
+
}
|
|
374
|
+
if (opts.target) {
|
|
375
|
+
const target = resolveWriteTarget(config, opts.target);
|
|
376
|
+
return { ...target, source: { ...target.source, path: path.resolve(target.source.path) } };
|
|
377
|
+
}
|
|
378
|
+
if (opts.stashDir) {
|
|
379
|
+
const root = path.resolve(opts.stashDir);
|
|
380
|
+
return {
|
|
381
|
+
source: { kind: "filesystem", name: "stash", path: root, adapterId: detectAdapterId(root) },
|
|
382
|
+
config: { type: "filesystem", name: "stash", path: root, writable: true },
|
|
383
|
+
};
|
|
384
|
+
}
|
|
385
|
+
const target = resolveWriteTarget(config);
|
|
386
|
+
return { ...target, source: { ...target.source, path: path.resolve(target.source.path) } };
|
|
387
|
+
}
|
|
574
388
|
export async function akmConsolidate(opts = {}) {
|
|
575
389
|
const startMs = Date.now();
|
|
576
390
|
// Derive a stable PROV-DM token for this run. Callers (e.g. akmImprove)
|
|
@@ -578,16 +392,52 @@ export async function akmConsolidate(opts = {}) {
|
|
|
578
392
|
// standalone `akm consolidate` gets a self-contained token.
|
|
579
393
|
const sourceRun = opts.sourceRun ?? `consolidate-${startMs}`;
|
|
580
394
|
const config = opts.config ?? loadConfig();
|
|
581
|
-
const
|
|
582
|
-
|
|
395
|
+
const writeTarget = resolveConsolidationWriteTarget(opts, config);
|
|
396
|
+
opts = { ...opts, target: writeTarget.source.name, writeTarget };
|
|
397
|
+
opts = { ...opts, improveProfile: opts.improveProfile ?? resolveImproveStrategy(undefined, config).config };
|
|
398
|
+
const stashDir = writeTarget.source.path;
|
|
399
|
+
// WI-9.10: construct this run's RunContext from values already resolved
|
|
400
|
+
// above (sourceRun, config, stashDir) — no second config load, no new db
|
|
401
|
+
// handle. consolidate.ts has no `eventsCtx`/proposals-`ctx` option at all
|
|
402
|
+
// (WS-3a retired its only appendEvent usage; `emitProposal` here is always
|
|
403
|
+
// called with the default, seam-less ProposalsContext — see
|
|
404
|
+
// emitPromotionProposal below), so both get the safe empty-object default,
|
|
405
|
+
// behaviorally identical to `undefined` (EventsContext/ProposalsContext
|
|
406
|
+
// fields are all optional-chained by their consumers). `getLlmConfig`
|
|
407
|
+
// mirrors `planConsolidation`'s own resolution (`resolveConsolidateLlmConfig`)
|
|
408
|
+
// verbatim but lazily and independently — nothing calls `ctx.getLlmConfig`
|
|
409
|
+
// yet this stage, so this never duplicates real work, only the (pure,
|
|
410
|
+
// side-effect-free) resolution logic if invoked. consolidate has no `chat`
|
|
411
|
+
// seam (it drives the LLM directly via the HTTP client path, never through
|
|
412
|
+
// `chatCompletion`), so that field is left to its default.
|
|
413
|
+
const runContext = createRunContext({
|
|
414
|
+
stashDir,
|
|
415
|
+
config,
|
|
416
|
+
eventsCtx: {},
|
|
417
|
+
proposalsCtx: {},
|
|
418
|
+
getLlmConfig: () => {
|
|
419
|
+
const resolved = Object.hasOwn(opts, "llmConfig")
|
|
420
|
+
? (opts.llmConfig ?? undefined)
|
|
421
|
+
: resolveConsolidateLlmConfig(config, opts.improveProfile);
|
|
422
|
+
return resolved ?? null;
|
|
423
|
+
},
|
|
424
|
+
sourceRun,
|
|
425
|
+
dryRun: opts.dryRun ?? false,
|
|
426
|
+
signal: opts.signal,
|
|
427
|
+
});
|
|
428
|
+
const warnings = [];
|
|
429
|
+
if (!resolveProcessEnabled("consolidate", opts.improveProfile ?? resolveImproveStrategy(undefined, config).config)) {
|
|
583
430
|
return makeConsolidateResult({
|
|
584
|
-
|
|
431
|
+
// Sourced from runContext (identical value to `opts.dryRun ?? false`)
|
|
432
|
+
// so the constructed RunContext has a genuine downstream reference —
|
|
433
|
+
// consolidate's own content-read sites are out of this stage's stated
|
|
434
|
+
// item-2 scope (reflect + distill only; see the WI-9.10c report).
|
|
435
|
+
dryRun: runContext.dryRun,
|
|
585
436
|
target: opts.target ?? stashDir,
|
|
586
437
|
durationMs: Date.now() - startMs,
|
|
438
|
+
warnings,
|
|
587
439
|
});
|
|
588
440
|
}
|
|
589
|
-
const warnings = [];
|
|
590
|
-
checkForIncompleteJournal(stashDir, opts.recoveryMode ?? "abort", warnings);
|
|
591
441
|
// WS-3a: open one state.db handle shared by the body-embedding cache (dedup
|
|
592
442
|
// + cluster) and the judged-state cache. All callers in the function body
|
|
593
443
|
// receive this handle; it is closed in the `finally` block below.
|
|
@@ -600,7 +450,7 @@ export async function akmConsolidate(opts = {}) {
|
|
|
600
450
|
// State DB unavailable → skip the embedding cache for this run.
|
|
601
451
|
}
|
|
602
452
|
try {
|
|
603
|
-
return await akmConsolidateInner(opts, config, stashDir, startMs,
|
|
453
|
+
return await akmConsolidateInner(opts, config, stashDir, startMs, warnings, sharedStateDb);
|
|
604
454
|
}
|
|
605
455
|
finally {
|
|
606
456
|
sharedStateDb?.close();
|
|
@@ -642,14 +492,13 @@ function createConsolidateAccounting() {
|
|
|
642
492
|
}
|
|
643
493
|
/**
|
|
644
494
|
* Pass 1 — narrow the memory pool before any LLM work: drop stale DB entries,
|
|
645
|
-
*
|
|
646
|
-
*
|
|
647
|
-
*
|
|
648
|
-
*
|
|
649
|
-
* passes consume. Behavior-identical to the former inlined narrowing block.
|
|
495
|
+
* apply incremental-since narrowing, and cap to `opts.limit` (oldest-modified
|
|
496
|
+
* first). Returns an early envelope when the pool empties at any stage;
|
|
497
|
+
* otherwise returns the narrowed pool and the state the plan/apply passes
|
|
498
|
+
* consume. Behavior-identical to the former inlined narrowing block.
|
|
650
499
|
*/
|
|
651
|
-
async function narrowConsolidationPool(opts,
|
|
652
|
-
let memories = loadMemoriesForSource(opts.
|
|
500
|
+
async function narrowConsolidationPool(opts, stashDir, startMs, warnings) {
|
|
501
|
+
let memories = loadMemoriesForSource(opts.writeTarget?.source.path, stashDir, warnings);
|
|
653
502
|
// Pre-flight: filter out stale DB entries whose files no longer exist on
|
|
654
503
|
// disk. Without this, memories deleted by a prior run (but not yet
|
|
655
504
|
// reindexed) appear in chunk prompts, causing the LLM to generate plans
|
|
@@ -664,77 +513,12 @@ async function narrowConsolidationPool(opts, config, stashDir, startMs, warnings
|
|
|
664
513
|
// it was default-off and self-undoing (the next salience recompute
|
|
665
514
|
// unconditionally overwrote the demoted values). Continuous decay now lives
|
|
666
515
|
// in computeSalience's recency term, whose floor decays on a long half-life.)
|
|
667
|
-
// ── WS-3b Step 0c: Filter hot-probation assets from LLM merge pool ─────────
|
|
668
|
-
// Hot-probation assets (system-generated, not yet graduated from intake pass)
|
|
669
|
-
// are processed by the dedup pre-pass but excluded from the LLM clustering.
|
|
670
|
-
// This prevents noisy extractions from polluting LLM context. The dedup pass
|
|
671
|
-
// below still runs against them so they're cleaned up deterministically.
|
|
672
|
-
// DEFAULT OFF — only active when `processes.extract.hotProbation.enabled === true`
|
|
673
|
-
// (the flag that causes extract to tag new extractions as hot-probation).
|
|
674
|
-
// Without that flag no assets will ever carry the hot-probation marker, so
|
|
675
|
-
// running the filter loop would be pure unnecessary I/O over the full corpus.
|
|
676
|
-
const hotProbationEnabled = getImproveProcessConfig(config, "extract", opts.improveProfile)?.hotProbation
|
|
677
|
-
?.enabled === true;
|
|
678
|
-
let hotProbationCount = 0;
|
|
679
|
-
if (hotProbationEnabled) {
|
|
680
|
-
const hotProbationMemories = [];
|
|
681
|
-
const nonProbationMemories = [];
|
|
682
|
-
for (const m of memories) {
|
|
683
|
-
try {
|
|
684
|
-
const raw = fs.readFileSync(m.filePath, "utf8");
|
|
685
|
-
const parsed = parseFrontmatter(raw);
|
|
686
|
-
if (shouldSkipHotProbationInLlm(parsed.data)) {
|
|
687
|
-
hotProbationMemories.push(m);
|
|
688
|
-
hotProbationCount++;
|
|
689
|
-
}
|
|
690
|
-
else {
|
|
691
|
-
nonProbationMemories.push(m);
|
|
692
|
-
}
|
|
693
|
-
}
|
|
694
|
-
catch {
|
|
695
|
-
nonProbationMemories.push(m); // fail open
|
|
696
|
-
}
|
|
697
|
-
}
|
|
698
|
-
if (hotProbationCount > 0) {
|
|
699
|
-
warnings.push(`Hot-probation: ${hotProbationCount} hot-probation asset(s) routed to dedup-only pass (excluded from LLM merge pool).`);
|
|
700
|
-
memories = nonProbationMemories;
|
|
701
|
-
}
|
|
702
|
-
}
|
|
703
|
-
// ── Deterministic dedup pre-pass (#617) ─────────────────────────────────────
|
|
704
|
-
// Cheap, no-LLM fast path that collapses the obvious near-duplicates
|
|
705
|
-
// (`.derived` ↔ origin pairs + content twins) BEFORE the embedding-clustered
|
|
706
|
-
// LLM consolidation. DEFAULT OFF — when `dedup.enabled !== true` this is a
|
|
707
|
-
// no-op and the pass behaves byte-identically to today. Collapsed variants
|
|
708
|
-
// are pruned from the LLM pool so the model only ever sees genuinely
|
|
709
|
-
// distinct-but-related memories. Each dropped variant is archived (soft
|
|
710
|
-
// invalidation) before deletion, matching the LLM merge path.
|
|
711
|
-
// Dry-run never mutates the filesystem, so the dedup pre-pass is skipped
|
|
712
|
-
// entirely under `--dry-run` (the LLM plan preview below is unaffected).
|
|
713
|
-
let dedupCollapsed = 0;
|
|
714
|
-
if (opts.dedup?.enabled && !opts.dryRun) {
|
|
715
|
-
const dedupTimestamp = timestampForFilename();
|
|
716
|
-
const dedupResult = await runDeterministicDedup(stashDir, opts.dedup, config, (variantFilePath, variantName) => {
|
|
717
|
-
archiveMemory(variantFilePath, stashDir, `memory:${variantName}`, "collapsed by deterministic dedup pre-pass", -1, undefined, warnings);
|
|
718
|
-
backupFile(variantFilePath, getBackupDir(stashDir, dedupTimestamp), variantName);
|
|
719
|
-
}, opts.signal, sharedStateDb);
|
|
720
|
-
dedupCollapsed = dedupResult.collapsed;
|
|
721
|
-
warnings.push(...dedupResult.warnings);
|
|
722
|
-
if (dedupResult.consumedRefs.length > 0) {
|
|
723
|
-
const consumed = new Set(dedupResult.consumedRefs);
|
|
724
|
-
memories = memories.filter((m) => !consumed.has(`memory:${m.name}`));
|
|
725
|
-
warnings.push(`Deterministic dedup: collapsed ${dedupResult.collapsed} near-duplicate memor${dedupResult.collapsed === 1 ? "y" : "ies"} (no LLM) before chunking.`);
|
|
726
|
-
}
|
|
727
|
-
}
|
|
728
516
|
if (memories.length === 0) {
|
|
729
517
|
return {
|
|
730
518
|
done: true,
|
|
731
519
|
result: makeConsolidateResult({
|
|
732
520
|
dryRun: opts.dryRun ?? false,
|
|
733
521
|
target: opts.target ?? stashDir,
|
|
734
|
-
// #617: the deterministic dedup pre-pass may have emptied the pool by
|
|
735
|
-
// collapsing every remaining memory into a canonical. Surface those
|
|
736
|
-
// collapses in `deleted` so the run reports the work it actually did.
|
|
737
|
-
deleted: dedupCollapsed,
|
|
738
522
|
warnings,
|
|
739
523
|
durationMs: Date.now() - startMs,
|
|
740
524
|
}),
|
|
@@ -754,80 +538,11 @@ async function narrowConsolidationPool(opts, config, stashDir, startMs, warnings
|
|
|
754
538
|
};
|
|
755
539
|
}
|
|
756
540
|
}
|
|
757
|
-
// WS-5 perf telemetry
|
|
758
|
-
//
|
|
759
|
-
//
|
|
760
|
-
//
|
|
761
|
-
|
|
762
|
-
// `embedMs/cacheHits/cacheMisses` = accumulated from clusterMemoriesBySimilarity.
|
|
763
|
-
const perfMs = { dedupPoolSize: memories.length, judgedCacheSkipped: 0 };
|
|
764
|
-
// ── Judged-state cache narrowing (#581) ─────────────────────────────────────
|
|
765
|
-
// DEFAULT OFF. When enabled, skip every memory whose current content hash
|
|
766
|
-
// equals the hash recorded the last time the consolidate LLM judged it
|
|
767
|
-
// (judged-unchanged → no re-judge). This converts coverage from O(window) to
|
|
768
|
-
// O(changed/new) so one run can sweep the whole corpus while the LLM only
|
|
769
|
-
// sees genuinely new/changed memories. `currentHashByName` is populated for
|
|
770
|
-
// EVERY surviving memory (whether or not the cache is on) so the post-LLM
|
|
771
|
-
// recording step can upsert judged state without re-reading the files; when
|
|
772
|
-
// the cache is off it stays empty and the recording step is a no-op.
|
|
773
|
-
const judgedCacheEnabled = opts.judgedCache?.enabled !== false;
|
|
774
|
-
const currentHashByName = new Map();
|
|
775
|
-
if (judgedCacheEnabled) {
|
|
776
|
-
for (const m of memories) {
|
|
777
|
-
const h = computeMemoryContentHash(m.filePath);
|
|
778
|
-
if (h !== undefined)
|
|
779
|
-
currentHashByName.set(m.name, h);
|
|
780
|
-
}
|
|
781
|
-
let cachedMap = new Map();
|
|
782
|
-
{
|
|
783
|
-
// Use the shared state.db handle if available; open a local one otherwise.
|
|
784
|
-
const dbForJudged = sharedStateDb;
|
|
785
|
-
if (dbForJudged) {
|
|
786
|
-
try {
|
|
787
|
-
cachedMap = getConsolidationJudgedMap(dbForJudged, memories.map((m) => `memory:${m.name}`));
|
|
788
|
-
}
|
|
789
|
-
catch {
|
|
790
|
-
cachedMap = new Map();
|
|
791
|
-
}
|
|
792
|
-
}
|
|
793
|
-
else {
|
|
794
|
-
try {
|
|
795
|
-
cachedMap = withStateDb((localDb) => getConsolidationJudgedMap(localDb, memories.map((m) => `memory:${m.name}`)));
|
|
796
|
-
}
|
|
797
|
-
catch {
|
|
798
|
-
// State DB unavailable → fail open: judge the full pool this run.
|
|
799
|
-
cachedMap = new Map();
|
|
800
|
-
}
|
|
801
|
-
}
|
|
802
|
-
}
|
|
803
|
-
const beforeCount = memories.length;
|
|
804
|
-
memories = memories.filter((m) => {
|
|
805
|
-
const cur = currentHashByName.get(m.name);
|
|
806
|
-
// No readable hash → keep (fail open; let the LLM judge it).
|
|
807
|
-
if (cur === undefined)
|
|
808
|
-
return true;
|
|
809
|
-
const cached = cachedMap.get(`memory:${m.name}`);
|
|
810
|
-
// Skip only when previously judged AND content is byte-identical since.
|
|
811
|
-
return !(cached !== undefined && cached.content_hash === cur);
|
|
812
|
-
});
|
|
813
|
-
const skipped = beforeCount - memories.length;
|
|
814
|
-
perfMs.judgedCacheSkipped = skipped; // WS-5 perf telemetry
|
|
815
|
-
if (skipped > 0) {
|
|
816
|
-
warnings.push(`Judged-state cache: skipped ${skipped} memor${skipped === 1 ? "y" : "ies"} judged-unchanged (no LLM); ${memories.length} remain for judging.`);
|
|
817
|
-
}
|
|
818
|
-
if (memories.length === 0) {
|
|
819
|
-
return {
|
|
820
|
-
done: true,
|
|
821
|
-
result: makeConsolidateResult({
|
|
822
|
-
dryRun: opts.dryRun ?? false,
|
|
823
|
-
target: opts.target ?? stashDir,
|
|
824
|
-
deleted: dedupCollapsed,
|
|
825
|
-
warnings,
|
|
826
|
-
durationMs: Date.now() - startMs,
|
|
827
|
-
}),
|
|
828
|
-
};
|
|
829
|
-
}
|
|
830
|
-
}
|
|
541
|
+
// WS-5 perf telemetry: `dedupPoolSize` = memories entering the LLM pool
|
|
542
|
+
// (after incremental narrowing, before the limit cap). `llmPoolSize` =
|
|
543
|
+
// memories actually sent to the LLM. `embedMs/cacheHits/cacheMisses` =
|
|
544
|
+
// accumulated from clusterMemoriesBySimilarity.
|
|
545
|
+
const dedupPoolSize = memories.length;
|
|
831
546
|
if (opts.limit === undefined && memories.length > 150) {
|
|
832
547
|
warnings.push(`Consolidation: pool has ${memories.length} memories and no limit is set. Consider adding a limit to your consolidate config to prevent timeouts on slow LLM endpoints.`);
|
|
833
548
|
}
|
|
@@ -852,147 +567,56 @@ async function narrowConsolidationPool(opts, config, stashDir, startMs, warnings
|
|
|
852
567
|
warnings.push(`Consolidation: pool capped at ${opts.limit} of ${memories.length} memories (limit option, oldest-modified first).`);
|
|
853
568
|
memories = memories.slice(0, opts.limit);
|
|
854
569
|
}
|
|
855
|
-
return { done: false, memories,
|
|
570
|
+
return { done: false, memories, dedupPoolSize };
|
|
856
571
|
}
|
|
857
572
|
/**
|
|
858
573
|
* Pass 2 — turn the narrowed pool into an executable plan. Sizes chunks to the
|
|
859
574
|
* model context window, clusters by embedding similarity, injects the
|
|
860
575
|
* anti-collapse random fraction, applies the cold-start budget cap, runs the
|
|
861
|
-
* per-chunk LLM calls (with retry + failure-rate abort),
|
|
862
|
-
*
|
|
863
|
-
*
|
|
864
|
-
* plan-generation block.
|
|
576
|
+
* per-chunk LLM calls (with retry + failure-rate abort), and reconciles the
|
|
577
|
+
* per-chunk op arrays via {@link mergePlans}. Populates `accounting` in place.
|
|
578
|
+
* Behavior-identical to the former inlined plan-generation block.
|
|
865
579
|
*/
|
|
866
|
-
|
|
867
|
-
|
|
868
|
-
|
|
869
|
-
|
|
870
|
-
|
|
871
|
-
|
|
872
|
-
|
|
873
|
-
|
|
874
|
-
const
|
|
875
|
-
|
|
876
|
-
|
|
877
|
-
|
|
878
|
-
|
|
879
|
-
|
|
880
|
-
|
|
881
|
-
|
|
882
|
-
|
|
883
|
-
// keep it fixed and let computeSafeChunkSize vary the number of memories
|
|
884
|
-
// per chunk instead.
|
|
885
|
-
const bodyTruncation = 500;
|
|
886
|
-
const modelContextLength = llmConfig?.contextLength ?? DEFAULT_CONTEXT_LENGTH_TOKENS;
|
|
887
|
-
const chunkSize = computeSafeChunkSize(modelContextLength, bodyTruncation, opts.maxChunkSize);
|
|
888
|
-
// -- Phase A: plan generation -----------------------------------------------
|
|
889
|
-
const sourceName = opts.target ?? stashDir;
|
|
890
|
-
// WS-5: capture llmPoolSize = memories entering the LLM (after all filtering).
|
|
891
|
-
const llmPoolSize = memories.length;
|
|
892
|
-
// C-1 / #380: Pre-cluster memories by embedding similarity before chunking.
|
|
893
|
-
// This ensures that semantically similar memories land in the same LLM
|
|
894
|
-
// context window, allowing the model to detect and merge duplicates that
|
|
895
|
-
// would otherwise be split across chunks and survive indefinitely.
|
|
896
|
-
// mem0 arXiv:2504.19413, A-MEM arXiv:2502.12110.
|
|
897
|
-
// Fails open: if embeddings are unavailable or fail, original order is used.
|
|
898
|
-
const { ordered: clusteredMemories, embedTelemetry } = await clusterMemoriesBySimilarity(memories, config, sharedStateDb);
|
|
899
|
-
// WS-3b Anti-collapse step 8c: inject random (non-similar) clusters.
|
|
900
|
-
// A small fraction (default 5%) of the pool is shuffled into random positions
|
|
901
|
-
// so the pipeline isn't PURELY similarity-driven. This prevents rich-get-richer
|
|
902
|
-
// entrenchment where only the most-retrieved assets ever get consolidated.
|
|
903
|
-
// DEFAULT ON since R5 — opt out via antiCollapse.enabled: false.
|
|
904
|
-
let finalClusteredMemories = clusteredMemories;
|
|
905
|
-
{
|
|
906
|
-
const antiCollapseForCluster = getImproveProcessConfig(config, "consolidate", opts.improveProfile)?.antiCollapse ?? {};
|
|
907
|
-
if (antiCollapseForCluster.enabled !== false && clusteredMemories.length > 2) {
|
|
908
|
-
const fraction = antiCollapseForCluster.randomClusterFraction ?? 0.05;
|
|
909
|
-
const randomCount = Math.max(1, Math.floor(clusteredMemories.length * fraction));
|
|
910
|
-
// Pick `randomCount` positions to inject random (un-clustered) members.
|
|
911
|
-
// Use a seeded-ish shuffle: sort by hash of the name so it's deterministic
|
|
912
|
-
// per run but not strictly similarity-driven.
|
|
913
|
-
const shuffled = [...clusteredMemories].sort((a, b) => {
|
|
914
|
-
// Deterministic shuffle: compare sha256-ish (use name hash as proxy).
|
|
915
|
-
const ha = a.name.split("").reduce((acc, c) => ((acc << 5) - acc + c.charCodeAt(0)) | 0, 0);
|
|
916
|
-
const hb = b.name.split("").reduce((acc, c) => ((acc << 5) - acc + c.charCodeAt(0)) | 0, 0);
|
|
917
|
-
return ha - hb;
|
|
918
|
-
});
|
|
919
|
-
const randomSlice = shuffled.slice(0, randomCount);
|
|
920
|
-
const randomSet = new Set(randomSlice.map((m) => m.name));
|
|
921
|
-
// Insert random members at intervals through the clustered sequence.
|
|
922
|
-
const withRandom = [];
|
|
923
|
-
const interval = Math.max(2, Math.floor(clusteredMemories.length / randomCount));
|
|
924
|
-
let randomIdx = 0;
|
|
925
|
-
for (let i = 0; i < clusteredMemories.length; i++) {
|
|
926
|
-
const m = clusteredMemories[i];
|
|
927
|
-
if (m && !randomSet.has(m.name))
|
|
928
|
-
withRandom.push(m);
|
|
929
|
-
if (i > 0 && i % interval === 0 && randomIdx < randomSlice.length) {
|
|
930
|
-
const r = randomSlice[randomIdx++];
|
|
931
|
-
if (r)
|
|
932
|
-
withRandom.push(r);
|
|
933
|
-
}
|
|
934
|
-
}
|
|
935
|
-
// Append any remaining random members not yet inserted.
|
|
936
|
-
while (randomIdx < randomSlice.length) {
|
|
937
|
-
const r = randomSlice[randomIdx++];
|
|
938
|
-
if (r)
|
|
939
|
-
withRandom.push(r);
|
|
940
|
-
}
|
|
941
|
-
finalClusteredMemories = withRandom;
|
|
942
|
-
warnings.push(`Anti-collapse: injected ${randomCount} random (non-similarity-driven) cluster member(s) into consolidation pool (fraction=${fraction}).`);
|
|
580
|
+
/**
|
|
581
|
+
* Per-chunk judgedNoAction accounting: count memories the LLM saw inside a chunk
|
|
582
|
+
* but proposed no op for. Membership is by `memory:<name>` ref against the
|
|
583
|
+
* targets of each op (primary + secondaries for merge; ref otherwise). 2026-05-26:
|
|
584
|
+
* pre-fix this was a 78/119 (66%) silent drop in the cron run — no warning,
|
|
585
|
+
* event, or counter. See tuning investigation §Q2. Moved verbatim.
|
|
586
|
+
*/
|
|
587
|
+
function recordChunkJudgedNoAction(chunk, ops, accounting) {
|
|
588
|
+
const targetRefs = new Set();
|
|
589
|
+
for (const op of ops) {
|
|
590
|
+
if (op.op === "merge") {
|
|
591
|
+
targetRefs.add(op.primary);
|
|
592
|
+
for (const s of op.secondaries)
|
|
593
|
+
targetRefs.add(s);
|
|
594
|
+
}
|
|
595
|
+
else {
|
|
596
|
+
targetRefs.add(op.ref);
|
|
943
597
|
}
|
|
944
598
|
}
|
|
945
|
-
|
|
946
|
-
for (
|
|
947
|
-
|
|
948
|
-
|
|
949
|
-
|
|
950
|
-
|
|
951
|
-
// memories whose body would just produce a deterministic
|
|
952
|
-
// `dedup_pending_proposal` skip. Cuts ~110 wasted LLM proposals per
|
|
953
|
-
// 4h on this user's stack. See
|
|
954
|
-
// /tmp/akm-health-investigations/tuning-reasons-investigation.md §Q3.
|
|
955
|
-
const pendingProposalBodyHashes = loadPendingConsolidateProposalHashes(stashDir);
|
|
956
|
-
// ── Cold-start budget estimation ─────────────────────────────────────────────
|
|
957
|
-
// Estimate wall-clock cost BEFORE issuing any LLM calls. When a signal is
|
|
958
|
-
// provided and the estimated cost exceeds ~60% of the remaining budget we
|
|
959
|
-
// auto-reduce the pool and log the reduction so the run never starts work
|
|
960
|
-
// it cannot finish (avoiding SIGTERM mid-LLM-call).
|
|
961
|
-
//
|
|
962
|
-
// Formula: chunks.length × p90_chunk_seconds. The p90 comes from
|
|
963
|
-
// `opts.p90ChunkSecondsDefault` (caller-supplied, typically from the profile
|
|
964
|
-
// config); absent = 30 s (conservative default matching a medium local LLM).
|
|
965
|
-
//
|
|
966
|
-
// "Remaining budget" is read from a custom property on the AbortSignal if
|
|
967
|
-
// the caller (improve.ts) has attached one. Without it no auto-reduction
|
|
968
|
-
// fires but the check is still cheap to run.
|
|
969
|
-
if (chunks.length > 10 && opts.signal) {
|
|
970
|
-
const p90Chunk = opts.p90ChunkSecondsDefault ?? 30;
|
|
971
|
-
const estimatedSeconds = chunks.length * p90Chunk;
|
|
972
|
-
// remainingBudgetMs is a non-standard extension set by improve.ts when it
|
|
973
|
-
// creates the budget AbortController. Undefined = no budget information.
|
|
974
|
-
const budgetMs = opts.signal.remainingBudgetMs;
|
|
975
|
-
if (budgetMs !== undefined && budgetMs > 0) {
|
|
976
|
-
const remainingSeconds = budgetMs / 1000;
|
|
977
|
-
if (estimatedSeconds > remainingSeconds * 0.6) {
|
|
978
|
-
const safeCaps = Math.max(1, Math.floor((remainingSeconds * 0.6) / p90Chunk));
|
|
979
|
-
const removedChunks = chunks.length - safeCaps;
|
|
980
|
-
if (removedChunks > 0) {
|
|
981
|
-
const msg = `[consolidate] cold-start budget: estimated ${estimatedSeconds.toFixed(0)}s > 60% of remaining ${remainingSeconds.toFixed(0)}s; ` +
|
|
982
|
-
`reducing pool from ${chunks.length} to ${safeCaps} chunks (${removedChunks} deferred to next run).`;
|
|
983
|
-
warn(msg);
|
|
984
|
-
warnings.push(msg);
|
|
985
|
-
chunks.splice(safeCaps);
|
|
986
|
-
}
|
|
987
|
-
}
|
|
599
|
+
let chunkNoAction = 0;
|
|
600
|
+
for (const m of chunk) {
|
|
601
|
+
const memRef = conceptIdFromTypeName("memory", m.name);
|
|
602
|
+
if (!targetRefs.has(memRef)) {
|
|
603
|
+
chunkNoAction++;
|
|
604
|
+
accounting.judgedNoActionRefs.add(memRef);
|
|
988
605
|
}
|
|
989
606
|
}
|
|
990
|
-
|
|
991
|
-
|
|
992
|
-
|
|
993
|
-
|
|
994
|
-
|
|
995
|
-
|
|
607
|
+
accounting.judgedNoAction += chunkNoAction;
|
|
608
|
+
}
|
|
609
|
+
/**
|
|
610
|
+
* Per-chunk LLM judge loop — the heart of plan generation. Iterates the sized
|
|
611
|
+
* chunks, applies the budget-abort/failure-rate/all-hot guards, calls the model
|
|
612
|
+
* (with one retry), validates the returned ops, and accumulates the per-chunk
|
|
613
|
+
* judgedNoAction accounting. Extracted verbatim from `planConsolidation`: the
|
|
614
|
+
* abort-rate policy, all-hot early-exit, and the 2026-05-26 accounting invariant
|
|
615
|
+
* (`processed == actioned + judgedNoAction + Σ(skipReasons) + failedChunkMemories`)
|
|
616
|
+
* are byte-identical, and every counter-increment point is unmoved.
|
|
617
|
+
*/
|
|
618
|
+
async function judgeConsolidationChunks(args) {
|
|
619
|
+
const { chunks, opts, config, llmConfig, sourceName, bodyTruncation, pendingProposalBodyHashes, standardsContext, warnings, accounting, } = args;
|
|
996
620
|
const chunkOpsArrays = [];
|
|
997
621
|
// judgedNoAction tracks memories the LLM saw inside a chunk but proposed
|
|
998
622
|
// no op for. Computed per chunk as `chunk.length − unique(targetRefs in ops)`.
|
|
@@ -1000,14 +624,6 @@ async function planConsolidation(opts, config, stashDir, startMs, memories, warn
|
|
|
1000
624
|
// double-count fixes now live on `accounting`; every deterministic post-LLM
|
|
1001
625
|
// op rejection site calls `accounting.pushSkipReason`. See
|
|
1002
626
|
// `/tmp/akm-health-investigations/tuning-reasons-investigation.md` §Q2.
|
|
1003
|
-
//
|
|
1004
|
-
// Judged-state cache (#581): coarse outcome per memory NAME the LLM actually
|
|
1005
|
-
// judged in a successfully-parsed chunk this run. "actioned" = an op targeted
|
|
1006
|
-
// it; "no_action" = the LLM saw it and proposed nothing. Populated only when
|
|
1007
|
-
// the cache is enabled (otherwise it stays empty and the post-loop recording
|
|
1008
|
-
// step is a no-op). Memories in failed/aborted chunks are NOT recorded, so a
|
|
1009
|
-
// transient LLM failure never poisons the cache into skipping them next run.
|
|
1010
|
-
const judgedOutcomeByName = new Map();
|
|
1011
627
|
// C-6 / #392: Replace two-consecutive-failures abort with failure-rate threshold.
|
|
1012
628
|
// Consecutive-count policies are brittle against transient LM Studio reloads:
|
|
1013
629
|
// two transient failures abort the run even though the next chunk would succeed.
|
|
@@ -1061,7 +677,7 @@ async function planConsolidation(opts, config, stashDir, startMs, memories, warn
|
|
|
1061
677
|
// LLM-failure-rate abort policy — no request was attempted.
|
|
1062
678
|
if (chunk.length > 0 && chunk.every((m) => isHotCapturedMemory(m.filePath))) {
|
|
1063
679
|
for (const m of chunk)
|
|
1064
|
-
accounting.judgedNoActionRefs.add(
|
|
680
|
+
accounting.judgedNoActionRefs.add(conceptIdFromTypeName("memory", m.name));
|
|
1065
681
|
accounting.judgedNoAction += chunk.length;
|
|
1066
682
|
warn(`[consolidate] chunk ${chunkIdx + 1}/${chunks.length}: all ${chunk.length} memories are captureMode: hot — skipping LLM (judged no-action).`);
|
|
1067
683
|
continue;
|
|
@@ -1074,20 +690,34 @@ async function planConsolidation(opts, config, stashDir, startMs, memories, warn
|
|
|
1074
690
|
// asset-writers-investigation §5): providers with `supportsJsonSchema: true`
|
|
1075
691
|
// enforce the shape upstream; others fall through to
|
|
1076
692
|
// `parseEmbeddedJsonResponse` on the response side.
|
|
1077
|
-
const callChunkLlm =
|
|
693
|
+
const callChunkLlm = async (fallbackError) => {
|
|
694
|
+
// The gate runs with enabled:true (always open), so this guard is
|
|
695
|
+
// exactly the envelope the gated fn used to return first thing.
|
|
1078
696
|
if (!llmConfig)
|
|
1079
697
|
return { ok: false, error: "No LLM configured for consolidation" };
|
|
1080
|
-
|
|
1081
|
-
|
|
698
|
+
return callStructured({
|
|
699
|
+
feature: "memory_consolidation",
|
|
700
|
+
akmConfig: config,
|
|
701
|
+
enabled: true,
|
|
702
|
+
config: llmConfig,
|
|
703
|
+
messages: [
|
|
1082
704
|
{ role: "system", content: CONSOLIDATE_SYSTEM_PROMPT },
|
|
1083
705
|
{ role: "user", content: userPrompt },
|
|
1084
|
-
],
|
|
1085
|
-
|
|
1086
|
-
|
|
1087
|
-
|
|
1088
|
-
|
|
1089
|
-
|
|
1090
|
-
|
|
706
|
+
],
|
|
707
|
+
request: {
|
|
708
|
+
responseSchema: CONSOLIDATE_PLAN_JSON_SCHEMA,
|
|
709
|
+
enableThinking: false,
|
|
710
|
+
timeoutMs: llmConfig.timeoutMs,
|
|
711
|
+
signal: opts.signal,
|
|
712
|
+
},
|
|
713
|
+
parse: (raw) => ({ ok: true, content: raw ?? "" }),
|
|
714
|
+
// A transport throw was caught INSIDE the gated fn and returned as an
|
|
715
|
+
// {ok:false} envelope (never reaching the gate's fallback); onError
|
|
716
|
+
// reproduces that. The fallback fires only on wrapper timeout.
|
|
717
|
+
onError: (_cls, e) => ({ ok: false, error: String(e) }),
|
|
718
|
+
fallback: { ok: false, error: fallbackError },
|
|
719
|
+
});
|
|
720
|
+
};
|
|
1091
721
|
let raw = await callChunkLlm(`chunk ${chunkIdx + 1} failed`);
|
|
1092
722
|
if (!raw.ok) {
|
|
1093
723
|
// Single retry with 2s backoff before recording chunk as lost.
|
|
@@ -1109,9 +739,12 @@ async function planConsolidation(opts, config, stashDir, startMs, memories, warn
|
|
|
1109
739
|
}
|
|
1110
740
|
raw = retry;
|
|
1111
741
|
}
|
|
1112
|
-
|
|
742
|
+
// C9 action 1: AKM_DEBUG_LLM was a separate, undocumented env var for this
|
|
743
|
+
// one diagnostic; folded into the standard AKM_VERBOSE gate (warnVerbose)
|
|
744
|
+
// rather than kept as its own toggle.
|
|
745
|
+
{
|
|
1113
746
|
const preview = (raw.content ?? "").slice(0, 500);
|
|
1114
|
-
|
|
747
|
+
warnVerbose(`[akm:consolidate] chunk ${chunkIdx + 1} raw response (first 500 chars): ${preview}`);
|
|
1115
748
|
}
|
|
1116
749
|
const parsed = parseEmbeddedJsonResponse(raw.content);
|
|
1117
750
|
if (!parsed || !Array.isArray(parsed.operations)) {
|
|
@@ -1141,194 +774,179 @@ async function planConsolidation(opts, config, stashDir, startMs, memories, warn
|
|
|
1141
774
|
warnings.push(w);
|
|
1142
775
|
}
|
|
1143
776
|
}
|
|
1144
|
-
|
|
1145
|
-
// op for. Membership is by `memory:<name>` ref against the targets of
|
|
1146
|
-
// each op (primary + secondaries for merge; ref otherwise). 2026-05-26:
|
|
1147
|
-
// pre-fix this was a 78/119 (66%) silent drop in the cron run — no
|
|
1148
|
-
// warning, event, or counter. See tuning investigation §Q2.
|
|
1149
|
-
const targetRefs = new Set();
|
|
1150
|
-
for (const op of ops) {
|
|
1151
|
-
if (op.op === "merge") {
|
|
1152
|
-
targetRefs.add(op.primary);
|
|
1153
|
-
for (const s of op.secondaries)
|
|
1154
|
-
targetRefs.add(s);
|
|
1155
|
-
}
|
|
1156
|
-
else {
|
|
1157
|
-
targetRefs.add(op.ref);
|
|
1158
|
-
}
|
|
1159
|
-
}
|
|
1160
|
-
let chunkNoAction = 0;
|
|
1161
|
-
for (const m of chunk) {
|
|
1162
|
-
const memRef = `memory:${m.name}`;
|
|
1163
|
-
if (!targetRefs.has(memRef)) {
|
|
1164
|
-
chunkNoAction++;
|
|
1165
|
-
accounting.judgedNoActionRefs.add(memRef);
|
|
1166
|
-
// Judged-state cache (#581): the LLM saw this memory and proposed
|
|
1167
|
-
// nothing → record judged-unchanged so the next run can skip it.
|
|
1168
|
-
if (judgedCacheEnabled)
|
|
1169
|
-
judgedOutcomeByName.set(m.name, "no_action");
|
|
1170
|
-
}
|
|
1171
|
-
else if (judgedCacheEnabled) {
|
|
1172
|
-
// An op targeted this memory → it was judged + actioned.
|
|
1173
|
-
judgedOutcomeByName.set(m.name, "actioned");
|
|
1174
|
-
}
|
|
1175
|
-
}
|
|
1176
|
-
accounting.judgedNoAction += chunkNoAction;
|
|
777
|
+
recordChunkJudgedNoAction(chunk, ops, accounting);
|
|
1177
778
|
chunkOpsArrays.push(ops);
|
|
1178
779
|
}
|
|
1179
|
-
|
|
1180
|
-
|
|
1181
|
-
|
|
1182
|
-
//
|
|
1183
|
-
//
|
|
1184
|
-
//
|
|
1185
|
-
//
|
|
1186
|
-
|
|
1187
|
-
|
|
1188
|
-
|
|
1189
|
-
|
|
1190
|
-
|
|
1191
|
-
|
|
1192
|
-
|
|
1193
|
-
|
|
1194
|
-
|
|
1195
|
-
|
|
1196
|
-
|
|
1197
|
-
|
|
1198
|
-
|
|
1199
|
-
|
|
1200
|
-
|
|
1201
|
-
|
|
1202
|
-
|
|
1203
|
-
|
|
1204
|
-
|
|
1205
|
-
|
|
1206
|
-
|
|
1207
|
-
|
|
1208
|
-
|
|
1209
|
-
|
|
1210
|
-
|
|
1211
|
-
|
|
1212
|
-
|
|
780
|
+
return chunkOpsArrays;
|
|
781
|
+
}
|
|
782
|
+
async function planConsolidation(opts, config, stashDir, _startMs, memories, warnings, sharedStateDb, accounting) {
|
|
783
|
+
// Consolidation always uses the HTTP LLM client directly — never the agent
|
|
784
|
+
// CLI. The agent CLI is for interactive agent sessions (reflect, propose);
|
|
785
|
+
// structured JSON generation works better and faster via HTTP.
|
|
786
|
+
//
|
|
787
|
+
// Improve supplies a frozen connection; standalone consolidate resolves its
|
|
788
|
+
// selected strategy/default engine here.
|
|
789
|
+
const llmConfig = Object.hasOwn(opts, "llmConfig")
|
|
790
|
+
? (opts.llmConfig ?? undefined)
|
|
791
|
+
: resolveConsolidateLlmConfig(config, opts.improveProfile);
|
|
792
|
+
// Chunk sizing: derive a safe chunk size from the configured model context
|
|
793
|
+
// window so that the full prompt (system prompt + chunk user prompt) never
|
|
794
|
+
// exceeds the model's n_ctx limit. When no context length is configured we
|
|
795
|
+
// fall back to DEFAULT_CONTEXT_LENGTH_TOKENS (8 000) which is conservative
|
|
796
|
+
// enough for most 8K–16K local models.
|
|
797
|
+
//
|
|
798
|
+
// bodyTruncation caps the body excerpt included per memory in the prompt.
|
|
799
|
+
// Reducing it further than 500 chars degrades consolidation quality, so we
|
|
800
|
+
// keep it fixed and let computeSafeChunkSize vary the number of memories
|
|
801
|
+
// per chunk instead.
|
|
802
|
+
const bodyTruncation = 500;
|
|
803
|
+
const modelContextLength = llmConfig?.contextLength ?? DEFAULT_CONTEXT_LENGTH_TOKENS;
|
|
804
|
+
const chunkSize = computeSafeChunkSize(modelContextLength, bodyTruncation, opts.maxChunkSize);
|
|
805
|
+
// -- Phase A: plan generation -----------------------------------------------
|
|
806
|
+
const sourceName = opts.target ?? stashDir;
|
|
807
|
+
let budgetedMemories = memories;
|
|
808
|
+
if (opts.signal) {
|
|
809
|
+
const budgetMs = opts.signal.remainingBudgetMs;
|
|
810
|
+
if (budgetMs !== undefined) {
|
|
811
|
+
const p90Chunk = opts.p90ChunkSecondsDefault ?? 30;
|
|
812
|
+
const safeChunks = Math.max(0, Math.floor((Math.max(0, budgetMs) / 1000 / p90Chunk) * 0.6));
|
|
813
|
+
const cap = safeChunks * chunkSize;
|
|
814
|
+
if (cap < memories.length) {
|
|
815
|
+
budgetedMemories = memories
|
|
816
|
+
.map((entry) => {
|
|
817
|
+
let mtimeMs = 0;
|
|
818
|
+
try {
|
|
819
|
+
mtimeMs = fs.statSync(entry.filePath).mtimeMs;
|
|
820
|
+
}
|
|
821
|
+
catch {
|
|
822
|
+
// Missing files sort first and are filtered by the existing guards.
|
|
823
|
+
}
|
|
824
|
+
return { entry, mtimeMs };
|
|
825
|
+
})
|
|
826
|
+
.sort((a, b) => a.mtimeMs - b.mtimeMs || a.entry.name.localeCompare(b.entry.name))
|
|
827
|
+
.map(({ entry }) => entry)
|
|
828
|
+
.slice(0, cap);
|
|
829
|
+
const msg = `[consolidate] cold-start budget: reducing pool from ${memories.length} to ${budgetedMemories.length} memories (${safeChunks} safe chunks; remainder deferred).`;
|
|
830
|
+
warn(msg);
|
|
831
|
+
warnings.push(msg);
|
|
1213
832
|
}
|
|
1214
|
-
|
|
1215
|
-
|
|
833
|
+
}
|
|
834
|
+
}
|
|
835
|
+
// WS-5: capture llmPoolSize after every pre-LLM cap.
|
|
836
|
+
const llmPoolSize = budgetedMemories.length;
|
|
837
|
+
// C-1 / #380: Pre-cluster memories by embedding similarity before chunking.
|
|
838
|
+
// This ensures that semantically similar memories land in the same LLM
|
|
839
|
+
// context window, allowing the model to detect and merge duplicates that
|
|
840
|
+
// would otherwise be split across chunks and survive indefinitely.
|
|
841
|
+
// mem0 arXiv:2504.19413, A-MEM arXiv:2502.12110.
|
|
842
|
+
// Fails open: if embeddings are unavailable or fail, original order is used.
|
|
843
|
+
const { ordered: clusteredMemories, embedTelemetry } = await clusterMemoriesBySimilarity(budgetedMemories, config, sharedStateDb, opts.signal);
|
|
844
|
+
// WS-3b Anti-collapse step 8c: inject random (non-similar) clusters.
|
|
845
|
+
// A small fraction (default 5%) of the pool is shuffled into random positions
|
|
846
|
+
// so the pipeline isn't PURELY similarity-driven. This prevents rich-get-richer
|
|
847
|
+
// entrenchment where only the most-retrieved assets ever get consolidated.
|
|
848
|
+
// DEFAULT ON since R5 — opt out via antiCollapse.enabled: false.
|
|
849
|
+
let finalClusteredMemories = clusteredMemories;
|
|
850
|
+
{
|
|
851
|
+
const antiCollapseForCluster = getImproveProcessConfig("consolidate", opts.improveProfile)?.antiCollapse ??
|
|
852
|
+
{};
|
|
853
|
+
if (antiCollapseForCluster.enabled !== false && clusteredMemories.length > 2) {
|
|
854
|
+
const fraction = antiCollapseForCluster.randomClusterFraction ?? 0.05;
|
|
855
|
+
const randomCount = Math.max(1, Math.floor(clusteredMemories.length * fraction));
|
|
856
|
+
// Pick `randomCount` positions to inject random (un-clustered) members.
|
|
857
|
+
// Use a seeded-ish shuffle: sort by hash of the name so it's deterministic
|
|
858
|
+
// per run but not strictly similarity-driven.
|
|
859
|
+
const shuffled = [...clusteredMemories].sort((a, b) => {
|
|
860
|
+
// Deterministic shuffle: compare sha256-ish (use name hash as proxy).
|
|
861
|
+
const ha = a.name.split("").reduce((acc, c) => ((acc << 5) - acc + c.charCodeAt(0)) | 0, 0);
|
|
862
|
+
const hb = b.name.split("").reduce((acc, c) => ((acc << 5) - acc + c.charCodeAt(0)) | 0, 0);
|
|
863
|
+
return ha - hb;
|
|
864
|
+
});
|
|
865
|
+
const randomSlice = shuffled.slice(0, randomCount);
|
|
866
|
+
const randomSet = new Set(randomSlice.map((m) => m.name));
|
|
867
|
+
// Insert random members at intervals through the clustered sequence.
|
|
868
|
+
const withRandom = [];
|
|
869
|
+
const interval = Math.max(2, Math.floor(clusteredMemories.length / randomCount));
|
|
870
|
+
let randomIdx = 0;
|
|
871
|
+
for (let i = 0; i < clusteredMemories.length; i++) {
|
|
872
|
+
const m = clusteredMemories[i];
|
|
873
|
+
if (m && !randomSet.has(m.name))
|
|
874
|
+
withRandom.push(m);
|
|
875
|
+
if (i > 0 && i % interval === 0 && randomIdx < randomSlice.length) {
|
|
876
|
+
const r = randomSlice[randomIdx++];
|
|
877
|
+
if (r)
|
|
878
|
+
withRandom.push(r);
|
|
879
|
+
}
|
|
880
|
+
}
|
|
881
|
+
// Append any remaining random members not yet inserted.
|
|
882
|
+
while (randomIdx < randomSlice.length) {
|
|
883
|
+
const r = randomSlice[randomIdx++];
|
|
884
|
+
if (r)
|
|
885
|
+
withRandom.push(r);
|
|
1216
886
|
}
|
|
887
|
+
finalClusteredMemories = withRandom;
|
|
888
|
+
warnings.push(`Anti-collapse: injected ${randomCount} random (non-similarity-driven) cluster member(s) into consolidation pool (fraction=${fraction}).`);
|
|
1217
889
|
}
|
|
1218
890
|
}
|
|
891
|
+
const chunks = [];
|
|
892
|
+
for (let i = 0; i < finalClusteredMemories.length; i += chunkSize) {
|
|
893
|
+
chunks.push(finalClusteredMemories.slice(i, i + chunkSize));
|
|
894
|
+
}
|
|
895
|
+
// 2026-05-27 prompt-context fix: precompute body-hashes of pending
|
|
896
|
+
// consolidate proposals once, so the per-chunk prompt can annotate
|
|
897
|
+
// memories whose body would just produce a deterministic
|
|
898
|
+
// `dedup_pending_proposal` skip. Cuts ~110 wasted LLM proposals per
|
|
899
|
+
// 4h on this user's stack. See
|
|
900
|
+
// /tmp/akm-health-investigations/tuning-reasons-investigation.md §Q3.
|
|
901
|
+
const pendingProposalBodyHashes = loadPendingConsolidateProposalHashes(stashDir);
|
|
902
|
+
warn(`[consolidate] ${budgetedMemories.length} memories / ${chunks.length} chunk(s) / chunk_size=${chunkSize}` +
|
|
903
|
+
` / pending-proposal hashes: ${pendingProposalBodyHashes.size}`);
|
|
904
|
+
// Consolidate output merges memories (non-wiki) → stash authoring standards.
|
|
905
|
+
// Resolved ONCE per run and passed to each chunk prompt (facts not re-read
|
|
906
|
+
// per chunk).
|
|
907
|
+
const standardsContext = resolveStandardsContext("memories/_consolidated", stashDir);
|
|
908
|
+
const chunkOpsArrays = await judgeConsolidationChunks({
|
|
909
|
+
chunks,
|
|
910
|
+
opts,
|
|
911
|
+
config,
|
|
912
|
+
llmConfig,
|
|
913
|
+
sourceName,
|
|
914
|
+
bodyTruncation,
|
|
915
|
+
pendingProposalBodyHashes,
|
|
916
|
+
standardsContext,
|
|
917
|
+
warnings,
|
|
918
|
+
accounting,
|
|
919
|
+
});
|
|
1219
920
|
// Build the known-refs set from the already-filtered memory pool so
|
|
1220
921
|
// mergePlans() can reject LLM-hallucinated primary refs before execution.
|
|
1221
|
-
const knownRefs = new Set(
|
|
922
|
+
const knownRefs = new Set(budgetedMemories.map((m) => conceptIdFromTypeName("memory", m.name)));
|
|
1222
923
|
const { ops: allOps, warnings: mergeWarnings } = mergePlans(chunkOpsArrays, knownRefs);
|
|
1223
924
|
warnings.push(...mergeWarnings);
|
|
1224
|
-
return {
|
|
1225
|
-
|
|
1226
|
-
|
|
1227
|
-
|
|
1228
|
-
|
|
1229
|
-
|
|
1230
|
-
|
|
1231
|
-
* write block. Never invoked on the dry-run or aborted-confirm paths.
|
|
1232
|
-
*/
|
|
1233
|
-
async function applyConsolidationPlan(config, stashDir, sourceRun, memories, warnings, allOps, accounting, dedupCollapsed, activeProfile) {
|
|
1234
|
-
// -- Phase B + writes -------------------------------------------------------
|
|
1235
|
-
const target = resolveWriteTarget(config);
|
|
1236
|
-
const timestamp = timestampForFilename();
|
|
1237
|
-
const backupDir = getBackupDir(stashDir, timestamp);
|
|
1238
|
-
// Write journal before any mutations
|
|
1239
|
-
writeJournal(stashDir, allOps, timestamp);
|
|
1240
|
-
const counts = {
|
|
1241
|
-
merged: 0,
|
|
1242
|
-
deleted: 0,
|
|
1243
|
-
contradicted: 0, // C-3 / #382: count of contradiction edges written
|
|
1244
|
-
mergeFloorViolations: 0, // R5 §4.2: advisory merge-information-floor failures
|
|
1245
|
-
mergedSecondaries: 0,
|
|
1246
|
-
};
|
|
1247
|
-
const promoted = [];
|
|
1248
|
-
// Within-run dedup: track source refs for which a promote proposal was
|
|
1249
|
-
// already created this run. The LLM can return multiple promote ops for
|
|
1250
|
-
// different source memories that happen to have identical content (all are
|
|
1251
|
-
// duplicate memories), so we also need a content-hash guard below.
|
|
1252
|
-
const promotedSourceRefs = new Set();
|
|
1253
|
-
// Build a lookup map: ref → MemoryEntry
|
|
1254
|
-
const memoryByRef = new Map();
|
|
1255
|
-
for (const m of memories) {
|
|
1256
|
-
memoryByRef.set(`memory:${m.name}`, m);
|
|
1257
|
-
}
|
|
1258
|
-
const opCtx = {
|
|
1259
|
-
config,
|
|
1260
|
-
improveProfile: activeProfile,
|
|
1261
|
-
stashDir,
|
|
1262
|
-
sourceRun,
|
|
1263
|
-
target,
|
|
1264
|
-
backupDir,
|
|
1265
|
-
memoryByRef,
|
|
1266
|
-
promoted,
|
|
1267
|
-
promotedSourceRefs,
|
|
1268
|
-
warnings,
|
|
1269
|
-
counts,
|
|
1270
|
-
pushSkipReason: accounting.pushSkipReason,
|
|
925
|
+
return {
|
|
926
|
+
allOps,
|
|
927
|
+
totalChunks: chunks.length,
|
|
928
|
+
llmPoolSize,
|
|
929
|
+
deferredMemories: memories.length - budgetedMemories.length,
|
|
930
|
+
embedTelemetry,
|
|
931
|
+
sourceName,
|
|
1271
932
|
};
|
|
1272
|
-
// Thin dispatch over the op discriminator — each branch is now an isolated,
|
|
1273
|
-
// independently-testable handler that mutates `opCtx`.
|
|
1274
|
-
for (let opIndex = 0; opIndex < allOps.length; opIndex++) {
|
|
1275
|
-
const op = allOps[opIndex];
|
|
1276
|
-
const opDisplayRef = op.op === "merge" ? op.primary : op.op === "contradict" ? `${op.ref} ↔ ${op.contradictedByRef}` : op.ref;
|
|
1277
|
-
warn(`[consolidate] ${opIndex + 1}/${allOps.length} ${op.op} ${opDisplayRef}`);
|
|
1278
|
-
switch (op.op) {
|
|
1279
|
-
case "merge":
|
|
1280
|
-
await handleMergeOp(op, opIndex, opCtx);
|
|
1281
|
-
break;
|
|
1282
|
-
case "delete":
|
|
1283
|
-
await handleDeleteOp(op, opIndex, opCtx);
|
|
1284
|
-
break;
|
|
1285
|
-
case "promote":
|
|
1286
|
-
await handlePromoteOp(op, opCtx);
|
|
1287
|
-
break;
|
|
1288
|
-
case "contradict":
|
|
1289
|
-
await handleContradictOp(op, opCtx);
|
|
1290
|
-
break;
|
|
1291
|
-
}
|
|
1292
|
-
}
|
|
1293
|
-
const { merged, deleted, contradicted, mergeFloorViolations, mergedSecondaries } = counts;
|
|
1294
|
-
// 0.9.0 (issue #507): batch-at-boundary commit. The merge/delete loop above
|
|
1295
|
-
// wrote one merged primary and deleted N secondaries to the resolved target
|
|
1296
|
-
// with NO per-asset commit. If the target is a writable git source and any
|
|
1297
|
-
// asset was mutated, commit the whole batch ONCE here (stages .akm/ +
|
|
1298
|
-
// siblings together). No-op for filesystem/primary-stash targets.
|
|
1299
|
-
if (merged > 0 || deleted > 0) {
|
|
1300
|
-
commitWriteTargetBoundary(target, `Consolidate: ${merged} merged, ${deleted} removed`);
|
|
1301
|
-
}
|
|
1302
|
-
cleanupJournal(stashDir, timestamp);
|
|
1303
|
-
// [signoff 2026-06-15] TTL archive cleanup machinery RETIRED (WS-3a).
|
|
1304
|
-
// The elaborate archiveRetentionDays / archive-dir scan existed only to satisfy
|
|
1305
|
-
// the old irrecoverability constraint. Stashes are now git-backed, so git
|
|
1306
|
-
// history is the recovery path — no bespoke archive TTL needed. Any files in
|
|
1307
|
-
// .akm/archive/ will stay there harmlessly until the operator prunes them with
|
|
1308
|
-
// `git rm` or `find .akm/archive -mtime +90 -delete`. Changed N files this
|
|
1309
|
-
// run; recover any via `git show <sha>:<path>` or `git restore <path>`.
|
|
1310
|
-
if (merged > 0 || deleted > 0 || dedupCollapsed > 0) {
|
|
1311
|
-
const totalChanged = merged + deleted + dedupCollapsed;
|
|
1312
|
-
warnings.push(`Changed ${totalChanged} file(s) this run. Recover any via git if needed (git history is the backstop).`);
|
|
1313
|
-
}
|
|
1314
|
-
return { merged, deleted, contradicted, mergeFloorViolations, mergedSecondaries, promoted };
|
|
1315
933
|
}
|
|
1316
|
-
async function akmConsolidateInner(opts, config, stashDir, startMs,
|
|
934
|
+
async function akmConsolidateInner(opts, config, stashDir, startMs, warnings, sharedStateDb) {
|
|
1317
935
|
// -- Pass 1: narrow the memory pool (may early-return an envelope) ----------
|
|
1318
|
-
const narrowed = await narrowConsolidationPool(opts,
|
|
936
|
+
const narrowed = await narrowConsolidationPool(opts, stashDir, startMs, warnings);
|
|
1319
937
|
if (narrowed.done)
|
|
1320
938
|
return narrowed.result;
|
|
1321
|
-
const { memories,
|
|
939
|
+
const { memories, dedupPoolSize } = narrowed;
|
|
1322
940
|
// -- Pass 2: build the LLM plan (populates the shared accounting counters) ---
|
|
1323
941
|
const accounting = createConsolidateAccounting();
|
|
1324
|
-
const { allOps, totalChunks, llmPoolSize,
|
|
942
|
+
const { allOps, totalChunks, llmPoolSize, deferredMemories, embedTelemetry, sourceName } = await planConsolidation(opts, config, stashDir, startMs, memories, warnings, sharedStateDb, accounting);
|
|
1325
943
|
// -- Dry-run: show AI plan without executing any writes --------------------
|
|
1326
944
|
if (opts.dryRun) {
|
|
1327
945
|
return makeConsolidateResult({
|
|
1328
946
|
dryRun: true,
|
|
1329
947
|
previewOnly: true,
|
|
1330
948
|
target: sourceName,
|
|
1331
|
-
processed:
|
|
949
|
+
processed: llmPoolSize,
|
|
1332
950
|
failedChunks: accounting.totalChunksFailed,
|
|
1333
951
|
totalChunks,
|
|
1334
952
|
judgedNoAction: accounting.judgedNoAction,
|
|
@@ -1337,407 +955,64 @@ async function akmConsolidateInner(opts, config, stashDir, startMs, sourceRun, w
|
|
|
1337
955
|
// provably still 0 here (it only increments in the op-execution loop).
|
|
1338
956
|
mergedSecondaries: 0,
|
|
1339
957
|
failedChunkMemories: accounting.failedChunkMemories,
|
|
958
|
+
deferredMemories,
|
|
1340
959
|
planned: allOps,
|
|
1341
960
|
warnings,
|
|
1342
961
|
durationMs: Date.now() - startMs,
|
|
1343
962
|
});
|
|
1344
963
|
}
|
|
1345
964
|
warn(`[consolidate] plan: ${allOps.length} operation(s)`);
|
|
1346
|
-
//
|
|
1347
|
-
|
|
1348
|
-
|
|
1349
|
-
|
|
1350
|
-
|
|
1351
|
-
|
|
1352
|
-
|
|
1353
|
-
|
|
1354
|
-
|
|
1355
|
-
|
|
1356
|
-
|
|
1357
|
-
|
|
1358
|
-
|
|
1359
|
-
|
|
1360
|
-
|
|
1361
|
-
|
|
1362
|
-
|
|
1363
|
-
|
|
1364
|
-
|
|
1365
|
-
|
|
1366
|
-
|
|
1367
|
-
|
|
1368
|
-
|
|
1369
|
-
|
|
1370
|
-
judgedNoAction: accounting.judgedNoAction,
|
|
1371
|
-
skipReasons: accounting.skipReasons,
|
|
1372
|
-
// No merge executed on the abort path — mergedSecondaries is still 0.
|
|
1373
|
-
mergedSecondaries: 0,
|
|
1374
|
-
failedChunkMemories: accounting.failedChunkMemories,
|
|
1375
|
-
planned: allOps,
|
|
1376
|
-
warnings: [...warnings, nonInteractive ? "Non-interactive context: skipped apply." : "Aborted by user."],
|
|
1377
|
-
durationMs: Date.now() - startMs,
|
|
1378
|
-
});
|
|
1379
|
-
}
|
|
1380
|
-
}
|
|
965
|
+
// Destructive operations remain advisory. Promote is safe to execute because
|
|
966
|
+
// it emits a reviewable proposal rather than mutating an asset.
|
|
967
|
+
const promoted = [];
|
|
968
|
+
const promotionFailures = { count: 0 };
|
|
969
|
+
const memoryByRef = new Map(memories.map((memory) => [conceptIdFromTypeName("memory", memory.name), memory]));
|
|
970
|
+
const promoteContext = {
|
|
971
|
+
config,
|
|
972
|
+
stashDir,
|
|
973
|
+
sourceRun: opts.sourceRun ?? `consolidate-${startMs}`,
|
|
974
|
+
proposalsCtx: opts.proposalsCtx,
|
|
975
|
+
target: opts.writeTarget,
|
|
976
|
+
memoryByRef,
|
|
977
|
+
promoted,
|
|
978
|
+
promotedSourceRefs: new Set(),
|
|
979
|
+
promotionFailures,
|
|
980
|
+
warnings,
|
|
981
|
+
pushSkipReason: accounting.pushSkipReason,
|
|
982
|
+
llmConfig: Object.hasOwn(opts, "llmConfig")
|
|
983
|
+
? (opts.llmConfig ?? null)
|
|
984
|
+
: (resolveConsolidateLlmConfig(config, opts.improveProfile) ?? null),
|
|
985
|
+
};
|
|
986
|
+
for (const op of allOps) {
|
|
987
|
+
if (op.op === "promote")
|
|
988
|
+
await emitPromotionProposal(op, promoteContext);
|
|
1381
989
|
}
|
|
1382
|
-
|
|
1383
|
-
const { merged, deleted, contradicted, mergeFloorViolations, mergedSecondaries, promoted } = await applyConsolidationPlan(config, stashDir, sourceRun, memories, warnings, allOps, accounting, dedupCollapsed, opts.improveProfile);
|
|
1384
|
-
const runDurationMs = Date.now() - startMs;
|
|
1385
|
-
const budgetFraction = opts.runBudgetMs !== undefined && opts.runBudgetMs > 0 ? runDurationMs / opts.runBudgetMs : undefined;
|
|
1386
|
-
return {
|
|
1387
|
-
schemaVersion: 1,
|
|
1388
|
-
ok: true,
|
|
1389
|
-
shape: "consolidate-result",
|
|
1390
|
-
dryRun: false,
|
|
1391
|
-
previewOnly: false,
|
|
990
|
+
return makeConsolidateResult({
|
|
1392
991
|
target: sourceName,
|
|
1393
|
-
processed:
|
|
1394
|
-
merged,
|
|
1395
|
-
// #617: fold the deterministic dedup pre-pass collapses into the reported
|
|
1396
|
-
// deleted count. Each collapse removed exactly one variant file with NO
|
|
1397
|
-
// LLM call before the LLM pass ran on the pruned pool.
|
|
1398
|
-
deleted: deleted + dedupCollapsed,
|
|
1399
|
-
promoted,
|
|
1400
|
-
contradicted,
|
|
1401
|
-
mergeFloorViolations,
|
|
992
|
+
processed: llmPoolSize,
|
|
1402
993
|
failedChunks: accounting.totalChunksFailed,
|
|
1403
994
|
totalChunks,
|
|
1404
995
|
judgedNoAction: accounting.judgedNoAction,
|
|
1405
996
|
skipReasons: accounting.skipReasons,
|
|
1406
|
-
mergedSecondaries,
|
|
997
|
+
mergedSecondaries: 0,
|
|
1407
998
|
failedChunkMemories: accounting.failedChunkMemories,
|
|
999
|
+
deferredMemories,
|
|
1000
|
+
promoted,
|
|
1001
|
+
failedPromotions: promotionFailures.count,
|
|
1002
|
+
planned: allOps,
|
|
1408
1003
|
warnings,
|
|
1409
|
-
durationMs:
|
|
1004
|
+
durationMs: Date.now() - startMs,
|
|
1410
1005
|
perfTelemetry: {
|
|
1411
|
-
dedupPoolSize
|
|
1006
|
+
dedupPoolSize,
|
|
1412
1007
|
llmPoolSize,
|
|
1413
|
-
judgedCacheSkipped: perfMs.judgedCacheSkipped,
|
|
1414
1008
|
embedMs: embedTelemetry.embedMs,
|
|
1415
1009
|
embedCacheHits: embedTelemetry.cacheHits,
|
|
1416
1010
|
embedCacheMisses: embedTelemetry.cacheMisses,
|
|
1417
|
-
...(budgetFraction !== undefined ? { estimatedBudgetFractionUsed: budgetFraction } : {}),
|
|
1418
1011
|
},
|
|
1419
|
-
};
|
|
1420
|
-
}
|
|
1421
|
-
/** Execute one `merge` op (behavior-identical to the former inlined branch). */
|
|
1422
|
-
export async function handleMergeOp(op, opIndex, ctx) {
|
|
1423
|
-
const { config, stashDir, target, backupDir, memoryByRef, warnings, pushSkipReason, counts } = ctx;
|
|
1424
|
-
// Accounting helper: emit a per-participant skipReason for failed
|
|
1425
|
-
// merges so primary + every loaded-memory secondary land in the
|
|
1426
|
-
// structured skip histogram. Pre-2026-05-26 only the primary was
|
|
1427
|
-
// counted (1 skipReason per failed merge), leaving N secondaries
|
|
1428
|
-
// unaccounted for in the `processed == actioned + noAction + Σskips`
|
|
1429
|
-
// invariant — the source of the 4–11 silent leaks per run.
|
|
1430
|
-
const emitMergeFailureSkips = (reason) => {
|
|
1431
|
-
if (memoryByRef.has(op.primary))
|
|
1432
|
-
pushSkipReason("merge", op.primary, reason);
|
|
1433
|
-
for (const secRef of op.secondaries) {
|
|
1434
|
-
if (memoryByRef.has(secRef))
|
|
1435
|
-
pushSkipReason("merge", secRef, reason);
|
|
1436
|
-
}
|
|
1437
|
-
};
|
|
1438
|
-
const primaryEntry = memoryByRef.get(op.primary);
|
|
1439
|
-
if (!primaryEntry) {
|
|
1440
|
-
// This fires when a prior op in the same run consumed this ref as a
|
|
1441
|
-
// secondary and Fix-A pruned it from memoryByRef. It should NOT fire
|
|
1442
|
-
// for hallucinated primaries (those are dropped by mergePlans() before
|
|
1443
|
-
// reaching here). If this counter is non-zero, suspect an intra-run
|
|
1444
|
-
// cross-chunk race, not a filter regression.
|
|
1445
|
-
warnings.push(`Merge: primary ${op.primary} not found in loaded memories (pruned by prior op this run) — skipping.`);
|
|
1446
|
-
emitMergeFailureSkips("merge_primary_missing");
|
|
1447
|
-
return;
|
|
1448
|
-
}
|
|
1449
|
-
// Defense-in-depth: even if the entry is in memoryByRef (pre-flight ran
|
|
1450
|
-
// before this run's own ops), the file may have been deleted by a
|
|
1451
|
-
// concurrent process or an edge case the pre-flight filter missed.
|
|
1452
|
-
if (!fs.existsSync(primaryEntry.filePath)) {
|
|
1453
|
-
warnings.push(`Merge: primary ${op.primary} file gone at execution time (stale entry) — skipping.`);
|
|
1454
|
-
emitMergeFailureSkips("merge_primary_file_gone");
|
|
1455
|
-
return;
|
|
1456
|
-
}
|
|
1457
|
-
// Phase B: generate merged content
|
|
1458
|
-
const secondaryBodies = [];
|
|
1459
|
-
for (const secRef of op.secondaries) {
|
|
1460
|
-
const secEntry = memoryByRef.get(secRef);
|
|
1461
|
-
if (!secEntry) {
|
|
1462
|
-
warnings.push(`Merge: secondary ${secRef} not found — skipping merge op.`);
|
|
1463
|
-
// No accounting impact: a missing secondary is a phantom ref and
|
|
1464
|
-
// never contributed to any chunk's targetRefs reduction. We still
|
|
1465
|
-
// continue the loop to gather the remaining valid secondaries.
|
|
1466
|
-
continue;
|
|
1467
|
-
}
|
|
1468
|
-
secondaryBodies.push(secRef);
|
|
1469
|
-
}
|
|
1470
|
-
if (secondaryBodies.length === 0) {
|
|
1471
|
-
warnings.push(`Merge: ${op.primary} has no valid secondaries — skipping.`);
|
|
1472
|
-
emitMergeFailureSkips("merge_no_valid_secondaries");
|
|
1473
|
-
return;
|
|
1474
|
-
}
|
|
1475
|
-
// Pre-flight hot guard — skip the LLM call entirely if any participant
|
|
1476
|
-
// is hot or unparseable. Without this, mixed chunks still send hot merges
|
|
1477
|
-
// to the planner which proposes them; generateMergedContent() is then
|
|
1478
|
-
// called, produces output without `description`, and the skip is
|
|
1479
|
-
// misattributed to merge_missing_description instead of the real cause.
|
|
1480
|
-
const preflightParticipants = [op.primary, ...op.secondaries];
|
|
1481
|
-
const preflightBlocked = preflightParticipants.flatMap((ref) => {
|
|
1482
|
-
const e = memoryByRef.get(ref);
|
|
1483
|
-
if (!e)
|
|
1484
|
-
return [];
|
|
1485
|
-
const verdict = consolidateGuardStatus(e.filePath);
|
|
1486
|
-
if (verdict === "hot" || verdict === "unparseable")
|
|
1487
|
-
return [{ ref, verdict }];
|
|
1488
|
-
return [];
|
|
1489
1012
|
});
|
|
1490
|
-
if (preflightBlocked.length > 0) {
|
|
1491
|
-
const detail = preflightBlocked.map((p) => `${p.ref} (${p.verdict})`).join(", ");
|
|
1492
|
-
warnings.push(`Merge: refused for ${op.primary} — ${preflightBlocked.length} participant(s) blocked by hot/unparseable frontmatter guard (pre-flight): ${detail}`);
|
|
1493
|
-
emitMergeFailureSkips("merge_participant_blocked");
|
|
1494
|
-
return;
|
|
1495
|
-
}
|
|
1496
|
-
let primaryBody = "";
|
|
1497
|
-
try {
|
|
1498
|
-
primaryBody = fs.readFileSync(primaryEntry.filePath, "utf8");
|
|
1499
|
-
}
|
|
1500
|
-
catch {
|
|
1501
|
-
warnings.push(`Merge: could not read primary ${op.primary} — skipping.`);
|
|
1502
|
-
emitMergeFailureSkips("merge_read_failed");
|
|
1503
|
-
return;
|
|
1504
|
-
}
|
|
1505
|
-
const mergeResult = await generateMergedContent(config, op.primary, primaryBody, op.secondaries, memoryByRef, ctx.improveProfile);
|
|
1506
|
-
if ("error" in mergeResult) {
|
|
1507
|
-
warnings.push(`Merge: ${mergeResult.error} for ${mergeResult.detail}.`);
|
|
1508
|
-
emitMergeFailureSkips(mergeResult.error);
|
|
1509
|
-
return;
|
|
1510
|
-
}
|
|
1511
|
-
let mergedContent = mergeResult.content;
|
|
1512
|
-
// Validate frontmatter of merged content — must have a `---` block
|
|
1513
|
-
// with at minimum a `description` field. We parse via the hand-rolled
|
|
1514
|
-
// parser (cheap) AND require non-empty description. This guards against
|
|
1515
|
-
// the historical defect where merged memories were written back with
|
|
1516
|
-
// empty `description` and later polluted the promote path.
|
|
1517
|
-
let parsedMerged;
|
|
1518
|
-
try {
|
|
1519
|
-
parsedMerged = parseFrontmatter(mergedContent);
|
|
1520
|
-
}
|
|
1521
|
-
catch {
|
|
1522
|
-
warnings.push(`Merge: merged content for ${op.primary} has invalid frontmatter — skipping.`);
|
|
1523
|
-
emitMergeFailureSkips("merge_invalid_frontmatter");
|
|
1524
|
-
return;
|
|
1525
|
-
}
|
|
1526
|
-
if (parsedMerged.frontmatter === null) {
|
|
1527
|
-
warnings.push(`Merge: merged content for ${op.primary} has no frontmatter block — skipping.`);
|
|
1528
|
-
emitMergeFailureSkips("merge_invalid_frontmatter");
|
|
1529
|
-
return;
|
|
1530
|
-
}
|
|
1531
|
-
const mergedDesc = parsedMerged.data.description;
|
|
1532
|
-
if (typeof mergedDesc !== "string" || mergedDesc.trim().length === 0) {
|
|
1533
|
-
warnings.push(`Merge: merged content for ${op.primary} missing description — skipping.`);
|
|
1534
|
-
emitMergeFailureSkips("merge_missing_description");
|
|
1535
|
-
return;
|
|
1536
|
-
}
|
|
1537
|
-
const truncReason = detectTruncatedDescription(mergedDesc);
|
|
1538
|
-
if (truncReason) {
|
|
1539
|
-
warnings.push(`Merge: merged content for ${op.primary} has truncated description (${truncReason}) — skipping.`);
|
|
1540
|
-
emitMergeFailureSkips("merge_truncated_description");
|
|
1541
|
-
return;
|
|
1542
|
-
}
|
|
1543
|
-
// captureMode:hot guard — refuse the merge if ANY participating memory
|
|
1544
|
-
// (primary or secondary) was user-captured or has unparseable frontmatter
|
|
1545
|
-
// (could have hidden a hot flag). Hot memories are user-explicit and
|
|
1546
|
-
// must not be deleted/overwritten by the consolidate LLM. 14 user
|
|
1547
|
-
// memories were silent-deleted by consolidate before this guard landed;
|
|
1548
|
-
// recovery required copying from .akm/archive/ by hand.
|
|
1549
|
-
const mergeParticipants = [op.primary, ...op.secondaries];
|
|
1550
|
-
const blockedParticipants = mergeParticipants.flatMap((ref) => {
|
|
1551
|
-
const e = memoryByRef.get(ref);
|
|
1552
|
-
if (!e)
|
|
1553
|
-
return [];
|
|
1554
|
-
const verdict = consolidateGuardStatus(e.filePath);
|
|
1555
|
-
if (verdict === "hot" || verdict === "unparseable")
|
|
1556
|
-
return [{ ref, verdict }];
|
|
1557
|
-
return [];
|
|
1558
|
-
});
|
|
1559
|
-
if (blockedParticipants.length > 0) {
|
|
1560
|
-
const detail = blockedParticipants.map((p) => `${p.ref} (${p.verdict})`).join(", ");
|
|
1561
|
-
warnings.push(`Merge: refused for ${op.primary} — ${blockedParticipants.length} participant(s) blocked by hot/unparseable frontmatter guard: ${detail}`);
|
|
1562
|
-
emitMergeFailureSkips("merge_participant_blocked");
|
|
1563
|
-
return;
|
|
1564
|
-
}
|
|
1565
|
-
// WS-3b: Anti-collapse generation guard (step 8a).
|
|
1566
|
-
// DEFAULT ON since R5 (opt out via antiCollapse.enabled: false). Refuses
|
|
1567
|
-
// to merge two assets both above generation N (default 2) — prevents the
|
|
1568
|
-
// pipeline from building ever-deeper LLM-merged trees that lose the
|
|
1569
|
-
// source fidelity of the original episodes.
|
|
1570
|
-
const antiCollapseConfig = getImproveProcessConfig(config, "consolidate", ctx.improveProfile)?.antiCollapse ?? {};
|
|
1571
|
-
if (antiCollapseConfig.enabled !== false) {
|
|
1572
|
-
const allParticipants = [op.primary, ...op.secondaries];
|
|
1573
|
-
// One read per participant: generation counter, stripped body (for the
|
|
1574
|
-
// information floor), and existing source_refs (for the provenance union).
|
|
1575
|
-
const participantInfo = allParticipants.map((ref) => {
|
|
1576
|
-
const e = memoryByRef.get(ref);
|
|
1577
|
-
if (!e)
|
|
1578
|
-
return { ref, generation: 0, body: "", sourceRefs: [] };
|
|
1579
|
-
try {
|
|
1580
|
-
const raw = fs.readFileSync(e.filePath, "utf8");
|
|
1581
|
-
const parsed = parseFrontmatter(raw);
|
|
1582
|
-
const fm = parsed.data;
|
|
1583
|
-
const sourceRefs = Array.isArray(fm.source_refs) ? fm.source_refs.map(String) : [];
|
|
1584
|
-
return { ref, generation: readAssetGeneration(fm), body: stripFrontmatterBody(raw), sourceRefs };
|
|
1585
|
-
}
|
|
1586
|
-
catch {
|
|
1587
|
-
return { ref, generation: 0, body: "", sourceRefs: [] };
|
|
1588
|
-
}
|
|
1589
|
-
});
|
|
1590
|
-
const sourceGenerations = participantInfo.map((p) => p.generation);
|
|
1591
|
-
const generationCheck = checkGenerationGuard(sourceGenerations, antiCollapseConfig);
|
|
1592
|
-
if (generationCheck.refused) {
|
|
1593
|
-
warnings.push(`Merge: ${generationCheck.reason}`);
|
|
1594
|
-
emitMergeFailureSkips("merge_generation_guard");
|
|
1595
|
-
return;
|
|
1596
|
-
}
|
|
1597
|
-
// WS-3b: Lexical diversity check (step 8b).
|
|
1598
|
-
// Low n-gram diversity ⇒ likely correlated-extraction artifact; raise merge threshold.
|
|
1599
|
-
if (antiCollapseConfig.lexicalDiversityCheck !== false) {
|
|
1600
|
-
const bodies = participantInfo.map((p) => p.body).filter((b) => b.length > 0);
|
|
1601
|
-
const diversityCheck = checkLexicalDiversity(bodies, antiCollapseConfig);
|
|
1602
|
-
if (diversityCheck.lowDiversity) {
|
|
1603
|
-
// Low-diversity cluster: just warn (don't refuse merge since the dedup
|
|
1604
|
-
// path handles exact twins). The warning surfaces in health telemetry.
|
|
1605
|
-
warnings.push(`Merge: cluster around ${op.primary} has low lexical diversity (${diversityCheck.diversity?.toFixed(2) ?? "?"} < 0.30) — likely correlated extraction; merge proceeds but review is recommended.`);
|
|
1606
|
-
}
|
|
1607
|
-
}
|
|
1608
|
-
// Inject generation counter into merged content frontmatter (step 8a).
|
|
1609
|
-
// merged.generation = max(sourceGenerations) + 1. source_refs is the
|
|
1610
|
-
// UNION of participants + everything they already cited (R5 §4.2 —
|
|
1611
|
-
// the old set-if-absent behavior dropped second-generation provenance).
|
|
1612
|
-
const provenanceUnion = [...new Set([...allParticipants, ...participantInfo.flatMap((p) => p.sourceRefs)])];
|
|
1613
|
-
mergedContent = injectGenerationFrontmatter(mergedContent, sourceGenerations, provenanceUnion);
|
|
1614
|
-
// R5 §4.2: merge-information floor — ADVISORY in v1. A merge that
|
|
1615
|
-
// shrinks provenance or genericizes below the retention floor is
|
|
1616
|
-
// counted + warned, never refused (promotion path: design doc §7).
|
|
1617
|
-
try {
|
|
1618
|
-
const mergedParsed = parseFrontmatter(mergedContent);
|
|
1619
|
-
const mergedFm = mergedParsed.data;
|
|
1620
|
-
const mergedSourceRefs = Array.isArray(mergedFm.source_refs) ? mergedFm.source_refs.map(String) : [];
|
|
1621
|
-
const floorCheck = checkMergeInformationFloor(mergedParsed.content, mergedSourceRefs, participantInfo, antiCollapseConfig);
|
|
1622
|
-
if (!floorCheck.passed) {
|
|
1623
|
-
counts.mergeFloorViolations++;
|
|
1624
|
-
warnings.push(`Merge: information floor advisory for ${op.primary}: ${floorCheck.reason ?? "unspecified"} — merge proceeds (v1 observe-only).`);
|
|
1625
|
-
}
|
|
1626
|
-
}
|
|
1627
|
-
catch {
|
|
1628
|
-
// Floor measurement is best-effort; never blocks the merge path.
|
|
1629
|
-
}
|
|
1630
|
-
}
|
|
1631
|
-
// Backup secondaries before deleting
|
|
1632
|
-
for (const secRef of op.secondaries) {
|
|
1633
|
-
const secEntry = memoryByRef.get(secRef);
|
|
1634
|
-
if (secEntry && fs.existsSync(secEntry.filePath)) {
|
|
1635
|
-
backupFile(secEntry.filePath, backupDir, secEntry.name);
|
|
1636
|
-
}
|
|
1637
|
-
}
|
|
1638
|
-
// Write merged primary
|
|
1639
|
-
try {
|
|
1640
|
-
const parsedPrimary = parseAssetRef(op.primary);
|
|
1641
|
-
await writeAssetToSource(target.source, target.config, parsedPrimary, mergedContent);
|
|
1642
|
-
}
|
|
1643
|
-
catch (e) {
|
|
1644
|
-
warnings.push(`Merge: write failed for ${op.primary}: ${String(e)}`);
|
|
1645
|
-
emitMergeFailureSkips("merge_write_failed");
|
|
1646
|
-
return;
|
|
1647
|
-
}
|
|
1648
|
-
// Archive and delete secondaries (P1-B: soft-invalidation)
|
|
1649
|
-
for (const secRef of op.secondaries) {
|
|
1650
|
-
const secEntry = memoryByRef.get(secRef);
|
|
1651
|
-
if (!secEntry)
|
|
1652
|
-
continue;
|
|
1653
|
-
if (fs.existsSync(secEntry.filePath)) {
|
|
1654
|
-
archiveMemory(secEntry.filePath, stashDir, secRef, "merged into primary", opIndex, op.primary, warnings);
|
|
1655
|
-
}
|
|
1656
|
-
try {
|
|
1657
|
-
const parsedSec = parseAssetRef(secRef);
|
|
1658
|
-
await deleteAssetFromSource(target.source, target.config, parsedSec);
|
|
1659
|
-
markJournalCompleted(stashDir, secRef);
|
|
1660
|
-
}
|
|
1661
|
-
catch (e) {
|
|
1662
|
-
warnings.push(`Merge: delete failed for ${secRef}: ${String(e)}`);
|
|
1663
|
-
}
|
|
1664
|
-
}
|
|
1665
|
-
markJournalCompleted(stashDir, op.primary);
|
|
1666
|
-
counts.merged++;
|
|
1667
|
-
// 2026-05-26 accounting-leak fix: `merged` is op-level, but each
|
|
1668
|
-
// successful merge actions `1 + secondaries.length` memories. Without
|
|
1669
|
-
// this counter the accounting invariant breaks by `secondaries.length`
|
|
1670
|
-
// per successful merge (chunk loop excluded all secondaries from
|
|
1671
|
-
// judgedNoAction via targetRefs, but only the primary is credited to
|
|
1672
|
-
// `merged`). Count only loaded-memory secondaries; phantom secondary
|
|
1673
|
-
// refs never affected any chunk's targetRefs in the first place.
|
|
1674
|
-
for (const secRef of op.secondaries) {
|
|
1675
|
-
if (memoryByRef.has(secRef))
|
|
1676
|
-
counts.mergedSecondaries++;
|
|
1677
|
-
}
|
|
1678
|
-
// Prune consumed refs from memoryByRef so later ops in this run cannot
|
|
1679
|
-
// reference an absorbed secondary as a merge primary and proceed with a
|
|
1680
|
-
// stale entry. Primary is rewritten (not deleted), so we only remove
|
|
1681
|
-
// secondaries; the primary ref remains valid under its new content.
|
|
1682
|
-
for (const secRef of op.secondaries) {
|
|
1683
|
-
memoryByRef.delete(secRef);
|
|
1684
|
-
}
|
|
1685
1013
|
}
|
|
1686
|
-
/** Execute one
|
|
1687
|
-
|
|
1688
|
-
const { stashDir, target, backupDir, memoryByRef, warnings, pushSkipReason, counts } = ctx;
|
|
1689
|
-
const entry = memoryByRef.get(op.ref);
|
|
1690
|
-
if (!entry) {
|
|
1691
|
-
warnings.push(`Delete: ${op.ref} not found in loaded memories — skipping.`);
|
|
1692
|
-
// Phantom ref: not in the batch so not in processed. Pushing to
|
|
1693
|
-
// skipReasons would inflate Σ(skipReasons) without a matching processed
|
|
1694
|
-
// entry, breaking the accounting invariant. Visibility is preserved via
|
|
1695
|
-
// the warnings array above.
|
|
1696
|
-
return;
|
|
1697
|
-
}
|
|
1698
|
-
// captureMode:hot guard — refuse to delete user-captured memories OR
|
|
1699
|
-
// memories whose frontmatter is unparseable (could have hidden the hot
|
|
1700
|
-
// flag). The consolidate LLM was deleting hot-captured user memos as
|
|
1701
|
-
// "redundant" — 14 such deletes were silently archived between
|
|
1702
|
-
// 2026-05-19 and 2026-05-20 before this guard. Hot memories are
|
|
1703
|
-
// user-explicit and may only be deleted by the user.
|
|
1704
|
-
const guard = consolidateGuardStatus(entry.filePath);
|
|
1705
|
-
if (guard === "hot" || guard === "unparseable") {
|
|
1706
|
-
warnings.push(`Delete: refused for ${op.ref} — ${guard === "hot" ? "captureMode:hot (user-explicit; never auto-delete)" : "frontmatter unparseable (cannot verify hot flag absent)"}. Reason from LLM: "${op.reason ?? "n/a"}"`);
|
|
1707
|
-
pushSkipReason("delete", op.ref, "captureMode_hot_refused");
|
|
1708
|
-
return;
|
|
1709
|
-
}
|
|
1710
|
-
if (fs.existsSync(entry.filePath)) {
|
|
1711
|
-
backupFile(entry.filePath, backupDir, entry.name);
|
|
1712
|
-
// P1-B: soft-invalidation archive before hard delete
|
|
1713
|
-
archiveMemory(entry.filePath, stashDir, op.ref, op.reason, opIndex, undefined, warnings);
|
|
1714
|
-
}
|
|
1715
|
-
try {
|
|
1716
|
-
const parsedRef = parseAssetRef(op.ref);
|
|
1717
|
-
await deleteAssetFromSource(target.source, target.config, parsedRef);
|
|
1718
|
-
markJournalCompleted(stashDir, op.ref);
|
|
1719
|
-
counts.deleted++;
|
|
1720
|
-
// Prune from memoryByRef so later ops in this run cannot reference a
|
|
1721
|
-
// deleted memory as a merge primary or secondary.
|
|
1722
|
-
memoryByRef.delete(op.ref);
|
|
1723
|
-
}
|
|
1724
|
-
catch (e) {
|
|
1725
|
-
// Distinguish "file already absent" from genuine failures. A prior run
|
|
1726
|
-
// may have deleted the file but the DB was not yet re-indexed, so the
|
|
1727
|
-
// ref still appeared in memoryByRef. The delete goal is already met.
|
|
1728
|
-
const msg = e instanceof Error ? e.message : String(e);
|
|
1729
|
-
if (msg.includes("not found in source")) {
|
|
1730
|
-
warnings.push(`Delete: ${op.ref} — file already absent (stale DB entry); skipping.`);
|
|
1731
|
-
pushSkipReason("delete", op.ref, "delete_already_gone");
|
|
1732
|
-
}
|
|
1733
|
-
else {
|
|
1734
|
-
warnings.push(`Delete: failed for ${op.ref}: ${String(e)}`);
|
|
1735
|
-
pushSkipReason("delete", op.ref, "delete_failed");
|
|
1736
|
-
}
|
|
1737
|
-
}
|
|
1738
|
-
}
|
|
1739
|
-
/** Execute one `promote` op (behavior-identical to the former inlined branch). */
|
|
1740
|
-
export async function handlePromoteOp(op, ctx) {
|
|
1014
|
+
/** Execute one reconciled promotion by emitting a reviewable proposal. */
|
|
1015
|
+
async function emitPromotionProposal(op, ctx) {
|
|
1741
1016
|
const { config, stashDir, sourceRun, target, memoryByRef, warnings, pushSkipReason, promoted, promotedSourceRefs } = ctx;
|
|
1742
1017
|
const entry = memoryByRef.get(op.ref);
|
|
1743
1018
|
if (!entry) {
|
|
@@ -1755,27 +1030,28 @@ export async function handlePromoteOp(op, ctx) {
|
|
|
1755
1030
|
pushSkipReason("promote", op.ref, "promote_already_promoted_this_run");
|
|
1756
1031
|
return;
|
|
1757
1032
|
}
|
|
1758
|
-
|
|
1759
|
-
|
|
1760
|
-
|
|
1761
|
-
|
|
1762
|
-
|
|
1763
|
-
|
|
1764
|
-
|
|
1765
|
-
|
|
1766
|
-
|
|
1767
|
-
|
|
1768
|
-
|
|
1769
|
-
|
|
1770
|
-
|
|
1771
|
-
|
|
1772
|
-
|
|
1033
|
+
const proposedName = op.knowledgeRef.split("/").filter(Boolean).at(-1) ??
|
|
1034
|
+
entry.name.split("/").filter(Boolean).at(-1) ??
|
|
1035
|
+
"promoted-memory";
|
|
1036
|
+
const slug = proposedName
|
|
1037
|
+
.replace(/[^a-z0-9-]/gi, "-")
|
|
1038
|
+
.replace(/-+/g, "-")
|
|
1039
|
+
.replace(/^-|-$/g, "")
|
|
1040
|
+
.toLowerCase();
|
|
1041
|
+
const knowledgeRef = conceptIdFromTypeName("knowledge", slug);
|
|
1042
|
+
parseRefInput(knowledgeRef);
|
|
1043
|
+
if (knowledgeRef !== op.knowledgeRef) {
|
|
1044
|
+
warnings.push(`Normalized generated ref "${op.knowledgeRef}" → "${knowledgeRef}"`);
|
|
1045
|
+
}
|
|
1046
|
+
// A pending proposal may carry a qualified item_ref, so compare its parsed
|
|
1047
|
+
// conceptId rather than exact display spelling.
|
|
1048
|
+
if (hasPendingProposalForConcept(stashDir, knowledgeRef)) {
|
|
1773
1049
|
warnings.push(`Skipping promote: pending proposal already exists for ${knowledgeRef}`);
|
|
1774
1050
|
pushSkipReason("promote", op.ref, "promote_pending_proposal_exists");
|
|
1775
1051
|
return;
|
|
1776
1052
|
}
|
|
1777
1053
|
// Idempotency: check if knowledge asset already exists
|
|
1778
|
-
const parsedKnowledgeRef =
|
|
1054
|
+
const parsedKnowledgeRef = parseRefInput(knowledgeRef);
|
|
1779
1055
|
const destPath = path.join(target.source.path, "knowledge", `${parsedKnowledgeRef.name}.md`);
|
|
1780
1056
|
if (fs.existsSync(destPath)) {
|
|
1781
1057
|
warnings.push(`Skipping promote: ${knowledgeRef} already exists in source`);
|
|
@@ -1791,9 +1067,7 @@ export async function handlePromoteOp(op, ctx) {
|
|
|
1791
1067
|
pushSkipReason("promote", op.ref, "promote_read_failed");
|
|
1792
1068
|
return;
|
|
1793
1069
|
}
|
|
1794
|
-
//
|
|
1795
|
-
// consolidate runs may still carry outer code fences or broken YAML.
|
|
1796
|
-
// Strip them here so we never propose a polluted asset.
|
|
1070
|
+
// Validate and normalize source content before proposing a promoted asset.
|
|
1797
1071
|
const promoteSanitized = sanitizeMergedContent(memoryContent);
|
|
1798
1072
|
if (!promoteSanitized.ok) {
|
|
1799
1073
|
warnings.push(`Promote: rejected ${op.ref} — source memory failed sanitization (${promoteSanitized.reason}).`);
|
|
@@ -1839,7 +1113,7 @@ export async function handlePromoteOp(op, ctx) {
|
|
|
1839
1113
|
const bodyHash = cacheHash(sourceBody);
|
|
1840
1114
|
const allPendingConsolidateProposals = listProposals(stashDir, { status: "pending" }).filter((p) => p.source === "consolidate");
|
|
1841
1115
|
const contentDupProposal = allPendingConsolidateProposals.find((p) => {
|
|
1842
|
-
return cacheHash(p
|
|
1116
|
+
return cacheHash(proposalContent(p)) === bodyHash;
|
|
1843
1117
|
});
|
|
1844
1118
|
if (contentDupProposal) {
|
|
1845
1119
|
warnings.push(`Skipping promote: identical body already pending as proposal ${contentDupProposal.id} (ref: ${contentDupProposal.ref}); skipping duplicate for ${op.ref} → ${knowledgeRef}`);
|
|
@@ -1876,9 +1150,10 @@ export async function handlePromoteOp(op, ctx) {
|
|
|
1876
1150
|
const mergedBodyFm = {
|
|
1877
1151
|
...(parsedMemory.data ?? {}),
|
|
1878
1152
|
description,
|
|
1153
|
+
xrefs: promoteProvenanceXrefs(parsedMemory.data?.xrefs, op.ref),
|
|
1879
1154
|
};
|
|
1880
1155
|
const serializedMergedFm = serializeFrontmatter(mergedBodyFm);
|
|
1881
|
-
const
|
|
1156
|
+
const promotedAssetContent = assembleAssetFromString(serializedMergedFm, parsedMemory.content);
|
|
1882
1157
|
// Pre-emit dedup against pending consolidate proposals from the
|
|
1883
1158
|
// same improve run (slug-variant match). The cross-run content-hash
|
|
1884
1159
|
// dedup inside `mergePlans` handles duplicates against existing
|
|
@@ -1895,13 +1170,16 @@ export async function handlePromoteOp(op, ctx) {
|
|
|
1895
1170
|
pushSkipReason("promote", op.ref, "promote_dedup_window");
|
|
1896
1171
|
return;
|
|
1897
1172
|
}
|
|
1898
|
-
const proposalResult =
|
|
1173
|
+
const proposalResult = emitProposal({ stashDir, proposalsCtx: ctx.proposalsCtx }, {
|
|
1899
1174
|
ref: knowledgeRef,
|
|
1175
|
+
target: { source: target.source.name, root: target.source.path },
|
|
1900
1176
|
source: "consolidate",
|
|
1901
1177
|
sourceRun,
|
|
1178
|
+
// §23.6 fingerprint model-id term (WI-6.4).
|
|
1179
|
+
...(ctx.llmConfig?.model ? { modelId: ctx.llmConfig.model } : {}),
|
|
1902
1180
|
payload: {
|
|
1903
|
-
content:
|
|
1904
|
-
frontmatter: { description },
|
|
1181
|
+
content: promotedAssetContent,
|
|
1182
|
+
frontmatter: { description, xrefs: [canonicalXref(op.ref)] },
|
|
1905
1183
|
},
|
|
1906
1184
|
...(typeof op.confidence === "number" ? { confidence: op.confidence } : {}),
|
|
1907
1185
|
});
|
|
@@ -1912,56 +1190,14 @@ export async function handlePromoteOp(op, ctx) {
|
|
|
1912
1190
|
else {
|
|
1913
1191
|
promoted.push(proposalResult.id);
|
|
1914
1192
|
promotedSourceRefs.add(op.ref);
|
|
1915
|
-
markJournalCompleted(stashDir, op.ref);
|
|
1916
1193
|
}
|
|
1917
1194
|
}
|
|
1918
1195
|
catch (e) {
|
|
1196
|
+
ctx.promotionFailures.count++;
|
|
1919
1197
|
warnings.push(`Promote: createProposal failed for ${op.ref}: ${String(e)}`);
|
|
1920
1198
|
pushSkipReason("promote", op.ref, "promote_create_failed");
|
|
1921
1199
|
}
|
|
1922
1200
|
}
|
|
1923
|
-
/** Execute one `contradict` op (behavior-identical to the former inlined branch). */
|
|
1924
|
-
export async function handleContradictOp(op, ctx) {
|
|
1925
|
-
const { stashDir, memoryByRef, warnings, pushSkipReason, counts } = ctx;
|
|
1926
|
-
// Confidence gate: surface-level topic overlap causes false positives
|
|
1927
|
-
// (investigation 2026-06-18). Require ≥0.92 confidence before writing
|
|
1928
|
-
// contradiction edges. Missing confidence field defaults to 1.0 for
|
|
1929
|
-
// backward compatibility with responses that predate this field.
|
|
1930
|
-
const opConfidence = typeof op.confidence === "number" ? op.confidence : 1.0;
|
|
1931
|
-
if (opConfidence < 0.92) {
|
|
1932
|
-
warnings.push(`Contradict: confidence ${opConfidence.toFixed(2)} below 0.92 threshold for ${op.ref} <-> ${op.contradictedByRef} — skipping.`);
|
|
1933
|
-
pushSkipReason("contradict", op.ref, "contradict_low_confidence");
|
|
1934
|
-
return;
|
|
1935
|
-
}
|
|
1936
|
-
// C-3 / #382: Write contradictedBy edges so resolveFamilyContradictions
|
|
1937
|
-
// (the SCC resolver in memory-improve.ts) has edges to work on.
|
|
1938
|
-
// Zep arXiv:2501.13956 §3 — unified belief-revision with contradiction edges.
|
|
1939
|
-
const entry = memoryByRef.get(op.ref);
|
|
1940
|
-
const contradictorEntry = memoryByRef.get(op.contradictedByRef);
|
|
1941
|
-
if (!entry) {
|
|
1942
|
-
warnings.push(`Contradict: ${op.ref} not found in loaded memories — skipping.`);
|
|
1943
|
-
// Phantom ref: not in processed, so no skipReason (same rationale as
|
|
1944
|
-
// delete_ref_missing).
|
|
1945
|
-
return;
|
|
1946
|
-
}
|
|
1947
|
-
if (!contradictorEntry) {
|
|
1948
|
-
warnings.push(`Contradict: ${op.contradictedByRef} not found — skipping.`);
|
|
1949
|
-
// op.ref IS in the batch (entry found above) so the skipReason is
|
|
1950
|
-
// correctly charged against a real processed memory.
|
|
1951
|
-
pushSkipReason("contradict", op.ref, "contradict_target_missing");
|
|
1952
|
-
return;
|
|
1953
|
-
}
|
|
1954
|
-
try {
|
|
1955
|
-
// Write the contradiction edge: op.ref is contradicted by op.contradictedByRef
|
|
1956
|
-
writeContradictEdge(entry.filePath, op.contradictedByRef);
|
|
1957
|
-
counts.contradicted++;
|
|
1958
|
-
markJournalCompleted(stashDir, op.ref);
|
|
1959
|
-
}
|
|
1960
|
-
catch (e) {
|
|
1961
|
-
warnings.push(`Contradict: failed to write edge for ${op.ref}: ${String(e)}`);
|
|
1962
|
-
pushSkipReason("contradict", op.ref, "contradict_write_failed");
|
|
1963
|
-
}
|
|
1964
|
-
}
|
|
1965
1201
|
// ── Helpers ─────────────────────────────────────────────────────────────────
|
|
1966
1202
|
/**
|
|
1967
1203
|
* Normalise a knowledge slug for variant-aware deduplication. Collapses:
|
|
@@ -1973,8 +1209,23 @@ export async function handleContradictOp(op, ctx) {
|
|
|
1973
1209
|
* Two slugs that normalise to the same string are considered the same asset
|
|
1974
1210
|
* for dedup purposes even if they don't share an exact ref.
|
|
1975
1211
|
*/
|
|
1212
|
+
/** The conceptId a proposal ref maps to, or undefined for an invalid ref. */
|
|
1213
|
+
function conceptIdForRef(ref) {
|
|
1214
|
+
try {
|
|
1215
|
+
const p = parseRefInput(ref);
|
|
1216
|
+
return conceptIdFromTypeName(p.type, p.name);
|
|
1217
|
+
}
|
|
1218
|
+
catch {
|
|
1219
|
+
return undefined;
|
|
1220
|
+
}
|
|
1221
|
+
}
|
|
1222
|
+
/** Is a pending proposal already queued for `conceptRef`'s concept? */
|
|
1223
|
+
function hasPendingProposalForConcept(stashDir, conceptRef) {
|
|
1224
|
+
const want = conceptIdForRef(conceptRef);
|
|
1225
|
+
return (want !== undefined && listProposals(stashDir, { status: "pending" }).some((p) => conceptIdForRef(p.ref) === want));
|
|
1226
|
+
}
|
|
1976
1227
|
function normalizeSlugForDedup(ref) {
|
|
1977
|
-
const slug = ref.
|
|
1228
|
+
const slug = parseRefInput(ref).name;
|
|
1978
1229
|
const monthRe = /(?:jan|feb|mar|apr|may|jun|jul|aug|sep|oct|nov|dec)/i;
|
|
1979
1230
|
const tokens = slug
|
|
1980
1231
|
.toLowerCase()
|
|
@@ -2029,22 +1280,10 @@ async function checkPreEmitDedup(opts) {
|
|
|
2029
1280
|
* everything changed or the index can't answer (fail-open to preserve merge
|
|
2030
1281
|
* correctness). `since` is an ISO timestamp.
|
|
2031
1282
|
*/
|
|
2032
|
-
/**
|
|
2033
|
-
* Parse a human-readable duration string (e.g. "30m", "24h", "7d") to an ISO
|
|
2034
|
-
* timestamp representing `now - duration`. Returns the input unchanged when it
|
|
2035
|
-
* doesn't match the pattern (assumed to already be an ISO timestamp).
|
|
2036
|
-
*/
|
|
2037
|
-
function parseSinceToIso(since) {
|
|
2038
|
-
// Canonical CLI unit grammar: `m` = minutes, `M` = months (see core/time.ts
|
|
2039
|
-
// DURATION_UNITS). Non-matching input is returned unchanged (assumed to
|
|
2040
|
-
// already be an ISO timestamp).
|
|
2041
|
-
const ms = parseDuration(since, DURATION_UNITS);
|
|
2042
|
-
if (ms === null)
|
|
2043
|
-
return since;
|
|
2044
|
-
return new Date(Date.now() - ms).toISOString();
|
|
2045
|
-
}
|
|
2046
1283
|
export function narrowToIncrementalCandidates(memories, since, warnings, neighborsPerChanged = 5) {
|
|
2047
|
-
|
|
1284
|
+
// Lenient by design: garbage `since` passes through unchanged and the ISO
|
|
1285
|
+
// string comparison below then selects nothing (see core/time.ts doc).
|
|
1286
|
+
const sinceIso = parseSinceToIsoLenient(since);
|
|
2048
1287
|
const isChanged = (m) => {
|
|
2049
1288
|
try {
|
|
2050
1289
|
return fs.statSync(m.filePath).mtime.toISOString() > sinceIso;
|
|
@@ -2064,7 +1303,7 @@ export function narrowToIncrementalCandidates(memories, since, warnings, neighbo
|
|
|
2064
1303
|
try {
|
|
2065
1304
|
db = openExistingDatabase();
|
|
2066
1305
|
for (const m of changed) {
|
|
2067
|
-
const id = findEntryIdByRef(db,
|
|
1306
|
+
const id = findEntryIdByRef(db, conceptIdFromTypeName("memory", m.name));
|
|
2068
1307
|
if (id === undefined)
|
|
2069
1308
|
continue;
|
|
2070
1309
|
for (const hit of getNeighborsByEntryId(db, id, neighborsPerChanged + 1)) {
|
|
@@ -2131,14 +1370,22 @@ function loadMemoriesForSource(source, stashDir, warnings) {
|
|
|
2131
1370
|
const memoriesDir = path.join(source ?? stashDir, "memories");
|
|
2132
1371
|
const fsStashDir = source ?? stashDir;
|
|
2133
1372
|
if (fs.existsSync(memoriesDir)) {
|
|
2134
|
-
|
|
2135
|
-
|
|
2136
|
-
|
|
2137
|
-
const
|
|
2138
|
-
|
|
2139
|
-
|
|
2140
|
-
|
|
2141
|
-
|
|
1373
|
+
const pending = [memoriesDir];
|
|
1374
|
+
while (pending.length > 0) {
|
|
1375
|
+
const current = pending.pop();
|
|
1376
|
+
for (const entry of fs.readdirSync(current, { withFileTypes: true })) {
|
|
1377
|
+
const filePath = path.join(current, entry.name);
|
|
1378
|
+
if (entry.isDirectory()) {
|
|
1379
|
+
pending.push(filePath);
|
|
1380
|
+
continue;
|
|
1381
|
+
}
|
|
1382
|
+
if (!entry.isFile() || !entry.name.endsWith(".md"))
|
|
1383
|
+
continue;
|
|
1384
|
+
const name = path.relative(memoriesDir, filePath).replace(/\.md$/, "").split(path.sep).join("/");
|
|
1385
|
+
if (!isConsolidationEligibleMemoryName(name))
|
|
1386
|
+
continue;
|
|
1387
|
+
memories.push({ name, filePath, description: "", tags: [], stashDir: fsStashDir });
|
|
1388
|
+
}
|
|
2142
1389
|
}
|
|
2143
1390
|
}
|
|
2144
1391
|
if (memories.length > 0) {
|
|
@@ -2147,150 +1394,3 @@ function loadMemoriesForSource(source, stashDir, warnings) {
|
|
|
2147
1394
|
}
|
|
2148
1395
|
return memories;
|
|
2149
1396
|
}
|
|
2150
|
-
async function generateMergedContent(config, primaryRef, primaryBody, secondaryRefs, memoryByRef, activeProfile) {
|
|
2151
|
-
// Only handle single-secondary merges per design (one call per merge op)
|
|
2152
|
-
const secRef = secondaryRefs[0];
|
|
2153
|
-
const secEntry = memoryByRef.get(secRef);
|
|
2154
|
-
if (!secEntry)
|
|
2155
|
-
return { error: "merge_read_failed", detail: `secondary ${secRef} not in memoryByRef` };
|
|
2156
|
-
let secBody = "";
|
|
2157
|
-
try {
|
|
2158
|
-
secBody = fs.readFileSync(secEntry.filePath, "utf8");
|
|
2159
|
-
}
|
|
2160
|
-
catch {
|
|
2161
|
-
return { error: "merge_read_failed", detail: `could not read secondary ${secRef}` };
|
|
2162
|
-
}
|
|
2163
|
-
const primaryFmKeys = Object.keys(parseFrontmatter(primaryBody).data);
|
|
2164
|
-
const secFmKeys = Object.keys(parseFrontmatter(secBody).data);
|
|
2165
|
-
const requiredFmKeys = [...new Set([...primaryFmKeys, ...secFmKeys])];
|
|
2166
|
-
const prompt = [
|
|
2167
|
-
"Merge these two memory assets into one. Output ONLY the merged markdown (with YAML frontmatter). Do not explain, do not use code fences.",
|
|
2168
|
-
"",
|
|
2169
|
-
"## OUTPUT FORMAT (MANDATORY)",
|
|
2170
|
-
"Return raw markdown content beginning DIRECTLY with the `---` frontmatter delimiter.",
|
|
2171
|
-
"DO NOT wrap your entire response in a code fence.",
|
|
2172
|
-
"",
|
|
2173
|
-
'GOOD: "---\\ndescription: ...\\n---\\nBody content."',
|
|
2174
|
-
'BAD: "```markdown\\n---\\ndescription: ...\\n---\\nBody content.\\n```"',
|
|
2175
|
-
'BAD: "```yaml\\n---\\ndescription: ...\\n---\\nBody content.\\n```"',
|
|
2176
|
-
"",
|
|
2177
|
-
"## FRONTMATTER RULES (MANDATORY)",
|
|
2178
|
-
"- The `updated:` field, if present, MUST be a real ISO date (e.g. `updated: 2026-05-20`). NEVER emit `updated: today`, `updated: now`, or `updated: {today: null}`. If you don't have a real date, OMIT the field — the post-processor will not invent one.",
|
|
2179
|
-
"- REQUIRED: The merged frontmatter MUST include a `description` field with a concise one-sentence summary of the merged asset's content. If neither source has a `description` field, synthesize one from the content.",
|
|
2180
|
-
requiredFmKeys.length > 0
|
|
2181
|
-
? `- CRITICAL: The merged frontmatter MUST include ALL of these keys from both source memories: ${requiredFmKeys.join(", ")}. Do NOT drop any of them.`
|
|
2182
|
-
: null,
|
|
2183
|
-
"",
|
|
2184
|
-
`=== Primary memory (${primaryRef}) ===`,
|
|
2185
|
-
primaryBody,
|
|
2186
|
-
"",
|
|
2187
|
-
`=== Secondary memory (${secRef}) ===`,
|
|
2188
|
-
secBody,
|
|
2189
|
-
]
|
|
2190
|
-
.filter((line) => line !== null)
|
|
2191
|
-
.join("\n");
|
|
2192
|
-
// Use the same per-process profile resolution as the chunk-plan call above
|
|
2193
|
-
// so the merge generation step doesn't silently revert to the default LLM.
|
|
2194
|
-
const llmConfig = resolveConsolidateLlmConfig(config, activeProfile);
|
|
2195
|
-
const result = await tryLlmFeature("memory_consolidation", config, async () => {
|
|
2196
|
-
if (!llmConfig)
|
|
2197
|
-
return { ok: false, error: "No LLM configured for consolidation" };
|
|
2198
|
-
try {
|
|
2199
|
-
const content = await chatCompletion(llmConfig, [{ role: "user", content: prompt }], {
|
|
2200
|
-
enableThinking: false,
|
|
2201
|
-
});
|
|
2202
|
-
return { ok: true, content };
|
|
2203
|
-
}
|
|
2204
|
-
catch (e) {
|
|
2205
|
-
return { ok: false, error: String(e) };
|
|
2206
|
-
}
|
|
2207
|
-
}, { ok: false, error: `merge content generation failed for ${primaryRef}` });
|
|
2208
|
-
if (!result.ok) {
|
|
2209
|
-
return {
|
|
2210
|
-
error: "merge_transport_failed",
|
|
2211
|
-
detail: result.error ?? `merge content generation failed for ${primaryRef}`,
|
|
2212
|
-
};
|
|
2213
|
-
}
|
|
2214
|
-
// Sanitize LLM output: strip outer code fences (defends against the
|
|
2215
|
-
// ```markdown … ``` leak observed in production), re-serialise frontmatter
|
|
2216
|
-
// through the yaml lib (fixes quote-escaping mistakes), and reject empty
|
|
2217
|
-
// or fence-only responses.
|
|
2218
|
-
const sanitized = sanitizeMergedContent(result.content ?? "");
|
|
2219
|
-
if (!sanitized.ok) {
|
|
2220
|
-
const reason = sanitized.reason;
|
|
2221
|
-
const isFenceError = reason === "UNBALANCED_CODE_FENCE" ||
|
|
2222
|
-
reason === "MISSING_FRONTMATTER_SENTINEL" ||
|
|
2223
|
-
reason === "MALFORMED_FRONTMATTER_BLOCK" ||
|
|
2224
|
-
reason === "FRONTMATTER_NOT_OBJECT";
|
|
2225
|
-
const mergeReason = isFenceError ? "merge_fence_rejected" : "merge_yaml_invalid";
|
|
2226
|
-
return { error: mergeReason, detail: `${primaryRef} — ${reason}` };
|
|
2227
|
-
}
|
|
2228
|
-
const mergedRaw = sanitized.result.content;
|
|
2229
|
-
// C-4 / #383: Content-preservation lint (mem0 §3.2, arXiv:2504.19413).
|
|
2230
|
-
// Guards against LLM-generated merged content that silently drops information
|
|
2231
|
-
// from the source assets. Two checks:
|
|
2232
|
-
// 1. Body size: merged body must be >= 50% of the larger source body.
|
|
2233
|
-
// 2. Frontmatter superset: merged frontmatter must contain all keys present
|
|
2234
|
-
// in both source frontmatters.
|
|
2235
|
-
// Failures return a discriminated error so the call site can emit a specific
|
|
2236
|
-
// skip-reason key in the histogram.
|
|
2237
|
-
try {
|
|
2238
|
-
const primaryFm = parseFrontmatter(primaryBody);
|
|
2239
|
-
const secFm = parseFrontmatter(secBody);
|
|
2240
|
-
const mergedFm = parseFrontmatter(mergedRaw);
|
|
2241
|
-
// Check body size — blended floor: max(ratio × largerLen, absoluteFloor).
|
|
2242
|
-
// Deduplication is expected, so the ratio is lower than the reflect gate
|
|
2243
|
-
// (0.3 vs 0.5). The absolute floor protects very short memory pairs where
|
|
2244
|
-
// the ratio alone would produce a near-zero threshold.
|
|
2245
|
-
const primaryBodyLen = (primaryFm.content ?? "").trim().length;
|
|
2246
|
-
const secBodyLen = (secFm.content ?? "").trim().length;
|
|
2247
|
-
const mergedBodyLen = (mergedFm.content ?? "").trim().length;
|
|
2248
|
-
const largerBodyLen = Math.max(primaryBodyLen, secBodyLen);
|
|
2249
|
-
const mergeFloor = Math.max(MERGE_SHRINK_RATIO_MIN * largerBodyLen, MERGE_ABSOLUTE_FLOOR_CHARS);
|
|
2250
|
-
if (largerBodyLen > 0 && mergedBodyLen < mergeFloor) {
|
|
2251
|
-
return {
|
|
2252
|
-
error: "merge_content_too_short",
|
|
2253
|
-
detail: `${primaryRef} — merged body (${mergedBodyLen} chars) is less than floor (${Math.round(mergeFloor)} chars; max(${MERGE_SHRINK_RATIO_MIN}×${largerBodyLen}, ${MERGE_ABSOLUTE_FLOOR_CHARS}))`,
|
|
2254
|
-
};
|
|
2255
|
-
}
|
|
2256
|
-
// Check frontmatter superset — attempt repair before rejecting.
|
|
2257
|
-
const primaryKeys = Object.keys(primaryFm.data ?? {});
|
|
2258
|
-
const secKeys = Object.keys(secFm.data ?? {});
|
|
2259
|
-
const mergedKeys = new Set(Object.keys(mergedFm.data ?? {}));
|
|
2260
|
-
const missingKeys = [...new Set([...primaryKeys, ...secKeys])].filter((k) => !mergedKeys.has(k));
|
|
2261
|
-
if (missingKeys.length > 0) {
|
|
2262
|
-
// Inject missing keys from source FMs. Primary value wins on conflict.
|
|
2263
|
-
const repairedFmData = { ...mergedFm.data };
|
|
2264
|
-
for (const key of missingKeys) {
|
|
2265
|
-
repairedFmData[key] =
|
|
2266
|
-
key in primaryFm.data
|
|
2267
|
-
? primaryFm.data[key]
|
|
2268
|
-
: secFm.data[key];
|
|
2269
|
-
}
|
|
2270
|
-
normalizeUpdatedField(repairedFmData);
|
|
2271
|
-
const repairedYaml = serializeFrontmatter(repairedFmData);
|
|
2272
|
-
const bodyPart = typeof mergedFm.content === "string" ? mergedFm.content : "";
|
|
2273
|
-
return { content: assembleAssetFromString(repairedYaml, bodyPart) };
|
|
2274
|
-
}
|
|
2275
|
-
}
|
|
2276
|
-
catch {
|
|
2277
|
-
// parseFrontmatter failures are non-fatal — allow the merge to proceed.
|
|
2278
|
-
}
|
|
2279
|
-
return { content: mergedRaw };
|
|
2280
|
-
}
|
|
2281
|
-
async function promptConfirm(message) {
|
|
2282
|
-
process.stdout.write(message);
|
|
2283
|
-
return new Promise((resolve) => {
|
|
2284
|
-
let settled = false;
|
|
2285
|
-
const done = (answer) => {
|
|
2286
|
-
if (settled)
|
|
2287
|
-
return;
|
|
2288
|
-
settled = true;
|
|
2289
|
-
rl.close();
|
|
2290
|
-
resolve(answer);
|
|
2291
|
-
};
|
|
2292
|
-
const rl = readline.createInterface({ input: process.stdin, output: process.stdout });
|
|
2293
|
-
rl.once("line", (line) => done(line.trim().toLowerCase() === "y"));
|
|
2294
|
-
rl.once("close", () => done(false));
|
|
2295
|
-
});
|
|
2296
|
-
}
|