akm-cli 0.9.0-rc.0 → 0.9.0-rc.13
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +1283 -22
- package/README.md +62 -37
- package/SECURITY.md +46 -31
- package/dist/akm +162 -38
- package/dist/akm-migrate +44 -0
- package/dist/assets/backends/schtasks-template.xml +2 -1
- package/dist/assets/hints/cli-hints-full.md +268 -118
- package/dist/assets/hints/cli-hints-short.md +87 -24
- package/dist/assets/{profiles → improve-strategies}/catchup.json +3 -1
- package/dist/assets/{profiles → improve-strategies}/consolidate.json +3 -1
- package/dist/assets/{profiles → improve-strategies}/default.json +6 -7
- package/dist/assets/improve-strategies/frequent.json +15 -0
- package/dist/assets/{profiles → improve-strategies}/graph-refresh.json +4 -2
- package/dist/assets/{profiles → improve-strategies}/memory-focus.json +4 -1
- package/dist/assets/{profiles → improve-strategies}/proactive-maintenance.json +5 -5
- package/dist/assets/{profiles → improve-strategies}/quick.json +4 -2
- package/dist/assets/improve-strategies/reflect-distill.json +30 -0
- package/dist/assets/{profiles → improve-strategies}/thorough.json +1 -1
- package/dist/assets/prompts/consolidate-system.md +5 -5
- package/dist/assets/prompts/extract-session.md +2 -6
- package/dist/assets/prompts/memory-infer-user.md +2 -3
- package/dist/assets/prompts/reflect-llm-framed-contract.md +11 -0
- package/dist/assets/prompts/reflect-llm-schema-contract.md +3 -0
- package/dist/assets/prompts/reflect-output-repair.md +3 -0
- package/dist/assets/prompts/workflow-unit-preamble.md +26 -0
- package/dist/assets/stash-skeleton/README.md +38 -10
- package/dist/assets/stash-skeleton/facts/conventions/assets/agent.md +8 -0
- package/dist/assets/stash-skeleton/facts/conventions/assets/command.md +8 -0
- package/dist/assets/stash-skeleton/facts/conventions/assets/fact.md +14 -1
- package/dist/assets/stash-skeleton/facts/conventions/assets/knowledge.md +13 -1
- package/dist/assets/stash-skeleton/facts/conventions/assets/lesson.md +9 -1
- package/dist/assets/stash-skeleton/facts/conventions/assets/memory.md +11 -0
- package/dist/assets/stash-skeleton/facts/conventions/assets/script.md +9 -0
- package/dist/assets/stash-skeleton/facts/conventions/assets/skill.md +9 -0
- package/dist/assets/stash-skeleton/facts/conventions/assets/workflow.md +8 -0
- package/dist/assets/stash-skeleton/facts/conventions/backlinks.md +100 -0
- package/dist/assets/stash-skeleton/facts/conventions/domains.md +64 -0
- package/dist/assets/stash-skeleton/facts/conventions/organization.md +136 -0
- package/dist/assets/tasks/core/extract.yml +3 -2
- package/dist/assets/tasks/core/improve.yml +2 -1
- package/dist/assets/tasks/core/index-refresh.yml +1 -0
- package/dist/assets/tasks/core/sync.yml +1 -0
- package/dist/assets/tasks/core/version-check.yml +2 -1
- package/dist/assets/tasks/improve/akm-graph-refresh-weekly.yml +5 -0
- package/dist/assets/tasks/improve/akm-improve-catchup.yml +8 -0
- package/dist/assets/tasks/improve/akm-improve-consolidate.yml +5 -0
- package/dist/assets/tasks/improve/akm-improve-frequent.yml +5 -0
- package/dist/assets/tasks/improve/akm-improve-nightly.yml +5 -0
- package/dist/assets/templates/html/health.html +5 -4
- package/dist/assets/workflows/workflow-template.md +31 -15
- package/dist/cli/invocation.js +279 -0
- package/dist/cli/parse-args.js +5 -90
- package/dist/cli/retired-commands.js +78 -0
- package/dist/cli/shared.js +158 -48
- package/dist/cli-node.mjs +2 -1
- package/dist/cli.js +747 -293
- package/dist/commands/agent/agent-dispatch.js +19 -18
- package/dist/commands/agent/agent-support.js +0 -24
- package/dist/commands/agent/contribute-cli.js +43 -97
- package/dist/commands/completions.js +80 -23
- package/dist/commands/config-cli.js +44 -281
- package/dist/commands/env/env-binding.js +99 -0
- package/dist/commands/env/env-cli.js +84 -224
- package/dist/commands/env/env.js +12 -163
- package/dist/commands/env/marker-path.js +6 -0
- package/dist/commands/env/secret-cli.js +45 -61
- package/dist/commands/env/secret.js +32 -62
- package/dist/commands/feedback-cli.js +179 -85
- package/dist/commands/health/accept-rate.js +58 -0
- package/dist/commands/health/advisories.js +7 -8
- package/dist/commands/health/checks.js +279 -94
- package/dist/commands/health/html-report.js +197 -578
- package/dist/commands/health/improve-metrics.js +277 -246
- package/dist/commands/health/llm-usage.js +19 -19
- package/dist/commands/health/md-report.js +16 -7
- package/dist/commands/health/metrics.js +67 -32
- package/dist/commands/health/renderers.js +47 -0
- package/dist/commands/health/report-view-model.js +508 -0
- package/dist/commands/health/stash-exposure.js +1 -1
- package/dist/commands/health/surfaces.js +16 -56
- package/dist/commands/health/task-runs.js +3 -67
- package/dist/{migrate-storage-node.mjs → commands/health/types-checks.js} +1 -5
- package/dist/commands/health/types-improve.js +29 -0
- package/dist/{output/text/save.js → commands/health/types-metrics.js} +1 -2
- package/dist/commands/health/types-result.js +7 -0
- package/dist/commands/health/types-runs.js +4 -0
- package/dist/commands/health/types-session-log.js +4 -0
- package/dist/commands/health/types-windows.js +4 -0
- package/dist/commands/health/types.js +26 -21
- package/dist/commands/health/windows.js +2 -3
- package/dist/commands/health.js +296 -167
- package/dist/commands/improve/anti-collapse.js +5 -5
- package/dist/commands/improve/autonomy-gate.js +68 -0
- package/dist/commands/improve/collapse-detector.js +65 -52
- package/dist/commands/improve/consolidate/chunking.js +9 -7
- package/dist/commands/improve/consolidate/eligibility.js +1 -23
- package/dist/commands/improve/consolidate/merge.js +4 -0
- package/dist/commands/improve/consolidate.js +454 -1354
- package/dist/commands/improve/content-hash.js +39 -0
- package/dist/commands/improve/distill/content-repair.js +4 -10
- package/dist/commands/improve/distill/promote-memory.js +89 -64
- package/dist/commands/improve/distill/quality-gate.js +118 -42
- package/dist/commands/improve/distill-guards.js +1 -1
- package/dist/commands/improve/distill-promotion-policy.js +33 -888
- package/dist/commands/improve/distill.js +607 -363
- package/dist/commands/improve/eligibility.js +165 -79
- package/dist/commands/improve/extract-cli.js +35 -126
- package/dist/commands/improve/extract-prompt.js +6 -35
- package/dist/commands/improve/extract.js +640 -391
- package/dist/commands/improve/feedback-valence.js +2 -12
- package/dist/commands/improve/improve-cli.js +134 -135
- package/dist/commands/improve/improve-result-file.js +30 -50
- package/dist/commands/improve/improve-run-types.js +4 -0
- package/dist/commands/improve/improve-strategies.js +135 -0
- package/dist/commands/improve/improve.js +904 -701
- package/dist/commands/improve/locks.js +64 -111
- package/dist/commands/improve/loop-stages.js +1110 -923
- package/dist/commands/improve/memory/derived-ref.js +124 -0
- package/dist/commands/improve/memory/memory-belief.js +79 -7
- package/dist/commands/improve/memory/memory-contradiction-detect.js +49 -52
- package/dist/commands/improve/memory/memory-improve.js +25 -37
- package/dist/commands/improve/outcome-loop.js +25 -88
- package/dist/commands/improve/preparation.js +1034 -813
- package/dist/commands/improve/proactive-maintenance.js +34 -9
- package/dist/commands/improve/proposal-envelope.js +31 -0
- package/dist/commands/improve/reflect.js +983 -794
- package/dist/commands/improve/run-context.js +119 -0
- package/dist/commands/improve/salience.js +24 -127
- package/dist/commands/improve/session-asset.js +7 -3
- package/dist/commands/improve/shared.js +14 -34
- package/dist/commands/improve/source-identity.js +28 -0
- package/dist/commands/improve/triage.js +20 -17
- package/dist/commands/lint/base-linter.js +340 -313
- package/dist/commands/lint/env-key-rules.js +31 -47
- package/dist/commands/lint/index.js +185 -30
- package/dist/commands/{events.js → log.js} +28 -38
- package/dist/commands/migrate-cli.js +54 -0
- package/dist/commands/migration-tool.js +55 -0
- package/dist/commands/observability-cli.js +70 -208
- package/dist/commands/proposal/diff-format.js +50 -0
- package/dist/commands/proposal/drain-policies.js +0 -6
- package/dist/commands/proposal/drain.js +91 -40
- package/dist/commands/proposal/proposal-cli.js +134 -132
- package/dist/commands/proposal/proposal-types.js +56 -0
- package/dist/commands/proposal/proposal.js +83 -65
- package/dist/commands/proposal/propose-cli.js +88 -0
- package/dist/commands/proposal/propose.js +105 -88
- package/dist/commands/proposal/repository.js +1303 -278
- package/dist/commands/proposal/validators/proposal-quality-validators.js +16 -6
- package/dist/commands/proposal/validators/proposal-validators.js +61 -12
- package/dist/commands/proposal/validators/proposals.js +6 -8
- package/dist/commands/read/curate.js +78 -73
- package/dist/commands/read/knowledge.js +510 -13
- package/dist/commands/read/registry-search.js +2 -2
- package/dist/commands/read/remember-cli.js +84 -15
- package/dist/commands/read/search-cli.js +203 -96
- package/dist/commands/read/search.js +126 -94
- package/dist/commands/read/show.js +226 -250
- package/dist/commands/registry-cli.js +34 -60
- package/dist/commands/remember.js +18 -57
- package/dist/commands/sources/add-cli.js +104 -49
- package/dist/commands/sources/bundle-cli.js +166 -0
- package/dist/commands/sources/bundle-config-ops.js +63 -0
- package/dist/commands/sources/info.js +27 -15
- package/dist/commands/sources/init.js +30 -40
- package/dist/commands/sources/installed-stashes.js +469 -172
- package/dist/commands/sources/migration-help.js +7 -4
- package/dist/commands/sources/schema-repair.js +10 -9
- package/dist/commands/sources/self-update.js +182 -121
- package/dist/commands/sources/source-add.js +169 -178
- package/dist/commands/sources/source-clone.js +144 -41
- package/dist/commands/sources/source-manage.js +94 -59
- package/dist/commands/sources/sources-cli.js +64 -205
- package/dist/commands/sources/stash-cli.js +91 -54
- package/dist/commands/sources/stash-skeleton.js +1 -1
- package/dist/commands/tasks/tasks-cli.js +106 -104
- package/dist/commands/tasks/tasks.js +445 -262
- package/dist/commands/workflow-cli.js +232 -121
- package/dist/core/action-contributors.js +1 -1
- package/dist/core/activation-policy.js +49 -0
- package/dist/core/adapter/adapters/agent-skills-adapter.js +181 -0
- package/dist/core/adapter/adapters/akm-adapter.js +528 -0
- package/dist/core/adapter/adapters/akm-lint.js +392 -0
- package/dist/core/adapter/adapters/akm-metadata.js +387 -0
- package/dist/core/adapter/adapters/akm-task-adapter.js +149 -0
- package/dist/core/adapter/adapters/akm-workflow-adapter.js +180 -0
- package/dist/core/adapter/adapters/claude-adapter.js +61 -0
- package/dist/core/adapter/adapters/dotenv-adapter.js +187 -0
- package/dist/core/adapter/adapters/generic-files-adapter.js +119 -0
- package/dist/core/adapter/adapters/index.js +80 -0
- package/dist/core/adapter/adapters/llm-wiki-adapter.js +419 -0
- package/dist/core/adapter/adapters/okf-adapter.js +391 -0
- package/dist/core/adapter/adapters/opencode-adapter.js +68 -0
- package/dist/core/adapter/adapters/shared.js +286 -0
- package/dist/core/adapter/adapters/tool-dir-shared.js +217 -0
- package/dist/core/adapter/adapters/website-snapshot-adapter.js +155 -0
- package/dist/core/adapter/bundle-adapter.js +4 -0
- package/dist/core/adapter/detect-adapter.js +17 -0
- package/dist/core/adapter/recognize-match.js +44 -0
- package/dist/core/adapter/registry.js +56 -0
- package/dist/core/adapter/types.js +4 -0
- package/dist/core/asset/akm-markdown.js +30 -0
- package/dist/core/asset/asset-placement.js +243 -0
- package/dist/core/asset/asset-ref.js +110 -79
- package/dist/core/asset/asset-serialize.js +20 -0
- package/dist/core/asset/frontmatter.js +28 -12
- package/dist/core/asset/markdown.js +40 -51
- package/dist/core/asset/resolve-ref.js +274 -0
- package/dist/core/asset/stash-meta.js +2 -2
- package/dist/core/bundle-id.js +51 -0
- package/dist/core/common.js +281 -86
- package/dist/core/config/config-io.js +42 -128
- package/dist/core/config/config-schema.js +233 -834
- package/dist/core/config/config-sources.js +162 -39
- package/dist/core/config/config-types.js +16 -11
- package/dist/core/config/config-version.js +29 -0
- package/dist/core/config/config-walker.js +126 -37
- package/dist/core/config/config.js +154 -331
- package/dist/core/config/deep-merge.js +41 -0
- package/dist/core/config/engine-semantics.js +28 -0
- package/dist/core/config/experimental.js +21 -0
- package/dist/core/config/schema/embedding.js +38 -0
- package/dist/core/config/schema/engines.js +116 -0
- package/dist/core/config/schema/experimental.js +47 -0
- package/dist/core/config/schema/feedback.js +31 -0
- package/dist/core/config/schema/improve-processes.js +389 -0
- package/dist/core/config/schema/improve.js +94 -0
- package/dist/core/config/schema/index-config.js +176 -0
- package/dist/core/config/schema/output.js +18 -0
- package/dist/core/config/schema/primitives.js +94 -0
- package/dist/core/config/schema/search.js +30 -0
- package/dist/core/config/schema/setup.js +18 -0
- package/dist/core/config/schema/sources-bundles.js +169 -0
- package/dist/core/config/schema/workflow.js +29 -0
- package/dist/core/env-secret-ref.js +155 -20
- package/dist/core/errors.js +17 -15
- package/dist/core/events-types.js +4 -0
- package/dist/core/events.js +46 -128
- package/dist/core/extra-params.js +62 -0
- package/dist/core/file-change.js +17 -0
- package/dist/core/file-lock.js +202 -57
- package/dist/core/fs-txn.js +392 -0
- package/dist/core/git-message.js +59 -0
- package/dist/core/improve-result.js +167 -0
- package/dist/core/json-schema.js +142 -0
- package/dist/core/lesson-lint.js +1 -17
- package/dist/core/logs-db.js +1 -1
- package/dist/core/maintenance-barrier.js +135 -0
- package/dist/core/migration-operation.js +44 -0
- package/dist/core/mutation-target.js +78 -0
- package/dist/core/paths.js +22 -25
- package/dist/core/platform.js +10 -0
- package/dist/core/recognition-util.js +128 -0
- package/dist/core/redaction.js +392 -0
- package/dist/core/standards/resolve-standards-context.js +36 -65
- package/dist/core/standards/resolve-stash-standards.js +2 -2
- package/dist/core/standards/resolve-type-conventions.js +5 -5
- package/dist/core/state/migrations.js +242 -11
- package/dist/core/state-db.js +98 -10
- package/dist/core/structured.js +1 -1
- package/dist/core/subprocess.js +303 -0
- package/dist/core/text-truncation.js +9 -5
- package/dist/core/time.js +20 -0
- package/dist/core/type-presentation.js +130 -0
- package/dist/core/warn.js +0 -3
- package/dist/core/write-source.js +834 -118
- package/dist/indexer/bundle-identity-guard.js +92 -0
- package/dist/indexer/db/graph-db.js +1 -25
- package/dist/indexer/db/llm-cache.js +1 -1
- package/dist/indexer/ensure-index.js +30 -9
- package/dist/indexer/graph/graph-boost.js +9 -30
- package/dist/indexer/graph/graph-extraction.js +41 -27
- package/dist/indexer/graph/graph-types.js +4 -0
- package/dist/indexer/index-writer-lock.js +93 -49
- package/dist/indexer/index-written-assets.js +100 -53
- package/dist/indexer/indexer.js +746 -329
- package/dist/indexer/init.js +18 -25
- package/dist/indexer/installations.js +142 -0
- package/dist/indexer/passes/dir-staleness.js +18 -10
- package/dist/indexer/passes/memory-inference.js +25 -15
- package/dist/indexer/passes/metadata.js +412 -243
- package/dist/indexer/scan/doc-to-entry.js +160 -0
- package/dist/indexer/scan/drain-dir.js +134 -0
- package/dist/indexer/search/db-search.js +292 -108
- package/dist/indexer/search/fts-query.js +64 -0
- package/dist/indexer/search/ranking-contributors.js +145 -25
- package/dist/indexer/search/ranking-types.js +4 -0
- package/dist/indexer/search/ranking.js +28 -71
- package/dist/indexer/search/search-attribution.js +67 -0
- package/dist/indexer/search/search-fields.js +18 -3
- package/dist/indexer/search/search-hit-enrichers.js +30 -40
- package/dist/indexer/search/search-source.js +157 -111
- package/dist/indexer/search/semantic-status.js +4 -1
- package/dist/indexer/usage/usage-events.js +10 -30
- package/dist/indexer/walk/file-context.js +3 -45
- package/dist/indexer/walk/matchers.js +42 -34
- package/dist/indexer/walk/path-resolver.js +11 -5
- package/dist/indexer/walk/walker.js +42 -14
- package/dist/integrations/agent/builder-shared.js +7 -0
- package/dist/integrations/agent/builders.js +5 -56
- package/dist/integrations/agent/config.js +3 -143
- package/dist/integrations/agent/detect.js +17 -2
- package/dist/integrations/agent/engine-resolution.js +231 -0
- package/dist/integrations/agent/index.js +1 -2
- package/dist/integrations/agent/model-aliases.js +16 -2
- package/dist/integrations/agent/profiles.js +36 -62
- package/dist/integrations/agent/prompts.js +46 -18
- package/dist/integrations/agent/runner-dispatch.js +93 -4
- package/dist/integrations/agent/runner.js +76 -208
- package/dist/integrations/agent/spawn.js +88 -196
- package/dist/integrations/harnesses/aider/agent-builder.js +114 -0
- package/dist/integrations/harnesses/aider/index.js +48 -0
- package/dist/integrations/harnesses/aider/result-extractor.js +53 -0
- package/dist/integrations/harnesses/amazonq/agent-builder.js +147 -0
- package/dist/integrations/harnesses/amazonq/index.js +45 -0
- package/dist/integrations/harnesses/amazonq/result-extractor.js +48 -0
- package/dist/integrations/harnesses/claude/agent-builder.js +46 -8
- package/dist/integrations/harnesses/claude/config-import.js +1 -3
- package/dist/integrations/harnesses/claude/index.js +24 -35
- package/dist/integrations/harnesses/claude/result-extractor.js +52 -0
- package/dist/integrations/harnesses/claude/session-log.js +27 -75
- package/dist/integrations/harnesses/codex/agent-builder.js +138 -0
- package/dist/integrations/harnesses/codex/index.js +52 -0
- package/dist/integrations/harnesses/codex/result-extractor.js +73 -0
- package/dist/integrations/harnesses/copilot/agent-builder.js +122 -0
- package/dist/integrations/harnesses/copilot/index.js +48 -0
- package/dist/integrations/harnesses/copilot/result-extractor.js +151 -0
- package/dist/integrations/harnesses/gemini/agent-builder.js +120 -0
- package/dist/integrations/harnesses/gemini/index.js +48 -0
- package/dist/integrations/harnesses/gemini/result-extractor.js +121 -0
- package/dist/integrations/harnesses/ids.js +24 -0
- package/dist/integrations/harnesses/index.js +54 -34
- package/dist/integrations/harnesses/opencode/agent-builder.js +23 -5
- package/dist/integrations/harnesses/opencode/config-import.js +1 -3
- package/dist/integrations/harnesses/opencode/index.js +14 -32
- package/dist/integrations/harnesses/opencode/session-log.js +67 -125
- package/dist/integrations/harnesses/opencode-sdk/harness.js +51 -0
- package/dist/integrations/harnesses/opencode-sdk/sdk-runner.js +681 -108
- package/dist/integrations/harnesses/openhands/agent-builder.js +128 -0
- package/dist/integrations/harnesses/openhands/index.js +48 -0
- package/dist/integrations/harnesses/openhands/result-extractor.js +103 -0
- package/dist/integrations/harnesses/pi/agent-builder.js +97 -0
- package/dist/integrations/harnesses/pi/index.js +45 -0
- package/dist/integrations/harnesses/pi/result-extractor.js +135 -0
- package/dist/integrations/harnesses/shared.js +17 -0
- package/dist/integrations/harnesses/types.js +43 -32
- package/dist/integrations/lockfile.js +211 -24
- package/dist/integrations/session-logs/index.js +36 -39
- package/dist/integrations/session-logs/provider-base.js +113 -0
- package/dist/llm/client.js +182 -110
- package/dist/llm/embedders/deterministic.js +2 -2
- package/dist/llm/embedders/remote.js +21 -9
- package/dist/llm/feature-gate.js +17 -57
- package/dist/llm/graph-extract.js +12 -13
- package/dist/llm/index-passes.js +8 -42
- package/dist/llm/memory-infer.js +144 -1
- package/dist/llm/metadata-enhance.js +45 -30
- package/dist/llm/structured-call.js +16 -8
- package/dist/llm/usage-persist.js +30 -5
- package/dist/llm/usage-telemetry.js +59 -6
- package/dist/output/cli-hints.js +1 -2
- package/dist/output/command-registry.js +27 -0
- package/dist/output/context.js +22 -7
- package/dist/output/format-exempt.js +80 -0
- package/dist/output/generic-render.js +251 -0
- package/dist/output/html-render.js +11 -16
- package/dist/output/render-registry.js +57 -0
- package/dist/output/renderers.js +14 -279
- package/dist/output/shapes/curate.js +10 -1
- package/dist/output/shapes/events.js +12 -7
- package/dist/output/shapes/helpers.js +58 -84
- package/dist/output/shapes/passthrough.js +11 -39
- package/dist/output/shapes/proposal/producer.js +15 -7
- package/dist/output/shapes/registry.js +12 -6
- package/dist/output/shapes.js +0 -9
- package/dist/output/text/{init.js → bundle-create.js} +3 -1
- package/dist/output/text/bundle-show.js +7 -0
- package/dist/output/text/command-format.js +562 -0
- package/dist/output/text/env.js +1 -3
- package/dist/output/text/events.js +8 -7
- package/dist/output/text/helpers.js +15 -1164
- package/dist/output/text/proposal/producer.js +4 -2
- package/dist/output/text/proposal-format.js +202 -0
- package/dist/output/text/registry-commands.js +1 -2
- package/dist/output/text/registry.js +12 -6
- package/dist/output/text/show-directives.js +117 -0
- package/dist/output/text/show-format.js +103 -0
- package/dist/output/text/sync.js +5 -0
- package/dist/output/text/workflow-format.js +332 -0
- package/dist/output/text/workflow.js +3 -2
- package/dist/output/text.js +10 -19
- package/dist/registry/factory.js +4 -6
- package/dist/registry/origin-resolve.js +16 -27
- package/dist/registry/providers/skills-sh.js +3 -3
- package/dist/registry/providers/static-index.js +15 -25
- package/dist/registry/resolve.js +43 -94
- package/dist/registry/semver.js +43 -0
- package/dist/runtime.js +81 -12
- package/dist/scripts/akm-migrate.js +35529 -0
- package/dist/setup/detect.js +5 -7
- package/dist/setup/detected-engines.js +136 -0
- package/dist/setup/engine-config.js +100 -0
- package/dist/setup/registry-stash-loader.js +3 -3
- package/dist/setup/semantic-assets.js +12 -9
- package/dist/setup/setup.js +444 -208
- package/dist/setup/steps/connection-shared.js +120 -0
- package/dist/setup/steps/connection.js +108 -305
- package/dist/setup/steps/platforms.js +13 -12
- package/dist/setup/steps/semantic.js +15 -3
- package/dist/setup/steps/sources.js +21 -15
- package/dist/setup/steps/stashdir.js +6 -4
- package/dist/setup/steps/tasks.js +236 -119
- package/dist/setup/steps.js +3 -2
- package/dist/sources/freshness.js +39 -0
- package/dist/sources/provider-factory.js +11 -17
- package/dist/sources/providers/filesystem.js +2 -3
- package/dist/sources/providers/git-install.js +278 -34
- package/dist/sources/providers/git-provider.js +54 -56
- package/dist/sources/providers/git-stash.js +420 -91
- package/dist/sources/providers/git.js +2 -2
- package/dist/sources/providers/npm.js +16 -19
- package/dist/sources/providers/provider-utils.js +47 -22
- package/dist/sources/providers/sync-from-ref.js +3 -9
- package/dist/sources/providers/website.js +2 -2
- package/dist/sources/resolve.js +11 -10
- package/dist/sources/snapshot-fetchers/types.js +4 -0
- package/dist/sources/{website-ingest.js → snapshot-fetchers/website-ingest.js} +110 -41
- package/dist/storage/database.js +60 -4
- package/dist/storage/engines/sqlite-migrations.js +156 -5
- package/dist/storage/locations.js +1 -2
- package/dist/storage/repositories/canaries-repository.js +1 -1
- package/dist/storage/repositories/events-repository.js +51 -11
- package/dist/storage/repositories/improve-runs-repository.js +6 -32
- package/dist/storage/repositories/index-connection.js +79 -0
- package/dist/storage/repositories/index-db.js +4 -3
- package/dist/storage/repositories/index-entries-repository.js +863 -0
- package/dist/{indexer/db/entry-mapper.js → storage/repositories/index-entry-mapper.js} +19 -2
- package/dist/storage/repositories/index-entry-types.js +4 -0
- package/dist/storage/repositories/index-fts-repository.js +167 -0
- package/dist/storage/repositories/index-llm-cache-repository.js +108 -0
- package/dist/storage/repositories/index-meta-repository.js +49 -0
- package/dist/{indexer/db/schema.js → storage/repositories/index-schema.js} +226 -100
- package/dist/storage/repositories/index-sql.js +12 -0
- package/dist/storage/repositories/index-utility-repository.js +356 -0
- package/dist/storage/repositories/index-vec-repository.js +250 -0
- package/dist/storage/repositories/outcome-repository.js +119 -0
- package/dist/storage/repositories/proposals-repository.js +317 -75
- package/dist/storage/repositories/registry-cache.js +1 -1
- package/dist/storage/repositories/salience-repository.js +172 -0
- package/dist/storage/repositories/task-history-repository.js +110 -3
- package/dist/storage/repositories/workflow-runs-repository.js +240 -19
- package/dist/tasks/backends/cron.js +169 -46
- package/dist/tasks/backends/exec-utils.js +76 -3
- package/dist/tasks/backends/index.js +6 -9
- package/dist/tasks/backends/launchd.js +292 -55
- package/dist/tasks/backends/schtasks.js +557 -70
- package/dist/tasks/backends/types.js +4 -0
- package/dist/tasks/command-executable.js +93 -0
- package/dist/tasks/embedded.js +56 -38
- package/dist/tasks/parser.js +156 -64
- package/dist/tasks/resolve-akm-bin.js +144 -51
- package/dist/tasks/runner.js +377 -209
- package/dist/tasks/schedule.js +108 -19
- package/dist/tasks/scheduler-invocation.js +296 -0
- package/dist/tasks/schema.js +1 -1
- package/dist/tasks/task-id.js +35 -0
- package/dist/tasks/validator.js +30 -16
- package/dist/text-import-hook.mjs +1 -1
- package/dist/workflows/authoring/authoring.js +104 -43
- package/dist/workflows/authoring/scope-key.js +1 -1
- package/dist/workflows/cli.js +0 -16
- package/dist/workflows/concurrency-policy.js +15 -0
- package/dist/workflows/exec/brief.js +450 -0
- package/dist/workflows/exec/frozen-judge.js +47 -0
- package/dist/workflows/exec/native-executor.js +1038 -0
- package/dist/workflows/exec/param-secrets.js +115 -0
- package/dist/workflows/exec/report.js +1460 -0
- package/dist/workflows/exec/run-workflow.js +602 -0
- package/dist/workflows/exec/scheduler.js +71 -0
- package/dist/workflows/exec/step-work.js +1190 -0
- package/dist/workflows/exec/unit-writer.js +23 -0
- package/dist/workflows/exec/workflow-engine-gate.js +67 -0
- package/dist/workflows/exec/worktree.js +171 -0
- package/dist/workflows/ir/compile.js +246 -0
- package/dist/workflows/ir/freeze.js +233 -0
- package/dist/workflows/ir/params.js +54 -0
- package/dist/workflows/ir/plan-hash.js +68 -0
- package/dist/workflows/ir/schema.js +540 -0
- package/dist/workflows/parser.js +878 -304
- package/dist/workflows/program/expressions.js +181 -0
- package/dist/workflows/program/schema.js +51 -0
- package/dist/workflows/renderer.js +100 -45
- package/dist/workflows/resource-limits.js +22 -0
- package/dist/workflows/runtime/agent-identity.js +59 -14
- package/dist/workflows/runtime/checkin.js +1 -1
- package/dist/workflows/runtime/plan-classifier.js +131 -0
- package/dist/workflows/runtime/runs.js +376 -119
- package/dist/workflows/runtime/unit-checkin.js +45 -0
- package/dist/workflows/runtime/unit-phases.js +20 -0
- package/dist/workflows/runtime/workflow-asset-loader.js +241 -40
- package/dist/workflows/schema.js +1 -11
- package/dist/workflows/validate-summary.js +2 -3
- package/dist/workflows/validator.js +52 -30
- package/docs/README.md +42 -78
- package/docs/migration/README.md +8 -0
- package/docs/migration/release-notes/0.6.0.md +1 -1
- package/docs/migration/release-notes/0.7.0.md +9 -8
- package/docs/migration/release-notes/0.9.0.md +158 -14
- package/docs/migration/v0.7-to-v0.8.md +46 -47
- package/docs/migration/v0.8-to-v0.9.md +844 -0
- package/docs/reference/README.md +12 -0
- package/docs/reference/data-and-telemetry.md +333 -0
- package/package.json +21 -17
- package/schemas/akm-asset-envelope.json +93 -0
- package/schemas/akm-config.json +4636 -0
- package/schemas/akm-task.json +87 -0
- package/schemas/akm-workflow.json +373 -0
- package/dist/akm-migrate-storage +0 -38
- package/dist/assets/help/help-accept.md +0 -12
- package/dist/assets/help/help-improve.md +0 -84
- package/dist/assets/help/help-proposals.md +0 -17
- package/dist/assets/help/help-propose.md +0 -17
- package/dist/assets/help/help-reject.md +0 -11
- package/dist/assets/profiles/frequent.json +0 -13
- package/dist/assets/profiles/recombine-only.json +0 -21
- package/dist/assets/profiles/reflect-distill.json +0 -30
- package/dist/assets/profiles/synthesize.json +0 -15
- package/dist/assets/prompts/procedural-system.md +0 -44
- package/dist/assets/prompts/recombine-system.md +0 -40
- package/dist/assets/prompts/staleness-detect-system.md +0 -6
- package/dist/assets/tasks/core/backup.yml +0 -4
- package/dist/assets/tasks/graph-refresh-weekly.yml +0 -10
- package/dist/assets/templates/html/default.html +0 -78
- package/dist/assets/templates/html/vendor/echarts.min.js +0 -45
- package/dist/assets/wiki/index-template.md +0 -12
- package/dist/assets/wiki/ingest-workflow-template.md +0 -83
- package/dist/assets/wiki/log-template.md +0 -8
- package/dist/assets/wiki/schema-template.md +0 -61
- package/dist/cli/config-migrate.js +0 -150
- package/dist/cli/config-validate.js +0 -39
- package/dist/commands/graph/graph-cli.js +0 -124
- package/dist/commands/graph/graph.js +0 -487
- package/dist/commands/improve/calibration.js +0 -161
- package/dist/commands/improve/dedup.js +0 -482
- package/dist/commands/improve/extract-watch.js +0 -140
- package/dist/commands/improve/hot-probation.js +0 -45
- package/dist/commands/improve/improve-auto-accept.js +0 -276
- package/dist/commands/improve/improve-profiles.js +0 -168
- package/dist/commands/improve/procedural.js +0 -398
- package/dist/commands/improve/recombine.js +0 -818
- package/dist/commands/improve/schema-similarity-gate.js +0 -168
- package/dist/commands/lint/agent-linter.js +0 -44
- package/dist/commands/lint/command-linter.js +0 -44
- package/dist/commands/lint/default-linter.js +0 -16
- package/dist/commands/lint/fact-linter.js +0 -39
- package/dist/commands/lint/knowledge-linter.js +0 -16
- package/dist/commands/lint/memory-linter.js +0 -61
- package/dist/commands/lint/registry.js +0 -41
- package/dist/commands/lint/skill-linter.js +0 -45
- package/dist/commands/lint/task-linter.js +0 -50
- package/dist/commands/lint/workflow-linter.js +0 -81
- package/dist/commands/proposal/legacy-import.js +0 -115
- package/dist/commands/sources/history.js +0 -196
- package/dist/commands/tasks/default-tasks.js +0 -186
- package/dist/commands/wiki-cli.js +0 -292
- package/dist/core/asset/asset-registry.js +0 -76
- package/dist/core/asset/asset-spec.js +0 -259
- package/dist/core/config/config-migration.js +0 -602
- package/dist/core/deep-merge.js +0 -38
- package/dist/core/eval/rank-metrics.js +0 -113
- package/dist/core/ripgrep/install.js +0 -163
- package/dist/core/ripgrep/resolve.js +0 -81
- package/dist/indexer/db/db.js +0 -1413
- package/dist/indexer/manifest.js +0 -170
- package/dist/indexer/passes/metadata-contributors.js +0 -31
- package/dist/indexer/usage/unmigrated-vaults-guard.js +0 -94
- package/dist/integrations/harnesses/opencode-sdk/index.js +0 -49
- package/dist/llm/call-ai.js +0 -62
- package/dist/llm/memory-infer-impl.js +0 -138
- package/dist/output/shapes/distill.js +0 -14
- package/dist/output/shapes/history.js +0 -11
- package/dist/output/text/distill.js +0 -6
- package/dist/output/text/enable-disable.js +0 -8
- package/dist/output/text/history.js +0 -6
- package/dist/output/text/wiki.js +0 -16
- package/dist/registry/build-index.js +0 -386
- package/dist/scripts/migrate-storage.js +0 -19108
- package/dist/scripts/migrations/import-fs-improve-runs-to-db.js +0 -9411
- package/dist/scripts/migrations/v16-to-v17.js +0 -141
- package/dist/setup/legacy-config.js +0 -106
- package/dist/storage/repositories/consolidation-repository.js +0 -38
- package/dist/storage/repositories/recombine-repository.js +0 -213
- package/dist/wiki/wiki-templates.js +0 -15
- package/dist/wiki/wiki.js +0 -1012
- package/dist/workflows/db.js +0 -215
- package/docs/data-and-telemetry.md +0 -226
- /package/dist/sources/{wiki-fetchers → snapshot-fetchers}/registry.js +0 -0
- /package/dist/sources/{wiki-fetchers → snapshot-fetchers}/youtube.js +0 -0
|
@@ -3,196 +3,104 @@
|
|
|
3
3
|
// file, You can obtain one at https://mozilla.org/MPL/2.0/.
|
|
4
4
|
import fs from "node:fs";
|
|
5
5
|
import path from "node:path";
|
|
6
|
-
import {
|
|
6
|
+
import { makeBundleRef } from "../../core/asset/asset-ref.js";
|
|
7
7
|
import { parseFrontmatter } from "../../core/asset/frontmatter.js";
|
|
8
|
+
import { typeNameFromConceptId } from "../../core/asset/resolve-ref.js";
|
|
8
9
|
import { daysToMs } from "../../core/common.js";
|
|
9
|
-
import {
|
|
10
|
-
import { rethrowIfTestIsolationError } from "../../core/errors.js";
|
|
10
|
+
import { loadConfig } from "../../core/config/config.js";
|
|
11
|
+
import { ConfigError, rethrowIfTestIsolationError } from "../../core/errors.js";
|
|
11
12
|
import { appendEvent, readEvents } from "../../core/events.js";
|
|
12
13
|
import { openStateDatabase, withStateDb } from "../../core/state-db.js";
|
|
13
14
|
import { info, warn } from "../../core/warn.js";
|
|
14
|
-
import { closeDatabase, getRetrievalCounts, getZeroResultSearches, openExistingDatabase } from "../../indexer/db/db.js";
|
|
15
15
|
import { countUsageEventsByType } from "../../indexer/usage/usage-events.js";
|
|
16
|
+
import { materializeLlmRunnerConnection } from "../../integrations/agent/runner.js";
|
|
16
17
|
import { getAvailableHarnesses } from "../../integrations/session-logs/index.js";
|
|
17
18
|
import { withLlmStage } from "../../llm/usage-telemetry.js";
|
|
18
|
-
import {
|
|
19
|
-
import {
|
|
19
|
+
import { closeDatabase, openExistingDatabase } from "../../storage/repositories/index-connection.js";
|
|
20
|
+
import { getZeroResultSearches } from "../../storage/repositories/index-entries-repository.js";
|
|
21
|
+
import { getRetrievalCounts } from "../../storage/repositories/index-utility-repository.js";
|
|
22
|
+
import { listStateProposals } from "../../storage/repositories/proposals-repository.js";
|
|
20
23
|
import { akmLint } from "../lint/index.js";
|
|
21
|
-
import { getProposal, listProposals } from "../proposal/repository.js";
|
|
22
24
|
import { runSchemaRepairPass } from "../sources/schema-repair.js";
|
|
23
|
-
import {
|
|
25
|
+
import { isAutonomyLaneAllowed } from "./autonomy-gate.js";
|
|
24
26
|
import { akmConsolidate } from "./consolidate.js";
|
|
25
27
|
// Eligibility / candidate-selection predicates live in ./eligibility.
|
|
26
28
|
import { buildLatestFeedbackTsMap, buildLatestProposalTsMap, buildUtilityMap, dedupeRefs, findAssetFilePath, isDistillCandidateRef, isLessonCandidate, isSignalDeltaEligible, } from "./eligibility.js";
|
|
27
29
|
import { akmExtract, countNewExtractCandidates } from "./extract.js";
|
|
28
30
|
import { computeValenceScore, FEEDBACK_WEIGHT, UTILITY_WEIGHT } from "./feedback-valence.js";
|
|
29
|
-
import { makeGateConfig, resolveExtractConfidence, runAutoAcceptGate } from "./improve-auto-accept.js";
|
|
30
|
-
import { resolveProcessEnabled } from "./improve-profiles.js";
|
|
31
31
|
import { applyMemoryCleanup } from "./memory/memory-improve.js";
|
|
32
32
|
import { computeProxyAdequacy, getAllAssetOutcomes, getOutcomeScoresByRef, OUTCOME_SCORE_MAX, outcomeScoreToSalience, updateAssetOutcome, } from "./outcome-loop.js";
|
|
33
33
|
import { DEFAULT_DUE_DAYS, DEFAULT_MAX_PER_RUN, selectProactiveMaintenanceRefs } from "./proactive-maintenance.js";
|
|
34
|
-
import { buildRankChangeReport, computeSalience, getAllRankScores, getAssetSalience,
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
* Reads `improve.calibration` from config. When `autoTune` is enabled, computes
|
|
42
|
-
* the calibration of recent gate decisions, derives a bounded threshold
|
|
43
|
-
* adjustment (clamped into the configured band, capped per step), logs it, and
|
|
44
|
-
* records a `calibration_autotune` event. Returns the new threshold (integer
|
|
45
|
-
* 0-100) when an adjustment was made, or `undefined` to leave the caller's
|
|
46
|
-
* threshold unchanged.
|
|
47
|
-
*
|
|
48
|
-
* WS-4 change: accepts an optional `phase` parameter. When provided, the tuned
|
|
49
|
-
* threshold is persisted to `improve_gate_thresholds` (state.db Migration 012)
|
|
50
|
-
* keyed by phase so `makeGateConfig` can read it back on the next run and each
|
|
51
|
-
* phase maintains its own calibrated threshold rather than a shared global.
|
|
52
|
-
*
|
|
53
|
-
* WS-4 ceiling: `maxThreshold` defaults to 85 (not 100) to prevent the gate
|
|
54
|
-
* converging to pure exploitation and shutting down Gap-3/4 novelty throughput.
|
|
55
|
-
*
|
|
56
|
-
* DEFAULT OFF: with no `improve.calibration` block (or `autoTune: false`) this
|
|
57
|
-
* returns `undefined` immediately, so the gate threshold is unchanged and
|
|
58
|
-
* behaviour is byte-identical to today.
|
|
59
|
-
*/
|
|
60
|
-
export function maybeAutoTuneThreshold(currentThreshold, config, stateDbPath, ctx, phase) {
|
|
61
|
-
const cal = config.improve?.calibration;
|
|
62
|
-
if (!cal?.autoTune)
|
|
63
|
-
return undefined;
|
|
64
|
-
// WS-4: default maxThreshold is now 85 (ceiling to prevent pure exploitation).
|
|
65
|
-
// Callers that explicitly set maxThreshold in config override this default.
|
|
66
|
-
const tuneConfig = {
|
|
67
|
-
autoTune: true,
|
|
68
|
-
minThreshold: cal.minThreshold ?? 0,
|
|
69
|
-
maxThreshold: cal.maxThreshold ?? 85,
|
|
70
|
-
maxStep: cal.maxStep ?? 5,
|
|
71
|
-
minSamples: cal.minSamples ?? 20,
|
|
72
|
-
targetAcceptRate: cal.targetAcceptRate ?? 0.9,
|
|
73
|
-
};
|
|
74
|
-
// Defensive: an inverted band disables tuning rather than clamping to nonsense.
|
|
75
|
-
if (tuneConfig.minThreshold > tuneConfig.maxThreshold)
|
|
76
|
-
return undefined;
|
|
77
|
-
const summary = withStateDb((db) => {
|
|
78
|
-
const allDecisions = listProposalGateDecisions(db);
|
|
79
|
-
// WS-4 fix: when called with a phase label, restrict calibration to that
|
|
80
|
-
// phase's decision pool so a reflect-dominated run cannot tighten the
|
|
81
|
-
// consolidate gate (or vice-versa). The gate field is `improve:<phase>`,
|
|
82
|
-
// matching what improve-auto-accept.ts stamps at line ~163.
|
|
83
|
-
const gateLabel = phase ? `improve:${phase}` : undefined;
|
|
84
|
-
const decisions = gateLabel ? allDecisions.filter((d) => d.gate === gateLabel) : allDecisions;
|
|
85
|
-
return summarizeCalibration(gateDecisionsToSamples(decisions));
|
|
86
|
-
}, { path: stateDbPath });
|
|
87
|
-
const result = computeThresholdAutoTune(currentThreshold, summary, tuneConfig);
|
|
88
|
-
if (!result.adjusted)
|
|
89
|
-
return undefined;
|
|
90
|
-
const appendEventFn = ctx?.appendEventFn ?? appendEvent;
|
|
91
|
-
const phaseLabel = phase ?? "global";
|
|
92
|
-
info(`[improve] calibration auto-tune (${phaseLabel}): threshold ${result.previousThreshold} -> ${result.newThreshold} ` +
|
|
93
|
-
`(${result.reason}; samples=${summary.samples}, acceptRate=${summary.overallAcceptRate}, ` +
|
|
94
|
-
`gap=${summary.calibrationGap}, band=[${tuneConfig.minThreshold},${tuneConfig.maxThreshold}])`);
|
|
95
|
-
try {
|
|
96
|
-
appendEventFn({
|
|
97
|
-
eventType: "calibration_autotune",
|
|
98
|
-
ref: "improve:calibration",
|
|
99
|
-
metadata: {
|
|
100
|
-
phase: phaseLabel,
|
|
101
|
-
previousThreshold: result.previousThreshold,
|
|
102
|
-
newThreshold: result.newThreshold,
|
|
103
|
-
delta: result.delta,
|
|
104
|
-
reason: result.reason,
|
|
105
|
-
samples: summary.samples,
|
|
106
|
-
overallAcceptRate: summary.overallAcceptRate,
|
|
107
|
-
calibrationGap: summary.calibrationGap,
|
|
108
|
-
minThreshold: tuneConfig.minThreshold,
|
|
109
|
-
maxThreshold: tuneConfig.maxThreshold,
|
|
110
|
-
},
|
|
111
|
-
});
|
|
112
|
-
}
|
|
113
|
-
catch (err) {
|
|
114
|
-
warn(`[improve] calibration auto-tune event not recorded: ${err instanceof Error ? err.message : String(err)}`);
|
|
115
|
-
}
|
|
116
|
-
// WS-4: Persist the per-phase threshold so makeGateConfig reads it on the
|
|
117
|
-
// next run. Best-effort — a write failure must not abort the improve run.
|
|
118
|
-
if (phase) {
|
|
119
|
-
try {
|
|
120
|
-
withStateDb((persistDb) => persistPhaseThreshold(persistDb, phase, result.newThreshold), {
|
|
121
|
-
path: stateDbPath,
|
|
122
|
-
});
|
|
123
|
-
}
|
|
124
|
-
catch (err) {
|
|
125
|
-
warn(`[improve] calibration auto-tune: failed to persist phase threshold for ${phase}: ${err instanceof Error ? err.message : String(err)}`);
|
|
126
|
-
}
|
|
34
|
+
import { buildRankChangeReport, computeSalience, getAllRankScores, getAssetSalience, getLastUseMsByRef, isContentEncodingRow, SALIENCE_NO_OP_DAMPEN_FACTOR, SALIENCE_NO_OP_DAMPEN_THRESHOLD, upsertAssetSalience, } from "./salience.js";
|
|
35
|
+
import { bareImproveRef, improveStateReadRefs } from "./source-identity.js";
|
|
36
|
+
function readAssetSalienceForImproveRef(db, ref, itemRef) {
|
|
37
|
+
for (const key of improveStateReadRefs(ref, itemRef)) {
|
|
38
|
+
const row = getAssetSalience(db, key);
|
|
39
|
+
if (row)
|
|
40
|
+
return row;
|
|
127
41
|
}
|
|
128
|
-
return
|
|
42
|
+
return undefined;
|
|
43
|
+
}
|
|
44
|
+
function readConsecutiveNoOpsForImproveRef(db, ref, itemRef) {
|
|
45
|
+
return readAssetSalienceForImproveRef(db, ref, itemRef)?.consecutive_no_ops ?? 0;
|
|
46
|
+
}
|
|
47
|
+
// ── Durable-state write keys ──────────────────────────────────────────────────
|
|
48
|
+
//
|
|
49
|
+
// The durable improve-state writers (salience + outcome) key by the resolved
|
|
50
|
+
// index entry's `item_ref` (`<bundle>//<conceptId>`) when the planner resolved
|
|
51
|
+
// one (`ImproveEligibleRef.itemRef`). Both writers share one key expression,
|
|
52
|
+
// `itemRefByRef.get(ref) ?? ref`.
|
|
53
|
+
//
|
|
54
|
+
// A direct `--scope <ref>` candidate does not flow through
|
|
55
|
+
// `collectEligibleRefsFromIndex`, so its conceptId `ref` is the write key.
|
|
56
|
+
//
|
|
57
|
+
// itemRefByRef is `ref → item_ref | undefined`, built once per pass from the
|
|
58
|
+
// candidate set.
|
|
59
|
+
/** `ref → item_ref | undefined` for a run's candidate set. */
|
|
60
|
+
function buildItemRefByRef(refs) {
|
|
61
|
+
const m = new Map();
|
|
62
|
+
for (const r of refs)
|
|
63
|
+
m.set(r.ref, r.itemRef);
|
|
64
|
+
return m;
|
|
65
|
+
}
|
|
66
|
+
/** Durable `asset_salience` write key: the entry's item_ref, else its conceptId `ref` (scope-ref fallback). */
|
|
67
|
+
function salienceWriteKey(ref, itemRefByRef) {
|
|
68
|
+
return itemRefByRef.get(ref) ?? ref;
|
|
69
|
+
}
|
|
70
|
+
/** Durable `asset_outcome` write key: the entry's item_ref, else its conceptId `ref` (scope-ref fallback). */
|
|
71
|
+
function outcomeWriteKey(ref, itemRefByRef) {
|
|
72
|
+
return itemRefByRef.get(ref) ?? ref;
|
|
73
|
+
}
|
|
74
|
+
/** Resolve an AKM asset type from a short or bundle-qualified conceptId. */
|
|
75
|
+
function assetTypeOf(ref) {
|
|
76
|
+
const tail = ref.includes("//") ? ref.slice(ref.indexOf("//") + 2) : ref;
|
|
77
|
+
return typeNameFromConceptId(tail)?.type ?? "";
|
|
129
78
|
}
|
|
130
79
|
/**
|
|
131
|
-
*
|
|
132
|
-
*
|
|
133
|
-
*
|
|
134
|
-
*
|
|
135
|
-
* 1. STRUCTURAL: this runs before extract in the improve pipeline (see
|
|
136
|
-
* `runImprovePreparationStage`). Consolidation therefore only ever judges
|
|
137
|
-
* PRIOR-run memories; current-run extract promotions are invisible to it.
|
|
138
|
-
*
|
|
139
|
-
* 2. SMARTER POOL-DELTA GATE: even among on-disk files, a memory whose only
|
|
140
|
-
* post-`lastConsolidateTs` mtime bump came from its OWN auto-accept
|
|
141
|
-
* promotion (i.e. it was just promoted by extract in the immediately
|
|
142
|
-
* preceding run and has not had a full improve cycle to settle) does NOT
|
|
143
|
-
* count as "work to do". We exclude those paths from the pool-delta check
|
|
144
|
-
* using the `promoted` events already emitted with each promotion's
|
|
145
|
-
* `assetPath`. A genuinely-settled prior memory — one edited by feedback,
|
|
146
|
-
* reflect, manual edit, or simply older than the last consolidate — still
|
|
147
|
-
* triggers the run. This is gate-option (a) from the issue (same-run /
|
|
148
|
-
* adjacent-run promotion exclusion), chosen over option (b) because there
|
|
149
|
-
* is no `extract_completed` event in the data model to gate against;
|
|
150
|
-
* `promoted` events with `assetPath` already carry exactly the signal we
|
|
151
|
-
* need, so the fix is non-invasive and provably correct.
|
|
80
|
+
* Evaluate the consolidation gate flags (volume trigger, #551 pool-delta
|
|
81
|
+
* cooldown, profile disable, #553 min-pool-size) up front, before any LLM call.
|
|
82
|
+
* Extracted verbatim from `runConsolidationPass` — logic is byte-identical.
|
|
152
83
|
*/
|
|
153
|
-
|
|
154
|
-
const { options, primaryStashDir, memorySummary, improveProfile,
|
|
155
|
-
const baseConfig = options.config ?? loadConfig();
|
|
84
|
+
function evaluateConsolidationEligibility(args) {
|
|
85
|
+
const { options, primaryStashDir, memorySummary, improveProfile, resolvedPlan } = args;
|
|
156
86
|
const MEMORY_VOLUME_THRESHOLD = options.memoryVolumeConsolidationThreshold ?? 100;
|
|
157
|
-
const hasLlm =
|
|
87
|
+
const hasLlm = resolvedPlan.processes.consolidate.runner !== null;
|
|
158
88
|
const volumeTriggered = typeof memorySummary.eligible === "number" && memorySummary.eligible > MEMORY_VOLUME_THRESHOLD && hasLlm;
|
|
159
|
-
// When volume triggers a consolidation pass, force-enable the consolidate
|
|
160
|
-
// process on the default improve profile so the gate accepts the run even
|
|
161
|
-
// if the user's config disabled it. We synthesise a new profile override
|
|
162
|
-
// rather than mutating connection settings.
|
|
163
|
-
const consolidationConfig = volumeTriggered
|
|
164
|
-
? {
|
|
165
|
-
...baseConfig,
|
|
166
|
-
profiles: {
|
|
167
|
-
...(baseConfig.profiles ?? {}),
|
|
168
|
-
improve: {
|
|
169
|
-
...(baseConfig.profiles?.improve ?? {}),
|
|
170
|
-
default: {
|
|
171
|
-
...(baseConfig.profiles?.improve?.default ?? {}),
|
|
172
|
-
processes: {
|
|
173
|
-
...(baseConfig.profiles?.improve?.default?.processes ?? {}),
|
|
174
|
-
consolidate: {
|
|
175
|
-
...(baseConfig.profiles?.improve?.default?.processes?.consolidate ?? {}),
|
|
176
|
-
enabled: true,
|
|
177
|
-
},
|
|
178
|
-
},
|
|
179
|
-
},
|
|
180
|
-
},
|
|
181
|
-
},
|
|
182
|
-
}
|
|
183
|
-
: baseConfig;
|
|
184
89
|
// 0.8.0 pool-delta gate for consolidate: re-eligible iff at least one
|
|
185
90
|
// memory file has been updated since the most recent successful
|
|
186
91
|
// consolidate_completed event. Time-based cooldowns produced the same
|
|
187
92
|
// synchronised-wave failure mode the reflect/distill cooldowns did; the
|
|
188
93
|
// pool-delta gate ties consolidation to actual work-to-do.
|
|
94
|
+
const sourceName = options.sourceName ?? options.writeTarget?.source.name ?? options.config?.defaultBundle ?? "stash";
|
|
189
95
|
const recentConsolidations = readEvents({ type: "consolidate_completed" });
|
|
190
96
|
const lastConsolidation = recentConsolidations.events
|
|
191
|
-
.filter((e) => e.metadata?.
|
|
97
|
+
.filter((e) => e.metadata?.source === sourceName && Number(e.metadata?.processed) > 0)
|
|
192
98
|
.sort((a, b) => new Date(b.ts ?? 0).getTime() - new Date(a.ts ?? 0).getTime())[0];
|
|
193
|
-
const lastConsolidateTs = lastConsolidation?.
|
|
99
|
+
const lastConsolidateTs = typeof lastConsolidation?.metadata?.completedThrough === "string"
|
|
100
|
+
? lastConsolidation.metadata.completedThrough
|
|
101
|
+
: lastConsolidation?.ts;
|
|
194
102
|
// #551 smarter gate: build the set of memory asset paths whose only delta
|
|
195
|
-
// since the last consolidate is their OWN
|
|
103
|
+
// since the last consolidate is their OWN promotion. Those files
|
|
196
104
|
// have not had a full improve cycle to settle, so they offer no merge /
|
|
197
105
|
// contradiction candidates yet — excluding them stops the gate firing on
|
|
198
106
|
// freshly-promoted single-source memories. We read `promoted` events emitted
|
|
@@ -235,21 +143,29 @@ export async function runConsolidationPass(args) {
|
|
|
235
143
|
if (!fs.existsSync(memoriesDir))
|
|
236
144
|
return false;
|
|
237
145
|
try {
|
|
238
|
-
|
|
239
|
-
|
|
240
|
-
|
|
241
|
-
const
|
|
242
|
-
|
|
243
|
-
|
|
244
|
-
|
|
245
|
-
|
|
246
|
-
|
|
247
|
-
|
|
248
|
-
|
|
249
|
-
|
|
250
|
-
|
|
146
|
+
const pending = [memoriesDir];
|
|
147
|
+
while (pending.length > 0) {
|
|
148
|
+
const current = pending.pop();
|
|
149
|
+
for (const entry of fs.readdirSync(current, { withFileTypes: true })) {
|
|
150
|
+
const filePath = path.join(current, entry.name);
|
|
151
|
+
if (entry.isDirectory()) {
|
|
152
|
+
pending.push(filePath);
|
|
153
|
+
continue;
|
|
154
|
+
}
|
|
155
|
+
if (!entry.isFile() || !entry.name.endsWith(".md"))
|
|
156
|
+
continue;
|
|
157
|
+
if (promotedSinceConsolidate.has(path.resolve(filePath)))
|
|
158
|
+
continue;
|
|
159
|
+
try {
|
|
160
|
+
if (fs.statSync(filePath).mtime.toISOString() > lastConsolidateTs)
|
|
161
|
+
return true;
|
|
162
|
+
}
|
|
163
|
+
catch {
|
|
164
|
+
// Ignore files that disappear during the scan.
|
|
165
|
+
}
|
|
251
166
|
}
|
|
252
|
-
}
|
|
167
|
+
}
|
|
168
|
+
return false;
|
|
253
169
|
}
|
|
254
170
|
catch {
|
|
255
171
|
return false;
|
|
@@ -272,6 +188,21 @@ export async function runConsolidationPass(args) {
|
|
|
272
188
|
// so a force-triggered run never trips the pool-size guard. The guard only
|
|
273
189
|
// engages when minPoolSize > 0 and the eligible pool is strictly below it.
|
|
274
190
|
const poolBelowMinSize = !volumeTriggered && minPoolSize > 0 && eligiblePoolSize < minPoolSize;
|
|
191
|
+
return {
|
|
192
|
+
volumeTriggered,
|
|
193
|
+
consolidationOnCooldown,
|
|
194
|
+
consolidateDisabledByProfile,
|
|
195
|
+
poolBelowMinSize,
|
|
196
|
+
eligiblePoolSize,
|
|
197
|
+
minPoolSize,
|
|
198
|
+
...(lastConsolidateTs ? { lastConsolidationTs: lastConsolidateTs } : {}),
|
|
199
|
+
};
|
|
200
|
+
}
|
|
201
|
+
export async function runConsolidationPass(args) {
|
|
202
|
+
const { options, primaryStashDir, memorySummary, improveProfile, resolvedPlan, eventsCtx, budgetSignal, runBudgetMs, } = args;
|
|
203
|
+
const baseConfig = options.config ?? loadConfig();
|
|
204
|
+
const consolidationConfig = baseConfig;
|
|
205
|
+
const { volumeTriggered, consolidationOnCooldown, consolidateDisabledByProfile, poolBelowMinSize, eligiblePoolSize, minPoolSize, lastConsolidationTs, } = evaluateConsolidationEligibility({ options, primaryStashDir, memorySummary, improveProfile, resolvedPlan });
|
|
275
206
|
let consolidation = {
|
|
276
207
|
schemaVersion: 1,
|
|
277
208
|
ok: true,
|
|
@@ -287,16 +218,6 @@ export async function runConsolidationPass(args) {
|
|
|
287
218
|
warnings: [],
|
|
288
219
|
durationMs: 0,
|
|
289
220
|
};
|
|
290
|
-
let gateAutoAcceptedCount = 0;
|
|
291
|
-
let gateAutoAcceptFailedCount = 0;
|
|
292
|
-
const consolidateGateCfg = makeGateConfig("consolidate", {
|
|
293
|
-
globalThreshold: options.autoAccept,
|
|
294
|
-
dryRun: options.dryRun ?? false,
|
|
295
|
-
stashDir: primaryStashDir,
|
|
296
|
-
config: consolidationConfig,
|
|
297
|
-
eventsCtx,
|
|
298
|
-
stateDbPath: eventsCtx?.dbPath,
|
|
299
|
-
}, { minimumThreshold: 95 });
|
|
300
221
|
if (consolidateDisabledByProfile) {
|
|
301
222
|
info("[improve] consolidation skipped (disabled by improve profile)");
|
|
302
223
|
}
|
|
@@ -306,7 +227,7 @@ export async function runConsolidationPass(args) {
|
|
|
306
227
|
// it via the dynamic skipReasons aggregation under `pool_below_min_size`.
|
|
307
228
|
appendEvent({
|
|
308
229
|
eventType: "improve_skipped",
|
|
309
|
-
ref: "
|
|
230
|
+
ref: "memories/_consolidation",
|
|
310
231
|
metadata: {
|
|
311
232
|
reason: "pool_below_min_size",
|
|
312
233
|
poolSize: eligiblePoolSize,
|
|
@@ -316,13 +237,18 @@ export async function runConsolidationPass(args) {
|
|
|
316
237
|
info(`[improve] consolidation skipped (pool ${eligiblePoolSize} < minPoolSize ${minPoolSize})`);
|
|
317
238
|
}
|
|
318
239
|
else if (!consolidationOnCooldown) {
|
|
240
|
+
const consolidationStartedAt = new Date().toISOString();
|
|
319
241
|
consolidation = await withLlmStage("consolidate", () => akmConsolidate({
|
|
320
242
|
...options.consolidateOptions,
|
|
321
243
|
config: consolidationConfig,
|
|
244
|
+
dryRun: options.dryRun ?? false,
|
|
322
245
|
stashDir: options.stashDir,
|
|
323
246
|
// Active profile for this improve run — lets consolidate's secondary
|
|
324
247
|
// process-config reads honor `--profile <name>` instead of `default`.
|
|
325
248
|
improveProfile,
|
|
249
|
+
llmConfig: resolvedPlan.processes.consolidate.runner
|
|
250
|
+
? materializeLlmRunnerConnection(resolvedPlan.processes.consolidate.runner)
|
|
251
|
+
: null,
|
|
326
252
|
autoTriggered: volumeTriggered,
|
|
327
253
|
// Tie consolidate proposals back to this improve invocation so
|
|
328
254
|
// accept-rate-per-run aggregation works. Mirrors reflect/propose/extract.
|
|
@@ -331,25 +257,9 @@ export async function runConsolidationPass(args) {
|
|
|
331
257
|
// recently-changed memories + graph neighbours — use this for frequent
|
|
332
258
|
// passes (quick-shredder). Leave absent in the nightly default profile for
|
|
333
259
|
// a full-pool sweep that catches stale-but-unmerged duplicates.
|
|
334
|
-
incrementalSince: improveProfile?.processes?.consolidate?.incrementalSince,
|
|
335
260
|
limit: improveProfile?.processes?.consolidate?.limit,
|
|
336
261
|
neighborsPerChanged: improveProfile?.processes?.consolidate?.neighborsPerChanged,
|
|
337
262
|
maxChunkSize: improveProfile?.processes?.consolidate?.maxChunkSize,
|
|
338
|
-
// #617 — deterministic near-duplicate dedup pre-pass. DEFAULT OFF; only
|
|
339
|
-
// runs when the profile explicitly sets `consolidate.dedup.enabled`.
|
|
340
|
-
dedup: improveProfile?.processes?.consolidate?.dedup,
|
|
341
|
-
// #581 — judged-state cache. DEFAULT OFF; only engages when the profile
|
|
342
|
-
// explicitly sets `consolidate.judgedCache.enabled`. Skips memories
|
|
343
|
-
// judged-unchanged since their last judge so one run sweeps the full
|
|
344
|
-
// corpus instead of narrowing to a time-window slice.
|
|
345
|
-
judgedCache: improveProfile?.processes?.consolidate?.judgedCache,
|
|
346
|
-
// Honor profile.autoAccept (already merged into options.autoAccept at the
|
|
347
|
-
// top of akmImprove). The CLI parser always supplies 90 when --auto-accept
|
|
348
|
-
// is absent, so ?? 90 is not needed here and would prevent --auto-accept=false
|
|
349
|
-
// (which maps to undefined) from disabling consolidation auto-accept.
|
|
350
|
-
// options.consolidateOptions.autoAccept (if explicitly provided by caller)
|
|
351
|
-
// still wins because the spread above runs first.
|
|
352
|
-
autoAccept: options.consolidateOptions?.autoAccept ?? options.autoAccept,
|
|
353
263
|
// WS-3a: forward budget signal for graceful abort on timeout, and pass
|
|
354
264
|
// the profile's p90 estimate for cold-start budget reduction.
|
|
355
265
|
signal: budgetSignal,
|
|
@@ -357,28 +267,25 @@ export async function runConsolidationPass(args) {
|
|
|
357
267
|
// WS-5: pass total run budget so perfTelemetry.estimatedBudgetFractionUsed
|
|
358
268
|
// can flag when consolidation alone exceeded the budget.
|
|
359
269
|
runBudgetMs,
|
|
360
|
-
}));
|
|
361
|
-
|
|
362
|
-
|
|
363
|
-
|
|
364
|
-
|
|
365
|
-
|
|
366
|
-
|
|
367
|
-
|
|
368
|
-
|
|
369
|
-
|
|
370
|
-
|
|
371
|
-
|
|
372
|
-
}), consolidateGateCfg);
|
|
373
|
-
gateAutoAcceptedCount += consolidateGr.promoted.length;
|
|
374
|
-
gateAutoAcceptFailedCount += consolidateGr.failed.length;
|
|
375
|
-
}
|
|
376
|
-
if (consolidation.processed > 0) {
|
|
270
|
+
}), { engine: resolvedPlan.processes.consolidate.runner?.engine, process: "consolidate" });
|
|
271
|
+
const sourceName = options.sourceName ?? options.writeTarget?.source.name ?? baseConfig.defaultBundle ?? "stash";
|
|
272
|
+
const complete = (consolidation.failedChunks ?? 0) === 0 &&
|
|
273
|
+
(consolidation.failedChunkMemories ?? 0) === 0 &&
|
|
274
|
+
(consolidation.failedPromotions ?? 0) === 0 &&
|
|
275
|
+
(consolidation.deferredMemories ?? 0) === 0;
|
|
276
|
+
const hasUnappliedAdvisoryOperations = consolidation.planned?.some((op) => op.op !== "promote") ?? false;
|
|
277
|
+
if (consolidation.ok &&
|
|
278
|
+
!consolidation.dryRun &&
|
|
279
|
+
complete &&
|
|
280
|
+
!hasUnappliedAdvisoryOperations &&
|
|
281
|
+
consolidation.processed > 0) {
|
|
377
282
|
appendEvent({
|
|
378
283
|
eventType: "consolidate_completed",
|
|
379
|
-
ref: "
|
|
284
|
+
ref: makeBundleRef(sourceName, "memories/_consolidation"),
|
|
380
285
|
metadata: {
|
|
381
286
|
processed: consolidation.processed,
|
|
287
|
+
source: sourceName,
|
|
288
|
+
completedThrough: consolidationStartedAt,
|
|
382
289
|
merged: consolidation.merged,
|
|
383
290
|
deleted: consolidation.deleted,
|
|
384
291
|
contradicted: consolidation.contradicted,
|
|
@@ -391,38 +298,30 @@ export async function runConsolidationPass(args) {
|
|
|
391
298
|
else {
|
|
392
299
|
appendEvent({
|
|
393
300
|
eventType: "improve_skipped",
|
|
394
|
-
ref: "
|
|
301
|
+
ref: "memories/_consolidation",
|
|
395
302
|
metadata: {
|
|
396
303
|
reason: "consolidation_no_memory_updates",
|
|
397
|
-
lastEventTs:
|
|
304
|
+
lastEventTs: lastConsolidationTs ?? null,
|
|
398
305
|
},
|
|
399
306
|
}, eventsCtx);
|
|
400
307
|
info("[improve] consolidation skipped (no memory updates since last run)");
|
|
401
308
|
}
|
|
402
309
|
// D9: track whether consolidation wrote any data so graph extraction can reindex if needed
|
|
403
|
-
const consolidationRan = !consolidateDisabledByProfile &&
|
|
404
|
-
|
|
405
|
-
|
|
406
|
-
|
|
407
|
-
|
|
408
|
-
|
|
409
|
-
maybeAutoTuneThreshold(consolidateGateCfg.phaseThreshold ?? options.autoAccept, consolidationConfig, consolidateTuneDbPath, undefined, "consolidate");
|
|
410
|
-
}
|
|
411
|
-
catch (err) {
|
|
412
|
-
warn(`[improve] calibration auto-tune (consolidate) skipped: ${err instanceof Error ? err.message : String(err)}`);
|
|
413
|
-
}
|
|
414
|
-
}
|
|
415
|
-
return { consolidation, consolidationRan, gateAutoAcceptedCount, gateAutoAcceptFailedCount };
|
|
310
|
+
const consolidationRan = !consolidateDisabledByProfile &&
|
|
311
|
+
!poolBelowMinSize &&
|
|
312
|
+
!consolidationOnCooldown &&
|
|
313
|
+
!consolidation.previewOnly &&
|
|
314
|
+
consolidation.processed > 0;
|
|
315
|
+
return { consolidation, consolidationRan };
|
|
416
316
|
}
|
|
417
317
|
/**
|
|
418
318
|
* Phase 0.4 — session-extract pass. Reads native session files through the
|
|
419
|
-
* SessionLogHarness registry, asks a bounded LLM for candidate proposals
|
|
420
|
-
*
|
|
421
|
-
*
|
|
422
|
-
* consolidation pass and accumulated here.
|
|
319
|
+
* SessionLogHarness registry, and asks a bounded LLM for candidate proposals.
|
|
320
|
+
* Failures are non-fatal (collected into `warnings`). Returns the extract
|
|
321
|
+
* results + any warnings collected along the way.
|
|
423
322
|
*/
|
|
424
323
|
async function runSessionExtractPass(args) {
|
|
425
|
-
const { options, primaryStashDir, improveProfile,
|
|
324
|
+
const { options, primaryStashDir, improveProfile, resolvedPlan, eventsCtx, budgetSignal } = args;
|
|
426
325
|
const warnings = [];
|
|
427
326
|
// Phase 0.4 — session-extract pass.
|
|
428
327
|
//
|
|
@@ -433,9 +332,10 @@ async function runSessionExtractPass(args) {
|
|
|
433
332
|
// / `akm feedback` invocations. Replaces the akm-plugin session-checkpoint
|
|
434
333
|
// hook with an on-demand pull pipeline.
|
|
435
334
|
//
|
|
436
|
-
//
|
|
437
|
-
// (#593: the gate respects the resolved
|
|
438
|
-
// hardcoded `default`
|
|
335
|
+
// Runs only when the ACTIVE strategy resolves
|
|
336
|
+
// `processes.extract.enabled: true` (#593: the gate respects the resolved
|
|
337
|
+
// improve strategy, not just the hardcoded `default` path the legacy feature
|
|
338
|
+
// flag read). Shipped `default` and `frequent` strategies leave this off.
|
|
439
339
|
// Each available harness gets one call with the default --since window;
|
|
440
340
|
// already-seen sessions (tracked in state.db.extract_sessions_seen) are
|
|
441
341
|
// skipped automatically so re-runs don't burn LLM calls on unchanged data.
|
|
@@ -443,30 +343,17 @@ async function runSessionExtractPass(args) {
|
|
|
443
343
|
// Failures are non-fatal — one harness throwing doesn't abort improve.
|
|
444
344
|
// The extract envelope's own `warnings` field surfaces what went wrong.
|
|
445
345
|
let extractResults;
|
|
446
|
-
// Seed the preparation-stage gate counters with consolidation's auto-accept
|
|
447
|
-
// gate results (#551: consolidation now runs in this stage), then accumulate
|
|
448
|
-
// extract's gate results on top.
|
|
449
|
-
let gateAutoAcceptedCount = seedGateAccepted;
|
|
450
|
-
let gateAutoAcceptFailedCount = seedGateFailed;
|
|
451
346
|
const extractConfig = options.config ?? loadConfig();
|
|
452
|
-
const extractGateCfg = makeGateConfig("extract", {
|
|
453
|
-
globalThreshold: options.autoAccept,
|
|
454
|
-
dryRun: options.dryRun ?? false,
|
|
455
|
-
stashDir: primaryStashDir,
|
|
456
|
-
config: extractConfig,
|
|
457
|
-
eventsCtx,
|
|
458
|
-
stateDbPath: eventsCtx?.dbPath,
|
|
459
|
-
});
|
|
460
347
|
// #554 minNewSessions gate: skip the entire extract pass (ensureIndex was
|
|
461
348
|
// already done upstream; here we elide every akmExtract/processSession call)
|
|
462
349
|
// when the NEW (unseen, in-window) candidate-session pool is below a minimum.
|
|
463
350
|
// 22% of improve runs produce zero memory-inference writes because extract
|
|
464
351
|
// finds no new sessions, yet still burns the full extract pipeline. Default 0
|
|
465
352
|
// (disabled) preserves existing always-run behaviour; only opted-in profiles
|
|
466
|
-
// (e.g. `frequent`) set it. Evaluated BEFORE any LLM
|
|
467
|
-
// LLM work AND writes nothing
|
|
468
|
-
//
|
|
469
|
-
//
|
|
353
|
+
// (e.g. a user-enabled `frequent` strategy) set it. Evaluated BEFORE any LLM
|
|
354
|
+
// call so a skip costs zero LLM work AND writes nothing. A skipped extract
|
|
355
|
+
// never flags work for the NEXT run's consolidation mtime-gate (the
|
|
356
|
+
// downstream trigger #554 asks us to suppress).
|
|
470
357
|
const EXTRACT_DEFAULT_MIN_NEW_SESSIONS = 0;
|
|
471
358
|
// Read from the ACTIVE resolved profile (not always `default`), matching how
|
|
472
359
|
// `extract.enabled` resolves — otherwise a non-default profile (e.g.
|
|
@@ -476,11 +363,22 @@ async function runSessionExtractPass(args) {
|
|
|
476
363
|
// #593/#594: the ACTIVE resolved improve profile is the single source of
|
|
477
364
|
// truth for whether extract runs. (Previously this also ANDed in the legacy
|
|
478
365
|
// `session_extraction` feature flag, which only reads
|
|
479
|
-
//
|
|
480
|
-
// profile a global kill switch, so a non-default profile enabling extract was
|
|
481
|
-
// silently overridden. The default profile is now just another profile.)
|
|
366
|
+
// a retired global feature path; the selected strategy is authoritative.)
|
|
482
367
|
// `akmExtract` re-checks the same active profile internally via `improveProfile`.
|
|
483
|
-
if (
|
|
368
|
+
if (resolvedPlan.processes.extract.enabled) {
|
|
369
|
+
const extractRunner = resolvedPlan.processes.extract.runner;
|
|
370
|
+
if (!extractRunner?.engine) {
|
|
371
|
+
throw new ConfigError("Resolved improve plan has no runner for enabled extract process.", "LLM_NOT_CONFIGURED");
|
|
372
|
+
}
|
|
373
|
+
const extractPlan = Object.freeze({
|
|
374
|
+
strategy: resolvedPlan.strategy.name,
|
|
375
|
+
engine: extractRunner.engine,
|
|
376
|
+
enabled: true,
|
|
377
|
+
process: resolvedPlan.processes.extract.config,
|
|
378
|
+
runner: extractRunner,
|
|
379
|
+
timeoutMs: extractRunner.timeoutMs === undefined ? 600_000 : extractRunner.timeoutMs,
|
|
380
|
+
embeddingConfig: Object.freeze(structuredClone(extractConfig.embedding)),
|
|
381
|
+
});
|
|
484
382
|
const availableHarnesses = options.extractHarnesses ?? getAvailableHarnesses();
|
|
485
383
|
// The guard engages only when minNewSessions > 0; 0 disables it entirely.
|
|
486
384
|
let belowMinNewSessions = false;
|
|
@@ -503,7 +401,7 @@ async function runSessionExtractPass(args) {
|
|
|
503
401
|
// skipReasons aggregation surfaces this under `below_min_new_sessions`.
|
|
504
402
|
appendEvent({
|
|
505
403
|
eventType: "improve_skipped",
|
|
506
|
-
ref: "
|
|
404
|
+
ref: "memories/_extract",
|
|
507
405
|
metadata: {
|
|
508
406
|
reason: "below_min_new_sessions",
|
|
509
407
|
newSessions: newCandidateCount,
|
|
@@ -521,25 +419,17 @@ async function runSessionExtractPass(args) {
|
|
|
521
419
|
type: h.name,
|
|
522
420
|
...(primaryStashDir !== undefined ? { stashDir: primaryStashDir } : {}),
|
|
523
421
|
config: extractConfig,
|
|
524
|
-
|
|
525
|
-
// config read the running profile, not always `default`.
|
|
526
|
-
improveProfile,
|
|
422
|
+
resolvedPlan: extractPlan,
|
|
527
423
|
dryRun: options.dryRun ?? false,
|
|
424
|
+
signal: budgetSignal,
|
|
528
425
|
...(options.extractHarnesses ? { harnesses: options.extractHarnesses } : {}),
|
|
529
426
|
// C2: pin extract's skip-tracking state.db open to the boundary path.
|
|
530
427
|
...(eventsCtx?.dbPath ? { stateDbPath: eventsCtx.dbPath } : {}),
|
|
531
|
-
|
|
428
|
+
// R25: extract's event emits reuse the run's events context
|
|
429
|
+
// (fast path when it carries the long-lived handle).
|
|
430
|
+
eventsCtx,
|
|
431
|
+
}), { engine: resolvedPlan.processes.extract.runner?.engine, process: "extract" });
|
|
532
432
|
extractResults.push(result);
|
|
533
|
-
{
|
|
534
|
-
const gr = await runAutoAcceptGate(primaryStashDir
|
|
535
|
-
? result.proposals.map((proposalId) => {
|
|
536
|
-
const proposal = getProposal(primaryStashDir, proposalId);
|
|
537
|
-
return { proposalId, confidence: resolveExtractConfidence(proposal) };
|
|
538
|
-
})
|
|
539
|
-
: [], extractGateCfg);
|
|
540
|
-
gateAutoAcceptedCount += gr.promoted.length;
|
|
541
|
-
gateAutoAcceptFailedCount += gr.failed.length;
|
|
542
|
-
}
|
|
543
433
|
}
|
|
544
434
|
catch (err) {
|
|
545
435
|
const msg = err instanceof Error ? err.message : String(err);
|
|
@@ -553,73 +443,57 @@ async function runSessionExtractPass(args) {
|
|
|
553
443
|
}
|
|
554
444
|
}
|
|
555
445
|
}
|
|
556
|
-
|
|
557
|
-
|
|
558
|
-
|
|
559
|
-
|
|
560
|
-
if (primaryStashDir && !options.dryRun && options.autoAccept !== undefined) {
|
|
561
|
-
const freshIds = new Set((extractResults ?? []).flatMap((r) => r.proposals));
|
|
562
|
-
const backlog = listProposals(primaryStashDir, { status: "pending" }).filter((p) => p.source === "extract" && !freshIds.has(p.id));
|
|
563
|
-
if (backlog.length > 0) {
|
|
564
|
-
const backlogCandidates = backlog.map((p) => ({
|
|
565
|
-
proposalId: p.id,
|
|
566
|
-
confidence: resolveExtractConfidence(p),
|
|
567
|
-
}));
|
|
568
|
-
const backlogGr = await runAutoAcceptGate(backlogCandidates, extractGateCfg);
|
|
569
|
-
gateAutoAcceptedCount += backlogGr.promoted.length;
|
|
570
|
-
gateAutoAcceptFailedCount += backlogGr.failed.length;
|
|
571
|
-
}
|
|
572
|
-
}
|
|
573
|
-
return { extractResults, gateAutoAcceptedCount, gateAutoAcceptFailedCount, warnings, extractGateCfg };
|
|
446
|
+
return {
|
|
447
|
+
extractResults,
|
|
448
|
+
warnings,
|
|
449
|
+
};
|
|
574
450
|
}
|
|
575
451
|
/**
|
|
576
452
|
* Phase 1 — validation + schema-repair pass. Scans postCleanupRefs for assets
|
|
577
453
|
* with structural problems (missing file, missing lesson description), attempts
|
|
578
454
|
* LLM schema repair, and returns the still-failing ref set + the repair records.
|
|
579
455
|
*/
|
|
580
|
-
async function runValidationAndRepairPass(args) {
|
|
581
|
-
const { postCleanupRefs, options, startMs, budgetMs, primaryStashDir } = args;
|
|
582
|
-
const
|
|
583
|
-
for (const candidate of postCleanupRefs) {
|
|
456
|
+
export async function runValidationAndRepairPass(args) {
|
|
457
|
+
const { postCleanupRefs, options, startMs, budgetMs, primaryStashDir, resolvedPlan, repairValidationFailures, schemaRepairFn = runSchemaRepairPass, } = args;
|
|
458
|
+
const validateCandidate = async (candidate) => {
|
|
584
459
|
try {
|
|
585
|
-
// #591: use the path pre-resolved at planning time when it is still on
|
|
586
|
-
// disk — a serial async DB lookup per ref cost ~500 s on a 9 000-ref
|
|
587
|
-
// stash. Fall back to findAssetFilePath only for refs that bypassed
|
|
588
|
-
// collectEligibleRefs' index scan or whose file moved since planning.
|
|
589
460
|
const filePath = candidate.filePath && fs.existsSync(candidate.filePath)
|
|
590
461
|
? candidate.filePath
|
|
591
462
|
: await findAssetFilePath(candidate.ref, options.stashDir);
|
|
592
|
-
if (!filePath)
|
|
593
|
-
|
|
594
|
-
|
|
595
|
-
|
|
596
|
-
if (path.extname(filePath).toLowerCase() !== ".md") {
|
|
597
|
-
continue;
|
|
598
|
-
}
|
|
463
|
+
if (!filePath)
|
|
464
|
+
return "file not found on disk";
|
|
465
|
+
if (path.extname(filePath).toLowerCase() !== ".md")
|
|
466
|
+
return undefined;
|
|
599
467
|
if (isLessonCandidate(candidate.ref)) {
|
|
600
|
-
const
|
|
601
|
-
const fm = parseFrontmatter(raw).data;
|
|
468
|
+
const fm = parseFrontmatter(fs.readFileSync(filePath, "utf8")).data;
|
|
602
469
|
if (!fm.description)
|
|
603
|
-
|
|
470
|
+
return "missing description";
|
|
604
471
|
}
|
|
472
|
+
return undefined;
|
|
605
473
|
}
|
|
606
|
-
catch (
|
|
607
|
-
|
|
474
|
+
catch (error) {
|
|
475
|
+
return String(error);
|
|
608
476
|
}
|
|
477
|
+
};
|
|
478
|
+
const validationFailures = [];
|
|
479
|
+
for (const candidate of postCleanupRefs) {
|
|
480
|
+
const reason = await validateCandidate(candidate);
|
|
481
|
+
if (reason)
|
|
482
|
+
validationFailures.push({ ref: candidate.ref, reason });
|
|
609
483
|
}
|
|
610
484
|
if (validationFailures.length > 0) {
|
|
611
|
-
info(`[improve] ${validationFailures.length} assets have validation issues (will attempt schema repair):`);
|
|
485
|
+
info(`[improve] ${validationFailures.length} assets have validation issues${repairValidationFailures ? " (will attempt schema repair)" : ""}:`);
|
|
612
486
|
for (const f of validationFailures)
|
|
613
487
|
info(` ${f.ref}: ${f.reason}`);
|
|
614
488
|
}
|
|
615
489
|
let schemaRepairs = [];
|
|
616
|
-
|
|
490
|
+
const repairedRefs = new Set();
|
|
617
491
|
// Schema repair pass: attempt to fix validation failures via LLM before skipping.
|
|
618
|
-
if (validationFailures.length > 0
|
|
619
|
-
const
|
|
620
|
-
const llmCfg =
|
|
492
|
+
if (validationFailures.length > 0) {
|
|
493
|
+
const validationRunner = resolvedPlan.processes.validation.runner;
|
|
494
|
+
const llmCfg = validationRunner ? materializeLlmRunnerConnection(validationRunner) : undefined;
|
|
621
495
|
if (llmCfg) {
|
|
622
|
-
const result = await
|
|
496
|
+
const result = await withLlmStage("validation", () => schemaRepairFn(validationFailures, {
|
|
623
497
|
startMs,
|
|
624
498
|
budgetMs,
|
|
625
499
|
llmConfig: llmCfg,
|
|
@@ -633,9 +507,17 @@ async function runValidationAndRepairPass(args) {
|
|
|
633
507
|
stashDir: primaryStashDir,
|
|
634
508
|
findFilePath: findAssetFilePath,
|
|
635
509
|
isLessonCandidateFn: isLessonCandidate,
|
|
636
|
-
});
|
|
510
|
+
}), { engine: resolvedPlan.processes.validation.runner?.engine, process: "validation" });
|
|
637
511
|
schemaRepairs = result.repairs;
|
|
638
|
-
|
|
512
|
+
// A repair result is advisory. Only a fresh structural read of the live
|
|
513
|
+
// asset can remove it from the failure set; queued content is not live.
|
|
514
|
+
const failedRefs = new Set(validationFailures.map((failure) => failure.ref));
|
|
515
|
+
const candidatesByRef = new Map(postCleanupRefs.map((candidate) => [candidate.ref, candidate]));
|
|
516
|
+
for (const ref of failedRefs) {
|
|
517
|
+
const candidate = candidatesByRef.get(ref);
|
|
518
|
+
if (candidate && !(await validateCandidate(candidate)))
|
|
519
|
+
repairedRefs.add(ref);
|
|
520
|
+
}
|
|
639
521
|
}
|
|
640
522
|
}
|
|
641
523
|
const validationFailureRefs = new Set(validationFailures.filter((f) => !repairedRefs.has(f.ref)).map((f) => f.ref));
|
|
@@ -645,32 +527,18 @@ async function runValidationAndRepairPass(args) {
|
|
|
645
527
|
return { validationFailures, validationFailureRefs, schemaRepairs };
|
|
646
528
|
}
|
|
647
529
|
export async function runImprovePreparationStage(args) {
|
|
648
|
-
const { scope, options, plannedRefs, memoryCleanupPlan, primaryStashDir, memorySummary, reindexFn, startMs, budgetMs, eventsCtx, initialCleanupWarnings, improveProfile, budgetSignal, } = args;
|
|
530
|
+
const { scope, options, plannedRefs, memoryCleanupPlan, primaryStashDir, memorySummary, reindexFn, startMs, budgetMs, eventsCtx, initialCleanupWarnings, improveProfile, resolvedPlan, strategyName, budgetSignal, } = args;
|
|
649
531
|
const actions = [];
|
|
650
532
|
const cleanupWarnings = initialCleanupWarnings ? [...initialCleanupWarnings] : [];
|
|
651
|
-
|
|
652
|
-
|
|
653
|
-
if (
|
|
654
|
-
|
|
655
|
-
if (fs.existsSync(memoryMdPath)) {
|
|
656
|
-
try {
|
|
657
|
-
const lines = fs.readFileSync(memoryMdPath, "utf8").split("\n").length;
|
|
658
|
-
const overBudget = lines >= 180;
|
|
659
|
-
memoryIndexHealth = { lineCount: lines, overBudget };
|
|
660
|
-
if (overBudget) {
|
|
661
|
-
cleanupWarnings.push(`MEMORY.md has ${lines} lines (budget: 200). Consolidation strongly recommended.`);
|
|
662
|
-
}
|
|
663
|
-
}
|
|
664
|
-
catch {
|
|
665
|
-
// best-effort
|
|
666
|
-
}
|
|
667
|
-
}
|
|
668
|
-
}
|
|
533
|
+
const memoryBudget = assessMemoryIndexBudget(primaryStashDir);
|
|
534
|
+
const memoryIndexHealth = memoryBudget.memoryIndexHealth;
|
|
535
|
+
if (memoryBudget.warning)
|
|
536
|
+
cleanupWarnings.push(memoryBudget.warning);
|
|
669
537
|
// Phase 0.3 — memory consolidation pass (#551).
|
|
670
538
|
//
|
|
671
539
|
// Consolidation runs BEFORE the session-extract pass. This is the structural
|
|
672
|
-
// half of the #551 fix: extract
|
|
673
|
-
//
|
|
540
|
+
// half of the #551 fix: extract promotions write brand-new memory .md files,
|
|
541
|
+
// which previously made the consolidation pool-delta gate fire
|
|
674
542
|
// unconditionally (any new file => "memory updated since last consolidate").
|
|
675
543
|
// By running consolidation first, the gate and akmConsolidate only ever see
|
|
676
544
|
// memories that existed at the start of the run — current-run extract
|
|
@@ -681,6 +549,7 @@ export async function runImprovePreparationStage(args) {
|
|
|
681
549
|
primaryStashDir,
|
|
682
550
|
memorySummary,
|
|
683
551
|
improveProfile,
|
|
552
|
+
resolvedPlan,
|
|
684
553
|
eventsCtx,
|
|
685
554
|
budgetSignal,
|
|
686
555
|
runBudgetMs: budgetMs,
|
|
@@ -690,13 +559,11 @@ export async function runImprovePreparationStage(args) {
|
|
|
690
559
|
options,
|
|
691
560
|
primaryStashDir,
|
|
692
561
|
improveProfile,
|
|
562
|
+
resolvedPlan,
|
|
693
563
|
eventsCtx,
|
|
694
|
-
|
|
695
|
-
seedGateFailed: consolidationPass.gateAutoAcceptFailedCount,
|
|
564
|
+
budgetSignal,
|
|
696
565
|
});
|
|
697
566
|
const extractResults = extractPass.extractResults;
|
|
698
|
-
const gateAutoAcceptedCount = extractPass.gateAutoAcceptedCount;
|
|
699
|
-
const gateAutoAcceptFailedCount = extractPass.gateAutoAcceptFailedCount;
|
|
700
567
|
if (extractPass.warnings.length > 0)
|
|
701
568
|
cleanupWarnings.push(...extractPass.warnings);
|
|
702
569
|
// eligibleCount = raw pre-filter count (before cooldown/signal/cleanup filters).
|
|
@@ -704,19 +571,167 @@ export async function runImprovePreparationStage(args) {
|
|
|
704
571
|
appendEvent({
|
|
705
572
|
eventType: "improve_invoked",
|
|
706
573
|
ref: scope.mode === "ref" ? scope.value : `improve:${scope.mode}:${scope.value ?? "all"}`,
|
|
707
|
-
metadata: { scope, dryRun: options.dryRun ?? false, eligibleCount: plannedRefs.length },
|
|
574
|
+
metadata: { strategy: strategyName, scope, dryRun: options.dryRun ?? false, eligibleCount: plannedRefs.length },
|
|
708
575
|
}, eventsCtx);
|
|
709
576
|
// ensureIndex now runs in akmImprove() BEFORE collectEligibleRefs so the
|
|
710
577
|
// eligible-ref query sees a populated `entries` table on the very first
|
|
711
578
|
// pass after a DB version upgrade (#339). Any failure messages from that
|
|
712
579
|
// earlier call were threaded in via args.initialCleanupWarnings.
|
|
580
|
+
const cleanup = await applyCleanupPass({
|
|
581
|
+
primaryStashDir,
|
|
582
|
+
memoryCleanupPlan,
|
|
583
|
+
plannedRefs,
|
|
584
|
+
reindexFn,
|
|
585
|
+
budgetSignal,
|
|
586
|
+
allowApply: isAutonomyLaneAllowed("memoryCleanup", options.config ?? loadConfig()),
|
|
587
|
+
});
|
|
588
|
+
const appliedCleanup = cleanup.appliedCleanup;
|
|
589
|
+
const postCleanupRefs = cleanup.postCleanupRefs;
|
|
590
|
+
actions.push(...cleanup.pruneActions);
|
|
591
|
+
cleanupWarnings.push(...cleanup.warnings);
|
|
592
|
+
const { validationFailures, validationFailureRefs, schemaRepairs } = await runValidationAndRepairPass({
|
|
593
|
+
postCleanupRefs,
|
|
594
|
+
options,
|
|
595
|
+
startMs,
|
|
596
|
+
budgetMs,
|
|
597
|
+
primaryStashDir,
|
|
598
|
+
resolvedPlan,
|
|
599
|
+
repairValidationFailures: resolvedPlan.processes.validation.enabled && options.repairValidationFailures !== false,
|
|
600
|
+
});
|
|
601
|
+
// Phase 0.5 — structural hygiene pass
|
|
602
|
+
let lintSummary;
|
|
603
|
+
if (primaryStashDir) {
|
|
604
|
+
try {
|
|
605
|
+
const lintResult = akmLint({ fix: false, dir: primaryStashDir });
|
|
606
|
+
lintSummary = { fixed: lintResult.summary.fixed, flagged: lintResult.summary.flagged };
|
|
607
|
+
}
|
|
608
|
+
catch {
|
|
609
|
+
// lint is best-effort; never block improve
|
|
610
|
+
}
|
|
611
|
+
}
|
|
612
|
+
const recentErrors = seedRecentErrorWindows(schemaRepairs);
|
|
613
|
+
const snapshot = buildSnapshotManifest({ postCleanupRefs, validationFailureRefs });
|
|
614
|
+
const gathered = gatherCandidates({
|
|
615
|
+
scope,
|
|
616
|
+
options,
|
|
617
|
+
primaryStashDir,
|
|
618
|
+
eventsCtx,
|
|
619
|
+
improveProfile,
|
|
620
|
+
resolvedPlan,
|
|
621
|
+
postCleanupRefs,
|
|
622
|
+
validationFailureRefs,
|
|
623
|
+
snapshot,
|
|
624
|
+
});
|
|
625
|
+
actions.push(...gathered.actions);
|
|
626
|
+
const scored = scoreSalience({
|
|
627
|
+
scope,
|
|
628
|
+
options,
|
|
629
|
+
primaryStashDir,
|
|
630
|
+
eventsCtx,
|
|
631
|
+
mergedRefs: gathered.mergedRefs,
|
|
632
|
+
eligibilitySourceByRef: gathered.eligibilitySourceByRef,
|
|
633
|
+
feedbackSummary: gathered.feedbackSummary,
|
|
634
|
+
retrievalCounts: gathered.retrievalCounts,
|
|
635
|
+
signalFiltered: gathered.signalFiltered,
|
|
636
|
+
proactiveRefs: gathered.proactiveRefs,
|
|
637
|
+
highSalienceRefs: gathered.highSalienceRefs,
|
|
638
|
+
});
|
|
639
|
+
const filtered = await filterEligibility({
|
|
640
|
+
scope,
|
|
641
|
+
options,
|
|
642
|
+
plannedRefs,
|
|
643
|
+
eventsCtx,
|
|
644
|
+
mergedRefs: scored.mergedRefs,
|
|
645
|
+
salienceMap: scored.salienceMap,
|
|
646
|
+
eligibilitySourceByRef: gathered.eligibilitySourceByRef,
|
|
647
|
+
distillOnlyRefs: gathered.distillOnlyRefs,
|
|
648
|
+
validationFailureRefs,
|
|
649
|
+
summary: {
|
|
650
|
+
fullySkippedCount: gathered.fullySkippedCount,
|
|
651
|
+
preCooldownCount: gathered.preCooldownCount,
|
|
652
|
+
signalAndRetrievalRefs: gathered.signalAndRetrievalRefs,
|
|
653
|
+
signalFiltered: gathered.signalFiltered,
|
|
654
|
+
},
|
|
655
|
+
});
|
|
656
|
+
return {
|
|
657
|
+
actions,
|
|
658
|
+
cleanupWarnings,
|
|
659
|
+
appliedCleanup,
|
|
660
|
+
memoryIndexHealth,
|
|
661
|
+
extract: extractResults,
|
|
662
|
+
actionableRefs: filtered.actionableRefs,
|
|
663
|
+
signalBearingSet: gathered.signalBearingSet,
|
|
664
|
+
validationFailures,
|
|
665
|
+
schemaRepairs,
|
|
666
|
+
lintSummary,
|
|
667
|
+
loopRefs: filtered.loopRefs,
|
|
668
|
+
distillCooledRefs: gathered.distillCooledRefs,
|
|
669
|
+
distillOnlyRefs: filtered.distillOnlyRefs,
|
|
670
|
+
coverageGaps: filtered.coverageGaps,
|
|
671
|
+
recentErrors,
|
|
672
|
+
utilityMap: scored.utilityMap,
|
|
673
|
+
consolidation: consolidationPass.consolidation,
|
|
674
|
+
consolidationRan: consolidationPass.consolidationRan,
|
|
675
|
+
...(gathered.proactiveMaintenanceSummary ? { proactiveMaintenance: gathered.proactiveMaintenanceSummary } : {}),
|
|
676
|
+
};
|
|
677
|
+
}
|
|
678
|
+
// ── preparation-stage passes (WI-7.6 decomposition, R31) ────────────────────
|
|
679
|
+
// The six-pass split prescribed by the chunk-7 brief §WI-7.6, adapted to the
|
|
680
|
+
// code as it exists at HEAD (anchors re-measured; see the chunk-7 ledger):
|
|
681
|
+
// snapshot-manifest → buildSnapshotManifest
|
|
682
|
+
// candidate-gather → gatherCandidates (+ its five lane/sub-passes)
|
|
683
|
+
// salience-score → scoreSalience (+ outcome/vector/persist sub-passes)
|
|
684
|
+
// valence-score → the two computeValenceScore call sites move VERBATIM
|
|
685
|
+
// inside the salience passes (pure fn; no separate pass)
|
|
686
|
+
// standards-context → does not exist in preparation.ts (assembly lives in
|
|
687
|
+
// extract.ts — recorded in the ledger, no empty pass)
|
|
688
|
+
// eligibility-filter → filterEligibility (+ replay/disk-check sub-passes)
|
|
689
|
+
// Every pass takes an args object and returns its results; the orchestrator
|
|
690
|
+
// folds them. Shared-by-reference structures (the ImproveEligibleRef objects,
|
|
691
|
+
// eligibilitySourceByRef, salienceMap, actions, recentErrors) keep their
|
|
692
|
+
// identity — attribution stamps must travel with the ref objects into the
|
|
693
|
+
// loop stage exactly as before.
|
|
694
|
+
/** Phase 0 — MEMORY.md budget check (200-line cap; warn at 180). */
|
|
695
|
+
function assessMemoryIndexBudget(primaryStashDir) {
|
|
696
|
+
let warning;
|
|
697
|
+
// Phase 0 — MEMORY.md budget check (200-line cap; warn at 180)
|
|
698
|
+
let memoryIndexHealth;
|
|
699
|
+
if (primaryStashDir) {
|
|
700
|
+
const memoryMdPath = path.join(primaryStashDir, "memories", "MEMORY.md");
|
|
701
|
+
if (fs.existsSync(memoryMdPath)) {
|
|
702
|
+
try {
|
|
703
|
+
const lines = fs.readFileSync(memoryMdPath, "utf8").split("\n").length;
|
|
704
|
+
const overBudget = lines >= 180;
|
|
705
|
+
memoryIndexHealth = { lineCount: lines, overBudget };
|
|
706
|
+
if (overBudget) {
|
|
707
|
+
warning = `MEMORY.md has ${lines} lines (budget: 200). Consolidation strongly recommended.`;
|
|
708
|
+
}
|
|
709
|
+
}
|
|
710
|
+
catch {
|
|
711
|
+
// best-effort
|
|
712
|
+
}
|
|
713
|
+
}
|
|
714
|
+
}
|
|
715
|
+
return { memoryIndexHealth, warning };
|
|
716
|
+
}
|
|
717
|
+
/**
|
|
718
|
+
* Memory-cleanup apply + prune-action recording + the post-cleanup reindex.
|
|
719
|
+
* Returns the surviving ref set and the prune actions/warnings for the
|
|
720
|
+
* orchestrator to fold (same order as the old inline pushes).
|
|
721
|
+
*/
|
|
722
|
+
async function applyCleanupPass(args) {
|
|
723
|
+
const { primaryStashDir, memoryCleanupPlan, plannedRefs, reindexFn, budgetSignal, allowApply } = args;
|
|
724
|
+
const pruneActions = [];
|
|
725
|
+
const warnings = [];
|
|
713
726
|
let appliedCleanup;
|
|
714
727
|
try {
|
|
715
728
|
appliedCleanup =
|
|
716
|
-
primaryStashDir && memoryCleanupPlan
|
|
729
|
+
primaryStashDir && memoryCleanupPlan && allowApply
|
|
730
|
+
? applyMemoryCleanup(primaryStashDir, memoryCleanupPlan)
|
|
731
|
+
: undefined;
|
|
717
732
|
}
|
|
718
733
|
catch (err) {
|
|
719
|
-
|
|
734
|
+
warnings.push(`applyMemoryCleanup failed: ${err instanceof Error ? err.message : String(err)}`);
|
|
720
735
|
}
|
|
721
736
|
const archivedRefs = appliedCleanup?.archived.map((record) => record.ref) ?? [];
|
|
722
737
|
const removed = new Set(archivedRefs);
|
|
@@ -730,7 +745,7 @@ export async function runImprovePreparationStage(args) {
|
|
|
730
745
|
const archived = appliedCleanup.archived.find((record) => record.ref === candidate.ref);
|
|
731
746
|
if (!archived)
|
|
732
747
|
continue;
|
|
733
|
-
|
|
748
|
+
pruneActions.push({
|
|
734
749
|
ref: candidate.ref,
|
|
735
750
|
mode: "memory-prune",
|
|
736
751
|
result: { ok: true, pruned: true, reason: candidate.reason },
|
|
@@ -738,31 +753,17 @@ export async function runImprovePreparationStage(args) {
|
|
|
738
753
|
}
|
|
739
754
|
if ((appliedCleanup.archived.length > 0 || appliedCleanup.beliefStateTransitions.length > 0) && primaryStashDir) {
|
|
740
755
|
try {
|
|
741
|
-
await reindexFn({ stashDir: primaryStashDir });
|
|
756
|
+
await reindexFn({ stashDir: primaryStashDir, signal: budgetSignal });
|
|
742
757
|
}
|
|
743
758
|
catch (err) {
|
|
744
|
-
|
|
759
|
+
warnings.push(`reindex after cleanup failed: ${err instanceof Error ? err.message : String(err)}`);
|
|
745
760
|
}
|
|
746
761
|
}
|
|
747
762
|
}
|
|
748
|
-
|
|
749
|
-
|
|
750
|
-
|
|
751
|
-
|
|
752
|
-
budgetMs,
|
|
753
|
-
primaryStashDir,
|
|
754
|
-
});
|
|
755
|
-
// Phase 0.5 — structural hygiene pass
|
|
756
|
-
let lintSummary;
|
|
757
|
-
if (primaryStashDir) {
|
|
758
|
-
try {
|
|
759
|
-
const lintResult = akmLint({ fix: true, dir: primaryStashDir });
|
|
760
|
-
lintSummary = { fixed: lintResult.summary.fixed, flagged: lintResult.summary.flagged };
|
|
761
|
-
}
|
|
762
|
-
catch {
|
|
763
|
-
// lint is best-effort; never block improve
|
|
764
|
-
}
|
|
765
|
-
}
|
|
763
|
+
return { appliedCleanup, postCleanupRefs, pruneActions, warnings };
|
|
764
|
+
}
|
|
765
|
+
/** Seed the per-originator rolling error windows from schema-repair errors. */
|
|
766
|
+
function seedRecentErrorWindows(schemaRepairs) {
|
|
766
767
|
// O-5 / #378: Per-originator rolling error windows.
|
|
767
768
|
// Reflexion (arXiv:2303.11366) warns that cross-task verbal critique
|
|
768
769
|
// contamination degrades below single-shot baseline. Each originator key
|
|
@@ -785,6 +786,11 @@ export async function runImprovePreparationStage(args) {
|
|
|
785
786
|
pushRecentError("schema-repair", errMsg);
|
|
786
787
|
}
|
|
787
788
|
}
|
|
789
|
+
return recentErrors;
|
|
790
|
+
}
|
|
791
|
+
/** Pass: snapshot-manifest — the three timestamp maps + the 30-day signal window. */
|
|
792
|
+
export function buildSnapshotManifest(args) {
|
|
793
|
+
const { postCleanupRefs, validationFailureRefs } = args;
|
|
788
794
|
// ── Phase 2: signal-delta eligibility sets built EARLY ────────────────────
|
|
789
795
|
// 0.8.0 replaces the flat time-based cooldowns (which produced synchronised
|
|
790
796
|
// waves whenever many refs cooled at the same instant — see the 2026-05-26
|
|
@@ -801,69 +807,235 @@ export async function runImprovePreparationStage(args) {
|
|
|
801
807
|
// The 30-day FEEDBACK_SIGNAL_WINDOW_DAYS bound still applies — only feedback
|
|
802
808
|
// events newer than that count as "current signal". Ancient one-off
|
|
803
809
|
// negatives don't permanently lock a ref into every run.
|
|
804
|
-
//
|
|
805
|
-
// High-retrieval refs (P0-A path) use a simpler "eligible once" rule: a
|
|
806
|
-
// ref with no feedback signal but retrievalCount ≥ threshold is eligible
|
|
807
|
-
// exactly once (no prior reflect proposal). Subsequent re-eligibility for
|
|
808
|
-
// those refs requires either a new feedback event (then the normal
|
|
809
|
-
// signal-delta gate applies) or human action. Documented limitation: this
|
|
810
|
-
// path does not re-fire on retrieval-count growth alone in 0.8.0; storing
|
|
811
|
-
// the retrieval count in proposal metadata for proper delta-tracking is
|
|
812
|
-
// captured as future work.
|
|
813
810
|
const FEEDBACK_SIGNAL_WINDOW_DAYS = 30;
|
|
814
811
|
const feedbackSinceCutoff = new Date(Date.now() - daysToMs(FEEDBACK_SIGNAL_WINDOW_DAYS)).toISOString();
|
|
815
812
|
// Build the three timestamp maps once across the entire postCleanupRefs set.
|
|
816
813
|
// Per-ref queries would be N+1 and the planner is already the hottest path
|
|
817
814
|
// in `akm improve`.
|
|
818
815
|
const candidateRefs = postCleanupRefs.filter((r) => !validationFailureRefs.has(r.ref)).map((r) => r.ref);
|
|
819
|
-
|
|
820
|
-
const
|
|
821
|
-
const
|
|
822
|
-
|
|
823
|
-
|
|
824
|
-
|
|
825
|
-
|
|
826
|
-
|
|
827
|
-
|
|
828
|
-
|
|
829
|
-
|
|
830
|
-
|
|
831
|
-
|
|
832
|
-
|
|
833
|
-
|
|
834
|
-
|
|
835
|
-
|
|
836
|
-
|
|
837
|
-
|
|
838
|
-
|
|
839
|
-
|
|
840
|
-
|
|
841
|
-
|
|
842
|
-
|
|
843
|
-
const eligibleRefs =
|
|
844
|
-
|
|
845
|
-
//
|
|
846
|
-
|
|
847
|
-
|
|
848
|
-
//
|
|
849
|
-
|
|
850
|
-
|
|
851
|
-
|
|
816
|
+
// Carry each candidate's item_ref into the feedback/proposal timestamp reads.
|
|
817
|
+
const itemRefByRef = buildItemRefByRef(postCleanupRefs);
|
|
818
|
+
const latestFeedbackTs = buildLatestFeedbackTsMap(candidateRefs, feedbackSinceCutoff, itemRefByRef);
|
|
819
|
+
const lastReflectProposalTs = buildLatestProposalTsMap(candidateRefs, "reflect", itemRefByRef);
|
|
820
|
+
const lastDistillProposalTs = buildLatestProposalTsMap(candidateRefs, "distill", itemRefByRef);
|
|
821
|
+
return { feedbackSinceCutoff, latestFeedbackTs, lastReflectProposalTs, lastDistillProposalTs };
|
|
822
|
+
}
|
|
823
|
+
/**
|
|
824
|
+
* Pass: candidate-gather — the signal-delta partition, the bulk feedback
|
|
825
|
+
* summary, retrieval signals, the Layer-2 proactive and Layer-3 high-salience
|
|
826
|
+
* rescue lanes, the merged candidate set, and lane attribution stamping.
|
|
827
|
+
*/
|
|
828
|
+
function gatherCandidates(args) {
|
|
829
|
+
const { scope, options, primaryStashDir, eventsCtx, improveProfile, resolvedPlan, postCleanupRefs } = args;
|
|
830
|
+
const { feedbackSinceCutoff, lastReflectProposalTs, lastDistillProposalTs } = args.snapshot;
|
|
831
|
+
const partition = partitionBySignalDelta({
|
|
832
|
+
scope,
|
|
833
|
+
options,
|
|
834
|
+
eventsCtx,
|
|
835
|
+
postCleanupRefs,
|
|
836
|
+
validationFailureRefs: args.validationFailureRefs,
|
|
837
|
+
snapshot: args.snapshot,
|
|
838
|
+
});
|
|
839
|
+
const actions = [...partition.actions];
|
|
840
|
+
const { distillCooledRefs, preCooldownCount, eligibleRefs, distillOnlyRefs, noFeedbackPool, fullySkippedCount } = partition;
|
|
841
|
+
// ── Phase 4: signal/feedback/utility/sort on the reduced set ──────────────
|
|
842
|
+
// Everything from here works on (eligibleRefs ∪ distillOnlyRefs) plus the
|
|
843
|
+
// deferred noFeedbackPool that may be rescued by the proactive-maintenance
|
|
844
|
+
// (Layer 2) or high-salience (Layer 3) fallbacks below. The fully-skipped
|
|
845
|
+
// bucket has already been routed and its aggregated event emitted; we
|
|
846
|
+
// deliberately avoid spending DB/CPU on refs that the signal-delta gate
|
|
847
|
+
// rejected with feedback already on record.
|
|
848
|
+
const processableRefs = [...eligibleRefs, ...distillOnlyRefs];
|
|
849
|
+
const feedbackSummary = buildFeedbackSummaryMap({
|
|
850
|
+
processableRefs,
|
|
851
|
+
noFeedbackPool,
|
|
852
|
+
eventsCtx,
|
|
853
|
+
feedbackSinceCutoff,
|
|
854
|
+
});
|
|
855
|
+
const signalFiltered = processableRefs.filter((candidate) => feedbackSummary.get(candidate.ref)?.hasSignal === true);
|
|
856
|
+
const signalBearingSet = new Set(signalFiltered.map((r) => r.ref));
|
|
857
|
+
// Zero-feedback candidates for the proactive/high-salience fallbacks:
|
|
858
|
+
// processableRefs without a recent signal, plus the deferred noFeedbackPool.
|
|
859
|
+
// Dedupe by ref (the two sources are disjoint by construction, but guard
|
|
860
|
+
// against overlap defensively).
|
|
861
|
+
const noFeedbackSeen = new Set();
|
|
862
|
+
const noFeedbackCandidates = [];
|
|
863
|
+
for (const r of [...processableRefs.filter((r) => !signalBearingSet.has(r.ref)), ...noFeedbackPool]) {
|
|
864
|
+
if (noFeedbackSeen.has(r.ref))
|
|
852
865
|
continue;
|
|
853
|
-
|
|
854
|
-
|
|
866
|
+
noFeedbackSeen.add(r.ref);
|
|
867
|
+
noFeedbackCandidates.push(r);
|
|
868
|
+
}
|
|
869
|
+
const { retrievalCounts, lastUseMsForProactive } = fetchRetrievalSignals({
|
|
870
|
+
options,
|
|
871
|
+
primaryStashDir,
|
|
872
|
+
signalFiltered,
|
|
873
|
+
noFeedbackCandidates,
|
|
874
|
+
});
|
|
875
|
+
const proactive = selectProactiveMaintenanceLane({
|
|
876
|
+
scope,
|
|
877
|
+
improveProfile,
|
|
878
|
+
resolvedPlan,
|
|
879
|
+
eventsCtx,
|
|
880
|
+
noFeedbackCandidates,
|
|
881
|
+
lastReflectProposalTs,
|
|
882
|
+
lastDistillProposalTs,
|
|
883
|
+
retrievalCounts,
|
|
884
|
+
lastUseMsForProactive,
|
|
885
|
+
});
|
|
886
|
+
const proactiveRefs = proactive.proactiveRefs;
|
|
887
|
+
const proactiveMaintenanceSummary = proactive.proactiveMaintenanceSummary;
|
|
888
|
+
const highSalienceRefs = selectHighSalienceLane({
|
|
889
|
+
options,
|
|
890
|
+
improveProfile,
|
|
891
|
+
eventsCtx,
|
|
892
|
+
noFeedbackCandidates,
|
|
893
|
+
proactiveRefs,
|
|
894
|
+
lastReflectProposalTs,
|
|
895
|
+
});
|
|
896
|
+
// Record an in-memory skip action for every zero-feedback ref that the
|
|
897
|
+
// partition loop deferred to the proactive/high-salience fallbacks but those
|
|
898
|
+
// lanes then declined (not due, below threshold, or a prior reflect proposal
|
|
899
|
+
// already on record). These never make it into mergedRefs, so without this
|
|
900
|
+
// they would silently vanish from the run summary. No DB event is written
|
|
901
|
+
// here — these refs carry no signal at all, so there is nothing for the skip
|
|
902
|
+
// histogram to aggregate; the action log alone preserves the per-ref audit
|
|
903
|
+
// trail (mirrors the fully-skipped action above).
|
|
904
|
+
const rescuedSet = new Set([...proactiveRefs, ...highSalienceRefs].map((r) => r.ref));
|
|
905
|
+
for (const r of noFeedbackPool) {
|
|
906
|
+
if (rescuedSet.has(r.ref))
|
|
855
907
|
continue;
|
|
856
|
-
|
|
857
|
-
|
|
858
|
-
|
|
859
|
-
|
|
860
|
-
|
|
861
|
-
|
|
862
|
-
|
|
863
|
-
|
|
864
|
-
|
|
865
|
-
|
|
866
|
-
|
|
908
|
+
actions.push({
|
|
909
|
+
ref: r.ref,
|
|
910
|
+
mode: "distill-skipped",
|
|
911
|
+
result: { ok: true, reason: "no new signal since last proposal" },
|
|
912
|
+
});
|
|
913
|
+
}
|
|
914
|
+
// If the user explicitly scoped to a single ref, always act on it —
|
|
915
|
+
// skip the signal/retrieval filter entirely. The filter exists to avoid
|
|
916
|
+
// noisy "improve everything" runs; it should not gate an intentional
|
|
917
|
+
// per-ref invocation where the user's explicit choice is the signal.
|
|
918
|
+
//
|
|
919
|
+
// For type/all scope: only process refs with usage signals (recent feedback
|
|
920
|
+
// or a proactive/high-salience rescue). A stash with no signals has 0
|
|
921
|
+
// eligible refs — usage is the gate. Run `akm feedback <ref> --positive` or
|
|
922
|
+
// retrieve assets to bring them into the eligible pool.
|
|
923
|
+
// Layer-2 proactive refs join the eligible set alongside feedback-signal
|
|
924
|
+
// refs. The three sources are disjoint by construction (proactive draws from
|
|
925
|
+
// noFeedbackCandidates, and high-salience draws from the remainder), but
|
|
926
|
+
// dedupe defensively so a ref can never enter the loop twice.
|
|
927
|
+
// `requireFeedbackSignal` still suppresses all fallback sources for callers
|
|
928
|
+
// that want feedback-only runs.
|
|
929
|
+
const signalAndRetrievalRefs = dedupeRefs([...signalFiltered, ...proactiveRefs, ...highSalienceRefs]);
|
|
930
|
+
const mergedRefs = scope.mode === "ref" ? processableRefs : options.requireFeedbackSignal ? signalFiltered : signalAndRetrievalRefs;
|
|
931
|
+
// ── Attribution tagging: stamp each ref with the eligibility lane that
|
|
932
|
+
// selected it ──────────────────────────────────────────────────────────────
|
|
933
|
+
// Every reflect/distill proposal must record WHICH lane chose its source asset
|
|
934
|
+
// so downstream accept/reject/revert/retrieval outcomes can be sliced by lane
|
|
935
|
+
// (does the PROACTIVE lane produce value vs the reactive lanes?). We build the
|
|
936
|
+
// lane map here — the one place all three lanes are known — and stamp it onto
|
|
937
|
+
// each ImproveEligibleRef object. Because the ref objects are shared by
|
|
938
|
+
// reference across buckets, the stamp travels with the ref through the sort,
|
|
939
|
+
// disk-check, and loop stages down to the reflect/distill event emit sites and
|
|
940
|
+
// createProposal calls. See EligibilitySource for the lane vocabulary.
|
|
941
|
+
//
|
|
942
|
+
// Precedence (prefer the most specific reactive signal):
|
|
943
|
+
// scope > signal-delta > proactive > high-salience
|
|
944
|
+
// A ref with real feedback is attributed to feedback even if it was also due
|
|
945
|
+
// for proactive maintenance or had high encoding salience. We apply lanes
|
|
946
|
+
// weakest-first so the strongest overwrites; the explicit --scope <ref> bypass
|
|
947
|
+
// wins outright (user intent).
|
|
948
|
+
const eligibilitySourceByRef = new Map();
|
|
949
|
+
for (const r of highSalienceRefs)
|
|
950
|
+
eligibilitySourceByRef.set(r.ref, "high-salience");
|
|
951
|
+
for (const r of proactiveRefs)
|
|
952
|
+
eligibilitySourceByRef.set(r.ref, "proactive");
|
|
953
|
+
for (const r of signalFiltered)
|
|
954
|
+
eligibilitySourceByRef.set(r.ref, "signal-delta");
|
|
955
|
+
if (scope.mode === "ref") {
|
|
956
|
+
// O-2 (#365): explicit --scope <ref> bypass — every ref in processableRefs
|
|
957
|
+
// arrived via the scopeRefBypass branch, so attribute the whole set to scope.
|
|
958
|
+
for (const r of processableRefs)
|
|
959
|
+
eligibilitySourceByRef.set(r.ref, "scope");
|
|
960
|
+
}
|
|
961
|
+
for (const r of mergedRefs) {
|
|
962
|
+
// "unknown" is a genuine fallback, never a silent alias for signal-delta:
|
|
963
|
+
// only refs we truly cannot attribute land here (none in practice, since
|
|
964
|
+
// mergedRefs is always a subset of the four lanes above).
|
|
965
|
+
r.eligibilitySource = eligibilitySourceByRef.get(r.ref) ?? "unknown";
|
|
966
|
+
}
|
|
967
|
+
return {
|
|
968
|
+
actions,
|
|
969
|
+
distillCooledRefs,
|
|
970
|
+
preCooldownCount,
|
|
971
|
+
distillOnlyRefs,
|
|
972
|
+
fullySkippedCount,
|
|
973
|
+
feedbackSummary,
|
|
974
|
+
signalFiltered,
|
|
975
|
+
signalBearingSet,
|
|
976
|
+
retrievalCounts,
|
|
977
|
+
proactiveRefs,
|
|
978
|
+
proactiveMaintenanceSummary,
|
|
979
|
+
highSalienceRefs,
|
|
980
|
+
signalAndRetrievalRefs,
|
|
981
|
+
mergedRefs,
|
|
982
|
+
eligibilitySourceByRef,
|
|
983
|
+
};
|
|
984
|
+
}
|
|
985
|
+
/**
|
|
986
|
+
* The signal-delta partition of postCleanupRefs into the four buckets (pass:
|
|
987
|
+
* candidate-gather, phase 3). The 2026-05-26 54-ref incident semantics move
|
|
988
|
+
* VERBATIM — see the phase-2/3 comments inside.
|
|
989
|
+
*/
|
|
990
|
+
export function partitionBySignalDelta(args) {
|
|
991
|
+
const { scope, options, eventsCtx, postCleanupRefs, validationFailureRefs } = args;
|
|
992
|
+
const { latestFeedbackTs, lastReflectProposalTs, lastDistillProposalTs } = args.snapshot;
|
|
993
|
+
const actions = [];
|
|
994
|
+
// Refs the distill signal-delta gate rejected at planning time. The main
|
|
995
|
+
// loop reads this to skip distill for these refs without re-checking
|
|
996
|
+
// eligibility per iteration.
|
|
997
|
+
const distillCooledRefs = new Set();
|
|
998
|
+
const preCooldownCount = postCleanupRefs.length;
|
|
999
|
+
// ── Phase 3: partition postCleanupRefs by signal-delta eligibility ────────
|
|
1000
|
+
// Three buckets (validation failures are excluded entirely):
|
|
1001
|
+
// eligibleRefs — reflect signal-delta passes (full reflect+distill
|
|
1002
|
+
// loop path; distill guard remains in the loop for
|
|
1003
|
+
// refs that fail the distill signal-delta gate).
|
|
1004
|
+
// distillOnlyRefs — reflect blocked but distill signal-delta passes
|
|
1005
|
+
// AND ref is a distill candidate.
|
|
1006
|
+
// noFeedbackPool — neither signal-delta gate passes *and* the ref has
|
|
1007
|
+
// no recent feedback signal at all. These are NOT
|
|
1008
|
+
// skipped here: they are handed to the proactive
|
|
1009
|
+
// (Layer 2) and high-salience (Layer 3) fallbacks
|
|
1010
|
+
// below so never-rated assets can still be improved.
|
|
1011
|
+
// Only refs those lanes decline are fully skipped.
|
|
1012
|
+
// fullySkippedCount — has stale feedback but no signal delta → genuine
|
|
1013
|
+
// skip (counted, aggregated event emitted post-loop),
|
|
1014
|
+
// excluded from sort.
|
|
1015
|
+
const eligibleRefs = [];
|
|
1016
|
+
const distillOnlyRefs = [];
|
|
1017
|
+
// Zero-(recent-)feedback refs deferred to the proactive/high-salience fallbacks.
|
|
1018
|
+
const noFeedbackPool = [];
|
|
1019
|
+
let fullySkippedCount = 0;
|
|
1020
|
+
// O-2 (#365): explicit --scope <ref> bypasses every gate (user intent wins).
|
|
1021
|
+
const scopeRefBypass = scope.mode === "ref";
|
|
1022
|
+
for (const r of postCleanupRefs) {
|
|
1023
|
+
if (validationFailureRefs.has(r.ref))
|
|
1024
|
+
continue;
|
|
1025
|
+
if (scopeRefBypass) {
|
|
1026
|
+
eligibleRefs.push(r);
|
|
1027
|
+
continue;
|
|
1028
|
+
}
|
|
1029
|
+
const reflectOk = isSignalDeltaEligible(r.ref, latestFeedbackTs, lastReflectProposalTs);
|
|
1030
|
+
const distillOk = isSignalDeltaEligible(r.ref, latestFeedbackTs, lastDistillProposalTs);
|
|
1031
|
+
const isDistillCandidate = isDistillCandidateRef(r.ref, options.stashDir);
|
|
1032
|
+
if (reflectOk) {
|
|
1033
|
+
if (!distillOk && isDistillCandidate) {
|
|
1034
|
+
// Reflect passes the gate, distill does not — emit the synthetic
|
|
1035
|
+
// distill-skipped action and event up-front so the in-loop guard
|
|
1036
|
+
// does not have to re-derive eligibility.
|
|
1037
|
+
distillCooledRefs.add(r.ref);
|
|
1038
|
+
actions.push({ ref: r.ref, mode: "distill-skipped", result: { ok: true, reason: "distill signal-delta" } });
|
|
867
1039
|
appendEvent({
|
|
868
1040
|
eventType: "improve_skipped",
|
|
869
1041
|
ref: r.ref,
|
|
@@ -883,16 +1055,16 @@ export async function runImprovePreparationStage(args) {
|
|
|
883
1055
|
}
|
|
884
1056
|
else if (!latestFeedbackTs.has(r.ref)) {
|
|
885
1057
|
// Neither signal-delta gate passes AND there is no recent feedback signal
|
|
886
|
-
// at all. Rather than skip outright, defer to the
|
|
887
|
-
//
|
|
888
|
-
//
|
|
1058
|
+
// at all. Rather than skip outright, defer to the proactive-maintenance
|
|
1059
|
+
// and high-salience fallbacks below: a never-rated asset is exactly what
|
|
1060
|
+
// those lanes are meant to rescue. Refs those lanes decline are skipped there.
|
|
889
1061
|
noFeedbackPool.push(r);
|
|
890
1062
|
}
|
|
891
1063
|
else {
|
|
892
1064
|
// Has feedback on record but no signal delta since the last proposal —
|
|
893
1065
|
// genuinely fully skipped. Counted here; a single aggregated
|
|
894
1066
|
// improve_skipped event is emitted after the loop (mirrors
|
|
895
|
-
//
|
|
1067
|
+
// strategy_filtered_all_passes) instead of one event per ref.
|
|
896
1068
|
fullySkippedCount++;
|
|
897
1069
|
actions.push({
|
|
898
1070
|
ref: r.ref,
|
|
@@ -903,7 +1075,7 @@ export async function runImprovePreparationStage(args) {
|
|
|
903
1075
|
}
|
|
904
1076
|
// Emit ONE aggregated skip event for the fully-skipped bucket rather than one
|
|
905
1077
|
// improve_skipped event per ref (#592 pattern, mirrors
|
|
906
|
-
//
|
|
1078
|
+
// strategy_filtered_all_passes above). The per-ref loop previously produced
|
|
907
1079
|
// ~11K state.db writes per run on a large stash, the dominant contributor to
|
|
908
1080
|
// 900 s timeouts. The in-memory `actions` log keeps the per-ref detail for the
|
|
909
1081
|
// run summary; no downstream consumer needs a per-ref DB audit trail (health's
|
|
@@ -918,26 +1090,23 @@ export async function runImprovePreparationStage(args) {
|
|
|
918
1090
|
},
|
|
919
1091
|
}, eventsCtx);
|
|
920
1092
|
}
|
|
921
|
-
|
|
922
|
-
|
|
923
|
-
|
|
924
|
-
|
|
925
|
-
|
|
926
|
-
|
|
927
|
-
|
|
928
|
-
|
|
929
|
-
|
|
930
|
-
|
|
931
|
-
|
|
932
|
-
|
|
933
|
-
|
|
934
|
-
// 2. processableRefs entries that turn out to carry no recent feedback
|
|
935
|
-
// *signal* once feedbackSummary is computed below.
|
|
936
|
-
// (1) is added here; (2) is folded in after feedbackSummary is built.
|
|
1093
|
+
return {
|
|
1094
|
+
actions,
|
|
1095
|
+
distillCooledRefs,
|
|
1096
|
+
preCooldownCount,
|
|
1097
|
+
eligibleRefs,
|
|
1098
|
+
distillOnlyRefs,
|
|
1099
|
+
noFeedbackPool,
|
|
1100
|
+
fullySkippedCount,
|
|
1101
|
+
};
|
|
1102
|
+
}
|
|
1103
|
+
/** Bulk per-ref feedback summary in a SINGLE readEvents pass (candidate-gather). */
|
|
1104
|
+
function buildFeedbackSummaryMap(args) {
|
|
1105
|
+
const { processableRefs, noFeedbackPool, eventsCtx, feedbackSinceCutoff } = args;
|
|
937
1106
|
// Gap 6: only surface feedback signals from the last 30 days so that
|
|
938
1107
|
// ancient one-off feedback events don't permanently lock an asset into
|
|
939
1108
|
// every improve run. Assets with only stale signals fall through to the
|
|
940
|
-
// high-
|
|
1109
|
+
// proactive/high-salience fallbacks or are skipped until new signals arrive.
|
|
941
1110
|
// (FEEDBACK_SIGNAL_WINDOW_DAYS / feedbackSinceCutoff are already defined in
|
|
942
1111
|
// Phase 2 above for the signal-delta gate; we reuse them here.)
|
|
943
1112
|
// Pre-compute feedback summary per ref in a SINGLE bulk read so we don't
|
|
@@ -946,22 +1115,25 @@ export async function runImprovePreparationStage(args) {
|
|
|
946
1115
|
// above: one readEvents() call fetches ALL feedback events, then we aggregate
|
|
947
1116
|
// in-memory by ref — O(1) DB opens regardless of candidate set size.
|
|
948
1117
|
// Cover processableRefs *and* the deferred noFeedbackPool so utility/feedback
|
|
949
|
-
// ratios are available for any noFeedbackPool ref
|
|
1118
|
+
// ratios are available for any noFeedbackPool ref the fallback lanes rescue below.
|
|
950
1119
|
//
|
|
951
1120
|
// Behavioral note: positive/negative COUNTS are all-time (same as the old
|
|
952
1121
|
// per-ref readEvents call which had no `since` filter); hasSignal is bounded
|
|
953
1122
|
// to feedbackSinceCutoff (same as the old inline `(e.ts ?? "") >= cutoff` guard).
|
|
954
1123
|
const feedbackSummary = new Map();
|
|
955
1124
|
{
|
|
956
|
-
const
|
|
1125
|
+
const feedbackCandidates = [...processableRefs, ...noFeedbackPool];
|
|
1126
|
+
const feedbackCandidateSet = new Set(feedbackCandidates.map((r) => r.ref));
|
|
1127
|
+
// Map each candidate's single durable event key back to its display ref.
|
|
1128
|
+
const feedbackRefByDurableKey = new Map(feedbackCandidates.flatMap((r) => improveStateReadRefs(r.ref, r.itemRef).map((key) => [key, r.ref])));
|
|
957
1129
|
if (feedbackCandidateSet.size > 0) {
|
|
958
1130
|
// Fetch ALL feedback events in one query (no ref filter, no since filter =
|
|
959
1131
|
// single full table scan). Filtering per-ref in memory avoids N sequential
|
|
960
1132
|
// state.db opens — the dominant FD-leak path on large stashes.
|
|
961
1133
|
const { events: allFeedbackEvents } = readEvents({ type: "feedback" }, eventsCtx);
|
|
962
1134
|
for (const e of allFeedbackEvents) {
|
|
963
|
-
const ref = e.ref;
|
|
964
|
-
if (!ref
|
|
1135
|
+
const ref = e.ref ? feedbackRefByDurableKey.get(e.ref) : undefined;
|
|
1136
|
+
if (!ref)
|
|
965
1137
|
continue;
|
|
966
1138
|
const entry = feedbackSummary.get(ref) ?? { hasSignal: false, positive: 0, negative: 0 };
|
|
967
1139
|
const meta = e.metadata;
|
|
@@ -987,22 +1159,11 @@ export async function runImprovePreparationStage(args) {
|
|
|
987
1159
|
}
|
|
988
1160
|
}
|
|
989
1161
|
}
|
|
990
|
-
|
|
991
|
-
|
|
992
|
-
|
|
993
|
-
|
|
994
|
-
|
|
995
|
-
// plus the deferred noFeedbackPool. Dedupe by ref (the two sources are
|
|
996
|
-
// disjoint by construction, but guard against overlap defensively).
|
|
997
|
-
const noFeedbackSeen = new Set();
|
|
998
|
-
const noFeedbackCandidates = [];
|
|
999
|
-
for (const r of [...processableRefs.filter((r) => !signalBearingSet.has(r.ref)), ...noFeedbackPool]) {
|
|
1000
|
-
if (noFeedbackSeen.has(r.ref))
|
|
1001
|
-
continue;
|
|
1002
|
-
noFeedbackSeen.add(r.ref);
|
|
1003
|
-
noFeedbackCandidates.push(r);
|
|
1004
|
-
}
|
|
1005
|
-
let highRetrievalRefs = [];
|
|
1162
|
+
return feedbackSummary;
|
|
1163
|
+
}
|
|
1164
|
+
/** Retrieval counts + last-use timestamps for the candidate pools (candidate-gather). */
|
|
1165
|
+
function fetchRetrievalSignals(args) {
|
|
1166
|
+
const { options, primaryStashDir, signalFiltered, noFeedbackCandidates } = args;
|
|
1006
1167
|
// Retrieval counts for the zero-feedback pool, hoisted so the Layer-2
|
|
1007
1168
|
// proactive-maintenance selector below can reuse them without a second DB pass.
|
|
1008
1169
|
// Also fetch lastUseMs here for the proactive-maintenance recency term (plan §WS-1
|
|
@@ -1012,67 +1173,64 @@ export async function runImprovePreparationStage(args) {
|
|
|
1012
1173
|
let dbForRetrieval;
|
|
1013
1174
|
try {
|
|
1014
1175
|
dbForRetrieval = openExistingDatabase();
|
|
1015
|
-
|
|
1016
|
-
|
|
1017
|
-
|
|
1018
|
-
|
|
1019
|
-
|
|
1020
|
-
|
|
1021
|
-
|
|
1022
|
-
|
|
1023
|
-
|
|
1024
|
-
|
|
1025
|
-
|
|
1026
|
-
|
|
1027
|
-
|
|
1028
|
-
|
|
1029
|
-
|
|
1030
|
-
|
|
1031
|
-
|
|
1032
|
-
|
|
1033
|
-
|
|
1034
|
-
// from rescuing genuinely never-retrieved assets — the fallback is for
|
|
1035
|
-
// *retrieved* assets, not silent ones. Tracking growth in retrieval count
|
|
1036
|
-
// would require persisting the count in proposal metadata; deferred to a
|
|
1037
|
-
// follow-up.
|
|
1038
|
-
highRetrievalRefs = noFeedbackCandidates.filter((r) => {
|
|
1039
|
-
const count = retrievalCounts.get(r.ref) ?? 0;
|
|
1040
|
-
return count > 0 && count >= RETRIEVAL_COUNT_THRESHOLD && !lastReflectProposalTs.has(r.ref);
|
|
1176
|
+
// usage_events lives in state.db (Chunk-8 WI-8.3); entries stay in index.db,
|
|
1177
|
+
// so the retrieval-count reads take both handles.
|
|
1178
|
+
const dbForRetrievalIndex = dbForRetrieval;
|
|
1179
|
+
withStateDb((stateDb) => {
|
|
1180
|
+
const showEventCount = countUsageEventsByType(stateDb, "show");
|
|
1181
|
+
if (showEventCount === 0) {
|
|
1182
|
+
warn("Warning: show events not yet in usage_events — zero-feedback fallback will match only search-retrieved assets.");
|
|
1183
|
+
}
|
|
1184
|
+
// Fetch retrieval counts for ALL candidates — not only the zero-feedback pool.
|
|
1185
|
+
// Previously only noFeedbackCandidates were looked up, so feedback-bearing refs
|
|
1186
|
+
// had retrievalFreq=0 in computeSalience(), collapsing their retrievalSalience
|
|
1187
|
+
// to 0 regardless of actual use. Two assets of the same type — one
|
|
1188
|
+
// heavily-retrieved, one never-touched — would receive identical rankScores.
|
|
1189
|
+
// Fix (WS-1 blocker 3): union the feedback pool into the lookup.
|
|
1190
|
+
const allCandidateRefs = [...new Set([...signalFiltered, ...noFeedbackCandidates].map((r) => r.ref))];
|
|
1191
|
+
retrievalCounts = getRetrievalCounts(dbForRetrievalIndex, stateDb, allCandidateRefs, {
|
|
1192
|
+
sourceName: options.sourceName,
|
|
1193
|
+
stashDir: primaryStashDir,
|
|
1194
|
+
});
|
|
1041
1195
|
});
|
|
1196
|
+
lastUseMsForProactive = getLastUseMsByRef(dbForRetrieval, noFeedbackCandidates.map((r) => r.ref), primaryStashDir);
|
|
1042
1197
|
}
|
|
1043
1198
|
catch (err) {
|
|
1044
1199
|
rethrowIfTestIsolationError(err);
|
|
1045
|
-
// best-effort: if DB unavailable,
|
|
1200
|
+
// best-effort: if DB unavailable, retrievalCounts/lastUseMsForProactive stay empty
|
|
1046
1201
|
}
|
|
1047
1202
|
finally {
|
|
1048
1203
|
if (dbForRetrieval)
|
|
1049
1204
|
closeDatabase(dbForRetrieval);
|
|
1050
1205
|
}
|
|
1051
|
-
|
|
1052
|
-
|
|
1053
|
-
|
|
1054
|
-
|
|
1055
|
-
|
|
1206
|
+
return { retrievalCounts, lastUseMsForProactive };
|
|
1207
|
+
}
|
|
1208
|
+
/** Layer 2 — the proactive-maintenance selector lane (candidate-gather). */
|
|
1209
|
+
function selectProactiveMaintenanceLane(args) {
|
|
1210
|
+
const { scope, improveProfile, resolvedPlan, eventsCtx, noFeedbackCandidates, lastReflectProposalTs, lastDistillProposalTs, retrievalCounts, lastUseMsForProactive, } = args;
|
|
1211
|
+
// ── Layer 2: PROACTIVE MAINTENANCE SELECTOR (second eligibility source) ────
|
|
1212
|
+
// The signal-delta gate only surfaces assets with fresh feedback. It never
|
|
1213
|
+
// revisits a stable, high-value asset on a schedule, so on a quiet stash
|
|
1214
|
+
// useful assets drift stale and are never refreshed. When the
|
|
1215
|
+
// `proactiveMaintenance` process is enabled (DEFAULT OFF)
|
|
1056
1216
|
// and the run is whole-stash / type scope, this selector ranks the eligible
|
|
1057
1217
|
// population by a composite maintenance priority, gates on staleness ("due"),
|
|
1058
1218
|
// bounds to top-N, and folds the winners into the SAME candidate set the other
|
|
1059
|
-
//
|
|
1219
|
+
// sources feed — so they flow through the existing #580 empty-diff /
|
|
1060
1220
|
// cosmetic suppression and additive-distill gates. It adds no new mutation
|
|
1061
1221
|
// logic of its own. The due gate doubles as the rotation cooldown: a freshly
|
|
1062
1222
|
// reflected asset is excluded until it ages back past `dueDays`, so successive
|
|
1063
1223
|
// runs rotate through the due pool rather than re-selecting the same heads.
|
|
1064
1224
|
let proactiveRefs = [];
|
|
1065
1225
|
let proactiveMaintenanceSummary;
|
|
1066
|
-
const proactiveEnabled = scope.mode !== "ref" &&
|
|
1226
|
+
const proactiveEnabled = scope.mode !== "ref" && resolvedPlan.processes.proactiveMaintenance.enabled;
|
|
1067
1227
|
if (proactiveEnabled) {
|
|
1068
1228
|
const pmCfg = improveProfile.processes?.proactiveMaintenance;
|
|
1069
1229
|
const dueDays = pmCfg?.dueDays ?? DEFAULT_DUE_DAYS;
|
|
1070
1230
|
const maxPerRun = pmCfg?.maxPerRun ?? pmCfg?.limit ?? DEFAULT_MAX_PER_RUN;
|
|
1071
1231
|
// Candidate population: the zero-feedback / non-signal pool — exactly the
|
|
1072
|
-
// assets the
|
|
1073
|
-
|
|
1074
|
-
const alreadySelected = new Set(highRetrievalRefs.map((r) => r.ref));
|
|
1075
|
-
const pmCandidates = noFeedbackCandidates.filter((r) => !alreadySelected.has(r.ref));
|
|
1232
|
+
// assets the signal-delta gate would NOT pick this run.
|
|
1233
|
+
const pmCandidates = noFeedbackCandidates;
|
|
1076
1234
|
const selection = selectProactiveMaintenanceRefs({
|
|
1077
1235
|
candidates: pmCandidates,
|
|
1078
1236
|
lastReflectTs: lastReflectProposalTs,
|
|
@@ -1116,6 +1274,11 @@ export async function runImprovePreparationStage(args) {
|
|
|
1116
1274
|
`(${selection.neverReflected} never reflected, dueDays=${dueDays}, maxPerRun=${maxPerRun})`);
|
|
1117
1275
|
}
|
|
1118
1276
|
}
|
|
1277
|
+
return { proactiveRefs, proactiveMaintenanceSummary };
|
|
1278
|
+
}
|
|
1279
|
+
/** Layer 3 — the high-salience admission gate (#608/#644; candidate-gather). */
|
|
1280
|
+
function selectHighSalienceLane(args) {
|
|
1281
|
+
const { options, improveProfile, eventsCtx, noFeedbackCandidates, proactiveRefs, lastReflectProposalTs } = args;
|
|
1119
1282
|
// ── Layer 3: HIGH-SALIENCE ADMISSION GATE (#608) ──────────────────────────
|
|
1120
1283
|
// Zero-feedback refs whose encoding_salience (set at distill time by
|
|
1121
1284
|
// scoreEncodingSalience) exceeds the configured salienceThreshold are admitted
|
|
@@ -1128,11 +1291,11 @@ export async function runImprovePreparationStage(args) {
|
|
|
1128
1291
|
//
|
|
1129
1292
|
// Cooldown: a ref qualifies at most once — when no prior reflect proposal
|
|
1130
1293
|
// exists for it (`!lastReflectProposalTs.has`). Without this guard the lane
|
|
1131
|
-
// re-selects the same high-salience refs on EVERY run (
|
|
1294
|
+
// re-selects the same high-salience refs on EVERY run (promotion emits a
|
|
1132
1295
|
// `promoted` event, not `feedback`, so the ref never leaves
|
|
1133
1296
|
// noFeedbackCandidates), burning LLM calls and churning the asset. This
|
|
1134
|
-
// mirrors the
|
|
1135
|
-
//
|
|
1297
|
+
// mirrors the same `!lastReflectProposalTs.has(r.ref)` once-per-asset
|
|
1298
|
+
// semantics the other "rescue" lanes share.
|
|
1136
1299
|
//
|
|
1137
1300
|
// Content-provenance gate (#644 follow-up): the row must ALSO carry a genuine
|
|
1138
1301
|
// content-derived encoding score (`isContentEncodingRow`). Otherwise the lane
|
|
@@ -1144,11 +1307,11 @@ export async function runImprovePreparationStage(args) {
|
|
|
1144
1307
|
// earn retrieval/feedback signal via the other lanes. This PRESERVES #608's
|
|
1145
1308
|
// intent — distilled assets (the lane's real targets) keep their real content
|
|
1146
1309
|
// score and still qualify — while cutting the type-stub waste. See §5 F1 of
|
|
1147
|
-
//
|
|
1310
|
+
// #608/#644.
|
|
1148
1311
|
const highSalienceRefs = [];
|
|
1149
1312
|
const salienceCfg = (options.config ?? loadConfig()).improve?.salience;
|
|
1150
1313
|
const salienceThreshold = salienceCfg?.salienceThreshold ?? 0.75;
|
|
1151
|
-
const
|
|
1314
|
+
const proactiveSelectedSet = new Set(proactiveRefs.map((r) => r.ref));
|
|
1152
1315
|
try {
|
|
1153
1316
|
withStateDb((dbForHighSalience) => {
|
|
1154
1317
|
// Derive the cap from the resolved reflect limit (mirrors improve.ts's
|
|
@@ -1156,15 +1319,15 @@ export async function runImprovePreparationStage(args) {
|
|
|
1156
1319
|
// collapse the lane to exactly 1 ref via the bare `?? 10` fallback.
|
|
1157
1320
|
const effectiveLimit = options.limit ?? improveProfile?.processes?.reflect?.limit ?? improveProfile.limit ?? 10;
|
|
1158
1321
|
const highSalienceCap = Math.max(1, Math.floor(effectiveLimit * 0.1));
|
|
1159
|
-
const candidates = noFeedbackCandidates.filter((r) => !
|
|
1322
|
+
const candidates = noFeedbackCandidates.filter((r) => !proactiveSelectedSet.has(r.ref));
|
|
1160
1323
|
// Collect ALL qualifying candidates, then take the top-N BY SCORE — the
|
|
1161
1324
|
// previous first-N-in-scan-order break meant a higher-salience candidate
|
|
1162
1325
|
// found later in the scan lost its slot to an earlier lower-scoring one.
|
|
1163
1326
|
const qualifying = [];
|
|
1164
1327
|
for (const r of candidates) {
|
|
1165
|
-
const row =
|
|
1328
|
+
const row = readAssetSalienceForImproveRef(dbForHighSalience, r.ref, r.itemRef);
|
|
1166
1329
|
if (row &&
|
|
1167
|
-
isContentEncodingRow(row
|
|
1330
|
+
isContentEncodingRow(row) &&
|
|
1168
1331
|
row.encoding_salience >= salienceThreshold &&
|
|
1169
1332
|
!lastReflectProposalTs.has(r.ref)) {
|
|
1170
1333
|
qualifying.push({ ref: r, score: row.encoding_salience });
|
|
@@ -1184,108 +1347,39 @@ export async function runImprovePreparationStage(args) {
|
|
|
1184
1347
|
info(`[improve] high-salience lane admitted ${highSalienceRefs.length} content-scored ref(s) ` +
|
|
1185
1348
|
`(threshold=${salienceThreshold}, requires content-derived encoding_source)`);
|
|
1186
1349
|
}
|
|
1187
|
-
|
|
1188
|
-
|
|
1189
|
-
|
|
1190
|
-
|
|
1191
|
-
|
|
1192
|
-
|
|
1193
|
-
|
|
1194
|
-
|
|
1195
|
-
|
|
1196
|
-
|
|
1197
|
-
|
|
1198
|
-
|
|
1199
|
-
|
|
1200
|
-
|
|
1201
|
-
|
|
1202
|
-
|
|
1203
|
-
}
|
|
1204
|
-
// If the user explicitly scoped to a single ref, always act on it —
|
|
1205
|
-
// skip the signal/retrieval filter entirely. The filter exists to avoid
|
|
1206
|
-
// noisy "improve everything" runs; it should not gate an intentional
|
|
1207
|
-
// per-ref invocation where the user's explicit choice is the signal.
|
|
1208
|
-
//
|
|
1209
|
-
// For type/all scope: only process refs with usage signals (recent feedback
|
|
1210
|
-
// or sufficient retrievals). A stash with no signals has 0 eligible refs —
|
|
1211
|
-
// usage is the gate. Run `akm feedback <ref> --positive` or retrieve assets
|
|
1212
|
-
// to bring them into the eligible pool.
|
|
1213
|
-
// Layer-2 proactive refs join the eligible set alongside feedback-signal and
|
|
1214
|
-
// high-retrieval (P0-A) refs. The four sources are disjoint by construction
|
|
1215
|
-
// (proactive draws from noFeedbackCandidates with the P0-A picks removed, and
|
|
1216
|
-
// high-salience draws from the remainder), but dedupe defensively so a ref can
|
|
1217
|
-
// never enter the loop twice. `requireFeedbackSignal` still suppresses all
|
|
1218
|
-
// fallback sources for callers that want feedback-only runs.
|
|
1219
|
-
const signalAndRetrievalRefs = dedupeRefs([
|
|
1220
|
-
...signalFiltered,
|
|
1221
|
-
...highRetrievalRefs,
|
|
1222
|
-
...proactiveRefs,
|
|
1223
|
-
...highSalienceRefs,
|
|
1224
|
-
]);
|
|
1225
|
-
let mergedRefs = scope.mode === "ref" ? processableRefs : options.requireFeedbackSignal ? signalFiltered : signalAndRetrievalRefs;
|
|
1226
|
-
// ── Attribution tagging: stamp each ref with the eligibility lane that
|
|
1227
|
-
// selected it ──────────────────────────────────────────────────────────────
|
|
1228
|
-
// Every reflect/distill proposal must record WHICH lane chose its source asset
|
|
1229
|
-
// so downstream accept/reject/revert/retrieval outcomes can be sliced by lane
|
|
1230
|
-
// (does the PROACTIVE lane produce value vs the reactive lanes?). We build the
|
|
1231
|
-
// lane map here — the one place all four lanes are known — and stamp it onto
|
|
1232
|
-
// each ImproveEligibleRef object. Because the ref objects are shared by
|
|
1233
|
-
// reference across buckets, the stamp travels with the ref through the sort,
|
|
1234
|
-
// disk-check, and loop stages down to the reflect/distill event emit sites and
|
|
1235
|
-
// createProposal calls. See EligibilitySource for the lane vocabulary.
|
|
1236
|
-
//
|
|
1237
|
-
// Precedence (prefer the most specific reactive signal):
|
|
1238
|
-
// scope > signal-delta > high-retrieval > proactive > high-salience
|
|
1239
|
-
// A ref with real feedback is attributed to feedback even if it was also due
|
|
1240
|
-
// for proactive maintenance or had high encoding salience. We apply lanes
|
|
1241
|
-
// weakest-first so the strongest overwrites; the explicit --scope <ref> bypass
|
|
1242
|
-
// wins outright (user intent).
|
|
1243
|
-
const eligibilitySourceByRef = new Map();
|
|
1244
|
-
for (const r of highSalienceRefs)
|
|
1245
|
-
eligibilitySourceByRef.set(r.ref, "high-salience");
|
|
1246
|
-
for (const r of proactiveRefs)
|
|
1247
|
-
eligibilitySourceByRef.set(r.ref, "proactive");
|
|
1248
|
-
for (const r of highRetrievalRefs)
|
|
1249
|
-
eligibilitySourceByRef.set(r.ref, "high-retrieval");
|
|
1250
|
-
for (const r of signalFiltered)
|
|
1251
|
-
eligibilitySourceByRef.set(r.ref, "signal-delta");
|
|
1252
|
-
if (scope.mode === "ref") {
|
|
1253
|
-
// O-2 (#365): explicit --scope <ref> bypass — every ref in processableRefs
|
|
1254
|
-
// arrived via the scopeRefBypass branch, so attribute the whole set to scope.
|
|
1255
|
-
for (const r of processableRefs)
|
|
1256
|
-
eligibilitySourceByRef.set(r.ref, "scope");
|
|
1257
|
-
}
|
|
1258
|
-
for (const r of mergedRefs) {
|
|
1259
|
-
// "unknown" is a genuine fallback, never a silent alias for signal-delta:
|
|
1260
|
-
// only refs we truly cannot attribute land here (none in practice, since
|
|
1261
|
-
// mergedRefs is always a subset of the four lanes above).
|
|
1262
|
-
r.eligibilitySource = eligibilitySourceByRef.get(r.ref) ?? "unknown";
|
|
1263
|
-
}
|
|
1350
|
+
return highSalienceRefs;
|
|
1351
|
+
}
|
|
1352
|
+
/**
|
|
1353
|
+
* Pass: salience-score — the WS-2 outcome loop, the WS-1 salience vector
|
|
1354
|
+
* computation (#644 provenance preserved), persistence + rank-change report,
|
|
1355
|
+
* and the forgetting-safety injection. The valence-score call sites
|
|
1356
|
+
* (computeValenceScore) live VERBATIM inside the outcome/persist sub-passes.
|
|
1357
|
+
* Mutates the shared eligibilitySourceByRef map and ref objects in place —
|
|
1358
|
+
* attribution identity is load-bearing (see the candidate-gather comments).
|
|
1359
|
+
*/
|
|
1360
|
+
function scoreSalience(args) {
|
|
1361
|
+
const { scope, options, primaryStashDir, eventsCtx, eligibilitySourceByRef, feedbackSummary, retrievalCounts, signalFiltered, proactiveRefs, highSalienceRefs, } = args;
|
|
1362
|
+
const mergedRefs = args.mergedRefs;
|
|
1363
|
+
// Chunk-5 flip F5e — resolve each candidate's durable item_ref ONCE for this
|
|
1364
|
+
// pass (the write/read key source for the outcome + salience state writers).
|
|
1365
|
+
const itemRefByRef = buildItemRefByRef(mergedRefs);
|
|
1264
1366
|
// WS-1 — Unified salience vector (S1 seam).
|
|
1265
1367
|
//
|
|
1266
|
-
//
|
|
1267
|
-
//
|
|
1268
|
-
// formula). WS-1 converges them into one `computeSalience()` call per ref, with
|
|
1368
|
+
// WS-1 converges utility, valence, and proactive-maintenance signals into one
|
|
1369
|
+
// `computeSalience()` call per ref, with
|
|
1269
1370
|
// three independently-stored sub-scores and one documented rankScore projection.
|
|
1270
1371
|
//
|
|
1271
|
-
// Migration note: if a profile still has `symmetricValence` set, emit a one-time
|
|
1272
|
-
// warning — its behaviour (symmetric |valence| attention) is now always-on as
|
|
1273
|
-
// part of the salience vector, so the knob is a no-op and will be removed in 0.10.
|
|
1274
|
-
if (improveProfile.symmetricValence === true) {
|
|
1275
|
-
warn("[improve] Profile option 'symmetricValence' is deprecated (WS-1 salience vector). " +
|
|
1276
|
-
"Symmetric valence is now always active; remove the option from your improve profile.");
|
|
1277
|
-
}
|
|
1278
1372
|
// Fetch last-use timestamps from the index DB for the full merged set so the
|
|
1279
1373
|
// recency term in retrievalSalience is genuinely decayable (plan §WS-1 step 2).
|
|
1280
1374
|
// This reuses the index DB opened earlier for retrieval counts; a separate
|
|
1281
1375
|
// lightweight open is used here to avoid holding the connection longer than needed.
|
|
1282
1376
|
let lastUseMsByRef = new Map();
|
|
1283
|
-
//
|
|
1377
|
+
// Health and outcome reporting consume the utility projection.
|
|
1284
1378
|
const utilityMap = buildUtilityMap(mergedRefs);
|
|
1285
1379
|
let dbForSalience;
|
|
1286
1380
|
try {
|
|
1287
1381
|
dbForSalience = openExistingDatabase();
|
|
1288
|
-
lastUseMsByRef = getLastUseMsByRef(dbForSalience, mergedRefs.map((r) => r.ref));
|
|
1382
|
+
lastUseMsByRef = getLastUseMsByRef(dbForSalience, mergedRefs.map((r) => r.ref), primaryStashDir);
|
|
1289
1383
|
}
|
|
1290
1384
|
catch (err) {
|
|
1291
1385
|
rethrowIfTestIsolationError(err);
|
|
@@ -1295,6 +1389,49 @@ export async function runImprovePreparationStage(args) {
|
|
|
1295
1389
|
if (dbForSalience)
|
|
1296
1390
|
closeDatabase(dbForSalience);
|
|
1297
1391
|
}
|
|
1392
|
+
const outcomeSalienceByRef = updateOutcomeScores({
|
|
1393
|
+
mergedRefs,
|
|
1394
|
+
itemRefByRef,
|
|
1395
|
+
feedbackSummary,
|
|
1396
|
+
retrievalCounts,
|
|
1397
|
+
lastUseMsByRef,
|
|
1398
|
+
utilityMap,
|
|
1399
|
+
primaryStashDir,
|
|
1400
|
+
eventsCtx,
|
|
1401
|
+
});
|
|
1402
|
+
const { salienceMap, nowForSalience } = computeSalienceVectors({
|
|
1403
|
+
mergedRefs,
|
|
1404
|
+
itemRefByRef,
|
|
1405
|
+
options,
|
|
1406
|
+
eventsCtx,
|
|
1407
|
+
retrievalCounts,
|
|
1408
|
+
lastUseMsByRef,
|
|
1409
|
+
utilityMap,
|
|
1410
|
+
outcomeSalienceByRef,
|
|
1411
|
+
});
|
|
1412
|
+
const pendingForgettingRefs = persistSalienceAndReportRanks({
|
|
1413
|
+
salienceMap,
|
|
1414
|
+
itemRefByRef,
|
|
1415
|
+
utilityMap,
|
|
1416
|
+
feedbackSummary,
|
|
1417
|
+
options,
|
|
1418
|
+
eventsCtx,
|
|
1419
|
+
nowForSalience,
|
|
1420
|
+
});
|
|
1421
|
+
const finalMergedRefs = applyForgettingSafety({
|
|
1422
|
+
pendingForgettingRefs,
|
|
1423
|
+
scope,
|
|
1424
|
+
mergedRefs,
|
|
1425
|
+
eligibilitySourceByRef,
|
|
1426
|
+
highSalienceRefs,
|
|
1427
|
+
proactiveRefs,
|
|
1428
|
+
signalFiltered,
|
|
1429
|
+
});
|
|
1430
|
+
return { mergedRefs: finalMergedRefs, utilityMap, lastUseMsByRef, salienceMap, nowForSalience };
|
|
1431
|
+
}
|
|
1432
|
+
/** WS-2 — update asset_outcome for the merged set; returns outcomeSalience by ref. */
|
|
1433
|
+
function updateOutcomeScores(args) {
|
|
1434
|
+
const { mergedRefs, itemRefByRef, feedbackSummary, retrievalCounts, lastUseMsByRef, utilityMap, primaryStashDir, eventsCtx, } = args;
|
|
1298
1435
|
// ── WS-2 Outcome loop ─────────────────────────────────────────────────────
|
|
1299
1436
|
//
|
|
1300
1437
|
// Update asset_outcome for every ref in the merged set BEFORE computing the
|
|
@@ -1336,7 +1473,9 @@ export async function runImprovePreparationStage(args) {
|
|
|
1336
1473
|
const valenceResult = computeValenceScore(fb);
|
|
1337
1474
|
try {
|
|
1338
1475
|
const result = updateAssetOutcome(outcomeDb, {
|
|
1339
|
-
|
|
1476
|
+
// Key by item_ref when resolved, else by the conceptId. Keep
|
|
1477
|
+
// rawOutcomeScores keyed by r.ref, its in-memory identity.
|
|
1478
|
+
ref: outcomeWriteKey(r.ref, itemRefByRef),
|
|
1340
1479
|
currentRetrievalCount: retrievalCounts.get(r.ref) ?? 0,
|
|
1341
1480
|
lastRetrievedAt: lastUseMsByRef.get(r.ref) ?? 0,
|
|
1342
1481
|
acceptedChangeCount: acceptedCountByRef.get(r.ref) ?? 0,
|
|
@@ -1361,10 +1500,7 @@ export async function runImprovePreparationStage(args) {
|
|
|
1361
1500
|
if (row.outcome_score > maxOutcomeScore)
|
|
1362
1501
|
maxOutcomeScore = row.outcome_score;
|
|
1363
1502
|
}
|
|
1364
|
-
//
|
|
1365
|
-
// existed can sit above the ceiling (live max was 3.13). Without this
|
|
1366
|
-
// clip they inflate the normalisation denominator and floor everyone
|
|
1367
|
-
// else's outcomeSalience (#691 follow-up).
|
|
1503
|
+
// Keep the normalization denominator within the writer's score bounds.
|
|
1368
1504
|
maxOutcomeScore = Math.min(maxOutcomeScore, OUTCOME_SCORE_MAX);
|
|
1369
1505
|
// Proxy-adequacy tripwire (two-tailed): inverted (corr < −0.3) and
|
|
1370
1506
|
// dead (|corr| < 0.1 at n ≥ 500) both emit health events.
|
|
@@ -1402,11 +1538,18 @@ export async function runImprovePreparationStage(args) {
|
|
|
1402
1538
|
}
|
|
1403
1539
|
// Also fetch outcome scores for refs NOT updated this run (stale or absent)
|
|
1404
1540
|
// so the outcomeSalience read path works for all refs in the batch.
|
|
1541
|
+
// Chunk-5 flip F5e — query by each missing ref's WRITE key (item_ref,
|
|
1542
|
+
// else bare) and map the stored-key result back to the bare `r.ref`
|
|
1543
|
+
// identity that outcomeSalienceByRef is keyed on.
|
|
1405
1544
|
const missingRefs = mergedRefs.map((r) => r.ref).filter((ref) => !rawOutcomeScores.has(ref));
|
|
1406
1545
|
if (missingRefs.length > 0) {
|
|
1407
|
-
const
|
|
1408
|
-
for (const
|
|
1409
|
-
|
|
1546
|
+
const refByWriteKey = new Map();
|
|
1547
|
+
for (const ref of missingRefs)
|
|
1548
|
+
refByWriteKey.set(outcomeWriteKey(ref, itemRefByRef), ref);
|
|
1549
|
+
const storedScores = getOutcomeScoresByRef(outcomeDb, [...refByWriteKey.keys()]);
|
|
1550
|
+
for (const [writeKey, score] of storedScores) {
|
|
1551
|
+
const bareRef = refByWriteKey.get(writeKey) ?? writeKey;
|
|
1552
|
+
outcomeSalienceByRef.set(bareRef, outcomeScoreToSalience(score, maxOutcomeScore));
|
|
1410
1553
|
}
|
|
1411
1554
|
}
|
|
1412
1555
|
}, { path: eventsCtx?.dbPath, borrowed: eventsCtx?.db });
|
|
@@ -1415,6 +1558,11 @@ export async function runImprovePreparationStage(args) {
|
|
|
1415
1558
|
rethrowIfTestIsolationError(err);
|
|
1416
1559
|
// best-effort: outcome failures never block salience computation
|
|
1417
1560
|
}
|
|
1561
|
+
return outcomeSalienceByRef;
|
|
1562
|
+
}
|
|
1563
|
+
/** WS-1 — compute the salience vector per ref (#644 provenance preserved). */
|
|
1564
|
+
function computeSalienceVectors(args) {
|
|
1565
|
+
const { mergedRefs, itemRefByRef, options, eventsCtx, retrievalCounts, lastUseMsByRef, utilityMap, outcomeSalienceByRef, } = args;
|
|
1418
1566
|
// Compute the salience vector for every ref in the merged set.
|
|
1419
1567
|
// retrievalCounts now covers the full candidate set (feedback-bearing + zero-feedback)
|
|
1420
1568
|
// so feedback refs get their genuine retrieval frequency, not a 0-floor fallback.
|
|
@@ -1440,9 +1588,8 @@ export async function runImprovePreparationStage(args) {
|
|
|
1440
1588
|
try {
|
|
1441
1589
|
withStateDb((dbForStoredEncoding) => {
|
|
1442
1590
|
for (const r of mergedRefs) {
|
|
1443
|
-
const
|
|
1444
|
-
|
|
1445
|
-
if (row && isContentEncodingRow(row, type)) {
|
|
1591
|
+
const row = readAssetSalienceForImproveRef(dbForStoredEncoding, r.ref, itemRefByRef.get(r.ref));
|
|
1592
|
+
if (row && isContentEncodingRow(row)) {
|
|
1446
1593
|
storedEncodingByRef.set(r.ref, row.encoding_salience);
|
|
1447
1594
|
}
|
|
1448
1595
|
}
|
|
@@ -1453,7 +1600,7 @@ export async function runImprovePreparationStage(args) {
|
|
|
1453
1600
|
// best-effort: if DB unavailable, fall back to type-weight stub (prior behaviour)
|
|
1454
1601
|
}
|
|
1455
1602
|
for (const r of mergedRefs) {
|
|
1456
|
-
const type =
|
|
1603
|
+
const type = assetTypeOf(r.ref);
|
|
1457
1604
|
const sizeBytes = (() => {
|
|
1458
1605
|
const fp = r.filePath;
|
|
1459
1606
|
if (!fp)
|
|
@@ -1482,6 +1629,40 @@ export async function runImprovePreparationStage(args) {
|
|
|
1482
1629
|
});
|
|
1483
1630
|
salienceMap.set(r.ref, vector);
|
|
1484
1631
|
}
|
|
1632
|
+
return { salienceMap, nowForSalience };
|
|
1633
|
+
}
|
|
1634
|
+
/** 1-indexed rank positions sorted by score desc (deterministic ref-asc tie-break). */
|
|
1635
|
+
function toRankPositions(scores) {
|
|
1636
|
+
const sorted = [...scores.entries()].sort(([refA, a], [refB, b]) => b !== a ? b - a : refA < refB ? -1 : refA > refB ? 1 : 0);
|
|
1637
|
+
return new Map(sorted.map(([ref], i) => [ref, i + 1]));
|
|
1638
|
+
}
|
|
1639
|
+
/**
|
|
1640
|
+
* Chunk-5 flip F5e — the durable salience write-key maps for one improve pass.
|
|
1641
|
+
* `wk(ref)` is a pool asset's write key (item_ref, else conceptId);
|
|
1642
|
+
* `normalizeStoredKey` maps each in-pool asset's durable spelling to its write
|
|
1643
|
+
* key (no stashSize double-count); and
|
|
1644
|
+
* `refByWriteKey` reverses a write key back to its filesystem-facing bare ref.
|
|
1645
|
+
*/
|
|
1646
|
+
function buildSalienceWriteKeyMaps(itemRefByRef) {
|
|
1647
|
+
const wk = (ref) => salienceWriteKey(ref, itemRefByRef);
|
|
1648
|
+
const normalizeStoredKey = new Map();
|
|
1649
|
+
const refByWriteKey = new Map();
|
|
1650
|
+
for (const [ref, itemRef] of itemRefByRef) {
|
|
1651
|
+
const writeKey = wk(ref);
|
|
1652
|
+
refByWriteKey.set(writeKey, ref);
|
|
1653
|
+
for (const spelling of improveStateReadRefs(ref, itemRef)) {
|
|
1654
|
+
normalizeStoredKey.set(spelling, writeKey);
|
|
1655
|
+
}
|
|
1656
|
+
}
|
|
1657
|
+
return { wk, normalizeStoredKey, refByWriteKey };
|
|
1658
|
+
}
|
|
1659
|
+
/** Persist salience vectors + the WS-1 step-7 rank-change/forgetting report. */
|
|
1660
|
+
function persistSalienceAndReportRanks(args) {
|
|
1661
|
+
const { salienceMap, itemRefByRef, utilityMap, feedbackSummary, options, eventsCtx, nowForSalience } = args;
|
|
1662
|
+
// Chunk-5 flip F5e — the WRITE-key space. salienceMap stays keyed by each
|
|
1663
|
+
// candidate's own short `r.ref`; the state.db boundary keys by item_ref when
|
|
1664
|
+
// available and otherwise by conceptId.
|
|
1665
|
+
const { wk, normalizeStoredKey, refByWriteKey } = buildSalienceWriteKeyMaps(itemRefByRef);
|
|
1485
1666
|
// Persist salience vectors to state.db (best-effort, non-blocking).
|
|
1486
1667
|
// The canonical store enables WS-3 homeostatic demotion and WS-2 outcome reads.
|
|
1487
1668
|
//
|
|
@@ -1504,7 +1685,7 @@ export async function runImprovePreparationStage(args) {
|
|
|
1504
1685
|
// stash-wide), but it is the most faithful comparison possible at cutover and
|
|
1505
1686
|
// allows the top-200→below-500 forgetting guard to fire if the formula change
|
|
1506
1687
|
// dramatically reorders the candidate pool.
|
|
1507
|
-
//
|
|
1688
|
+
// WS-1 step 7 — the stash-wide
|
|
1508
1689
|
// ordering was unreconstructable (no prior state.db snapshot), so this candidate-
|
|
1509
1690
|
// pool partial reconstruction is the documented resolution for the first-run case.
|
|
1510
1691
|
// Emit `improve_salience_first_run` to mark the cutover moment and include the
|
|
@@ -1534,7 +1715,21 @@ export async function runImprovePreparationStage(args) {
|
|
|
1534
1715
|
// Step 7: stash-wide rank-change report BEFORE overwriting the table.
|
|
1535
1716
|
//
|
|
1536
1717
|
// Load ALL existing rows so rank positions are stash-relative, not pool-relative.
|
|
1537
|
-
|
|
1718
|
+
// Source-scope by the `<bundle>//` prefix and fold each in-pool asset's
|
|
1719
|
+
// stored spelling onto its single write key so
|
|
1720
|
+
// the merge below never double-counts one asset across two spellings.
|
|
1721
|
+
const allStoredScores = getAllRankScores(stateDb);
|
|
1722
|
+
const existingAllScores = new Map();
|
|
1723
|
+
for (const [ref, score] of allStoredScores) {
|
|
1724
|
+
if (options.sourceName) {
|
|
1725
|
+
const boundary = ref.indexOf("//");
|
|
1726
|
+
const prefix = boundary >= 0 ? ref.slice(0, boundary) : undefined;
|
|
1727
|
+
const belongs = prefix === options.sourceName;
|
|
1728
|
+
if (!belongs)
|
|
1729
|
+
continue;
|
|
1730
|
+
}
|
|
1731
|
+
existingAllScores.set(normalizeStoredKey.get(ref) ?? ref, score);
|
|
1732
|
+
}
|
|
1538
1733
|
if (existingAllScores.size === 0) {
|
|
1539
1734
|
// Scenario A: first WS-1 run — table empty.
|
|
1540
1735
|
//
|
|
@@ -1546,7 +1741,7 @@ export async function runImprovePreparationStage(args) {
|
|
|
1546
1741
|
// Limitation: this covers only the current-run candidate pool, not the full
|
|
1547
1742
|
// stash. The stash-wide ordering was never persisted (asset_salience is a new
|
|
1548
1743
|
// WS-1 table), so this is the most faithful comparison possible at cutover.
|
|
1549
|
-
//
|
|
1744
|
+
// WS-1 step 7.
|
|
1550
1745
|
const reconstructedOldScores = new Map();
|
|
1551
1746
|
for (const ref of salienceMap.keys()) {
|
|
1552
1747
|
const utility = utilityMap.get(ref) ?? 0;
|
|
@@ -1555,12 +1750,8 @@ export async function runImprovePreparationStage(args) {
|
|
|
1555
1750
|
reconstructedOldScores.set(ref, utility * UTILITY_WEIGHT + attention * FEEDBACK_WEIGHT);
|
|
1556
1751
|
}
|
|
1557
1752
|
// Assign 1-indexed rank positions sorted by score desc (tie-break: ref asc).
|
|
1558
|
-
const
|
|
1559
|
-
|
|
1560
|
-
return new Map(sorted.map(([ref], i) => [ref, i + 1]));
|
|
1561
|
-
};
|
|
1562
|
-
const oldRanks = toRanks(reconstructedOldScores);
|
|
1563
|
-
const newRanks = toRanks(new Map([...salienceMap.entries()].map(([ref, v]) => [ref, v.rankScore])));
|
|
1753
|
+
const oldRanks = toRankPositions(reconstructedOldScores);
|
|
1754
|
+
const newRanks = toRankPositions(new Map([...salienceMap.entries()].map(([ref, v]) => [ref, v.rankScore])));
|
|
1564
1755
|
const firstRunReport = buildRankChangeReport(oldRanks, newRanks);
|
|
1565
1756
|
if (firstRunReport.forgettingCandidates.length > 0) {
|
|
1566
1757
|
warn(`[improve/salience] WS-1 first-run rank-change report: ${firstRunReport.forgettingCandidates.length} asset(s) fell from top-200 to below position 500 (cutover formula change). ` +
|
|
@@ -1575,7 +1766,7 @@ export async function runImprovePreparationStage(args) {
|
|
|
1575
1766
|
ref: undefined,
|
|
1576
1767
|
metadata: {
|
|
1577
1768
|
candidateCount: salienceMap.size,
|
|
1578
|
-
note: "first WS-1 salience run — partial reconstruction of old combinedEligibilityScore ordering for candidate pool (stash-wide ordering not available);
|
|
1769
|
+
note: "first WS-1 salience run — partial reconstruction of old combinedEligibilityScore ordering for candidate pool (stash-wide ordering not available); WS-1 step 7",
|
|
1579
1770
|
forgettingCandidates: firstRunReport.forgettingCandidates.length,
|
|
1580
1771
|
topDrops: firstRunReport.forgettingCandidates.slice(0, 10).map((e) => ({
|
|
1581
1772
|
ref: e.ref,
|
|
@@ -1594,15 +1785,13 @@ export async function runImprovePreparationStage(args) {
|
|
|
1594
1785
|
// picture of what the new ordering looks like after this run.
|
|
1595
1786
|
const mergedNewScores = new Map(existingAllScores);
|
|
1596
1787
|
for (const [ref, vector] of salienceMap) {
|
|
1597
|
-
|
|
1788
|
+
// Chunk-5 flip F5e — key this run's fresh scores by the WRITE key so
|
|
1789
|
+
// they overwrite (never duplicate) the same asset's normalized stored row.
|
|
1790
|
+
mergedNewScores.set(wk(ref), vector.rankScore);
|
|
1598
1791
|
}
|
|
1599
1792
|
// Assign 1-indexed rank positions sorted by score desc (tie-break: ref asc).
|
|
1600
|
-
const
|
|
1601
|
-
|
|
1602
|
-
return new Map(sorted.map(([ref], i) => [ref, i + 1]));
|
|
1603
|
-
};
|
|
1604
|
-
const oldRanks = toRanks(existingAllScores);
|
|
1605
|
-
const newRanks = toRanks(mergedNewScores);
|
|
1793
|
+
const oldRanks = toRankPositions(existingAllScores);
|
|
1794
|
+
const newRanks = toRankPositions(mergedNewScores);
|
|
1606
1795
|
const report = buildRankChangeReport(oldRanks, newRanks);
|
|
1607
1796
|
if (report.forgettingCandidates.length > 0) {
|
|
1608
1797
|
warn(`[improve/salience] WS-1 rank-change report: ${report.forgettingCandidates.length} asset(s) fell from top-200 to below position 500. ` +
|
|
@@ -1613,7 +1802,11 @@ export async function runImprovePreparationStage(args) {
|
|
|
1613
1802
|
// Collect refs for protective consolidation pass (plan §WS-1 step 7).
|
|
1614
1803
|
// These are force-included in the candidate pool (mergedRefs) after
|
|
1615
1804
|
// this try block, bypassing cooldown/signal-delta gating.
|
|
1616
|
-
|
|
1805
|
+
// Chunk-5 flip F5e — map an in-pool candidate's write-key spelling
|
|
1806
|
+
// back to its bare `r.ref` so applyForgettingSafety re-stamps the
|
|
1807
|
+
// existing pool ref; a genuinely stash-wide (out-of-pool) candidate
|
|
1808
|
+
// keeps its stored spelling and is synthesised as a fresh stub.
|
|
1809
|
+
pendingForgettingRefs = report.forgettingCandidates.map((e) => refByWriteKey.get(e.ref) ?? e.ref);
|
|
1617
1810
|
}
|
|
1618
1811
|
appendEvent({
|
|
1619
1812
|
eventType: "improve_salience_rank_change",
|
|
@@ -1631,7 +1824,8 @@ export async function runImprovePreparationStage(args) {
|
|
|
1631
1824
|
}, eventsCtx);
|
|
1632
1825
|
}
|
|
1633
1826
|
for (const [ref, vector] of salienceMap) {
|
|
1634
|
-
|
|
1827
|
+
// Persist salience under item_ref when resolved, else the conceptId.
|
|
1828
|
+
upsertAssetSalience(stateDb, wk(ref), vector, nowForSalience);
|
|
1635
1829
|
}
|
|
1636
1830
|
}, { path: eventsCtx?.dbPath });
|
|
1637
1831
|
}
|
|
@@ -1639,6 +1833,16 @@ export async function runImprovePreparationStage(args) {
|
|
|
1639
1833
|
rethrowIfTestIsolationError(err);
|
|
1640
1834
|
// best-effort: salience persistence failure never blocks ranking
|
|
1641
1835
|
}
|
|
1836
|
+
return pendingForgettingRefs;
|
|
1837
|
+
}
|
|
1838
|
+
/**
|
|
1839
|
+
* The protective forgetting-safety injection (plan §WS-1 step 7). Returns the
|
|
1840
|
+
* (possibly extended) mergedRefs; re-stamps lane attribution in precedence
|
|
1841
|
+
* order on the SHARED ref objects and eligibilitySourceByRef map.
|
|
1842
|
+
*/
|
|
1843
|
+
export function applyForgettingSafety(args) {
|
|
1844
|
+
const { pendingForgettingRefs, scope, eligibilitySourceByRef, highSalienceRefs, proactiveRefs, signalFiltered } = args;
|
|
1845
|
+
let mergedRefs = args.mergedRefs;
|
|
1642
1846
|
// ── Protective consolidation pass (plan §WS-1 step 7) ─────────────────────
|
|
1643
1847
|
// Forgetting candidates detected in scenario B are force-injected into
|
|
1644
1848
|
// mergedRefs here, BEFORE the effectiveScore sort, bypassing cooldown and
|
|
@@ -1657,8 +1861,8 @@ export async function runImprovePreparationStage(args) {
|
|
|
1657
1861
|
newForgettingRefs.push({ ref, reason: "scope-type", eligibilitySource: "forgetting-safety" });
|
|
1658
1862
|
}
|
|
1659
1863
|
// Always stamp the lane in the attribution map (overwrites weaker lanes;
|
|
1660
|
-
// stronger reactive signals — scope/signal-delta/
|
|
1661
|
-
//
|
|
1864
|
+
// stronger reactive signals — scope/signal-delta/proactive — are written
|
|
1865
|
+
// after this block so they take precedence).
|
|
1662
1866
|
eligibilitySourceByRef.set(ref, "forgetting-safety");
|
|
1663
1867
|
}
|
|
1664
1868
|
if (newForgettingRefs.length > 0) {
|
|
@@ -1666,25 +1870,23 @@ export async function runImprovePreparationStage(args) {
|
|
|
1666
1870
|
}
|
|
1667
1871
|
// Re-stamp attribution for any refs whose lane needs updating.
|
|
1668
1872
|
// Precedence (weakest → strongest, each overwrites the previous):
|
|
1669
|
-
// proactive <
|
|
1873
|
+
// proactive < forgetting-safety < signal-delta
|
|
1670
1874
|
// Scope mode is already excluded by the outer guard (`scope.mode !== "ref"`).
|
|
1671
|
-
// forgetting-safety sits above proactive
|
|
1672
|
-
//
|
|
1673
|
-
//
|
|
1674
|
-
//
|
|
1675
|
-
//
|
|
1875
|
+
// forgetting-safety sits above proactive so that a ref flagged as a
|
|
1876
|
+
// forgetting candidate is always visible to S5/WS-5 as such, even when it
|
|
1877
|
+
// was also due for a proactive maintenance run. signal-delta overrides
|
|
1878
|
+
// forgetting-safety because a ref with fresh feedback is reactive and
|
|
1879
|
+
// doesn't need the protective pass label for measurement purposes.
|
|
1676
1880
|
for (const r of highSalienceRefs)
|
|
1677
1881
|
eligibilitySourceByRef.set(r.ref, "high-salience");
|
|
1678
1882
|
for (const r of proactiveRefs)
|
|
1679
1883
|
eligibilitySourceByRef.set(r.ref, "proactive");
|
|
1680
|
-
|
|
1681
|
-
|
|
1682
|
-
// Apply forgetting-safety OVER proactive, high-retrieval, and high-salience
|
|
1683
|
-
// (already stamped in the loop above via
|
|
1884
|
+
// Apply forgetting-safety OVER proactive and high-salience (already
|
|
1885
|
+
// stamped in the loop above via
|
|
1684
1886
|
// `eligibilitySourceByRef.set(ref, "forgetting-safety")`). No-op here: the
|
|
1685
|
-
// set() calls above for proactive/high-
|
|
1686
|
-
//
|
|
1687
|
-
//
|
|
1887
|
+
// set() calls above for proactive/high-salience overwrite the earlier
|
|
1888
|
+
// forgetting-safety stamp — so we re-apply forgetting-safety now for those
|
|
1889
|
+
// refs that are both forgetting candidates AND in another fallback lane.
|
|
1688
1890
|
for (const ref of pendingForgettingRefs) {
|
|
1689
1891
|
eligibilitySourceByRef.set(ref, "forgetting-safety");
|
|
1690
1892
|
}
|
|
@@ -1697,6 +1899,158 @@ export async function runImprovePreparationStage(args) {
|
|
|
1697
1899
|
r.eligibilitySource = eligibilitySourceByRef.get(r.ref) ?? "unknown";
|
|
1698
1900
|
}
|
|
1699
1901
|
}
|
|
1902
|
+
return mergedRefs;
|
|
1903
|
+
}
|
|
1904
|
+
/**
|
|
1905
|
+
* Pass: eligibility-filter — replay selection (#610), the no-op dampener sort,
|
|
1906
|
+
* coverage gaps, the disk-existence guard, the --limit slice, and the summary
|
|
1907
|
+
* info emits.
|
|
1908
|
+
*/
|
|
1909
|
+
async function filterEligibility(args) {
|
|
1910
|
+
const { scope, options, plannedRefs, eventsCtx, salienceMap, eligibilitySourceByRef, distillOnlyRefs } = args;
|
|
1911
|
+
const { fullySkippedCount, preCooldownCount, signalAndRetrievalRefs, signalFiltered } = args.summary;
|
|
1912
|
+
const validationFailureRefs = args.validationFailureRefs;
|
|
1913
|
+
const replay = applyReplaySelection({
|
|
1914
|
+
scope,
|
|
1915
|
+
options,
|
|
1916
|
+
plannedRefs,
|
|
1917
|
+
eventsCtx,
|
|
1918
|
+
mergedRefs: args.mergedRefs,
|
|
1919
|
+
salienceMap,
|
|
1920
|
+
eligibilitySourceByRef,
|
|
1921
|
+
});
|
|
1922
|
+
const mergedRefs = replay.mergedRefs;
|
|
1923
|
+
const { replayRefSet, replayBudget } = replay;
|
|
1924
|
+
// Build no-op map for consolidation-selection dampener (plan §WS-1 step 8).
|
|
1925
|
+
// Reads consecutive_no_ops from the SAME pinned db handle used elsewhere in
|
|
1926
|
+
// this function. The effective score is used ONLY for processing/selection
|
|
1927
|
+
// order — the persisted rank_score in asset_salience is never mutated here.
|
|
1928
|
+
const noOpMap = new Map();
|
|
1929
|
+
try {
|
|
1930
|
+
const noOpDb = eventsCtx?.db ?? (eventsCtx?.dbPath ? openStateDatabase(eventsCtx.dbPath) : null);
|
|
1931
|
+
if (noOpDb) {
|
|
1932
|
+
const ownsNoOpDb = !eventsCtx?.db;
|
|
1933
|
+
try {
|
|
1934
|
+
for (const r of mergedRefs) {
|
|
1935
|
+
noOpMap.set(r.ref, readConsecutiveNoOpsForImproveRef(noOpDb, r.ref, r.itemRef));
|
|
1936
|
+
}
|
|
1937
|
+
}
|
|
1938
|
+
finally {
|
|
1939
|
+
if (ownsNoOpDb)
|
|
1940
|
+
noOpDb.close();
|
|
1941
|
+
}
|
|
1942
|
+
}
|
|
1943
|
+
}
|
|
1944
|
+
catch {
|
|
1945
|
+
// best-effort: dampener failure never blocks selection
|
|
1946
|
+
}
|
|
1947
|
+
// Sort by effective selection score (desc), with explicit ref-string tie-break
|
|
1948
|
+
// for determinism. The effective score applies the consolidation-selection
|
|
1949
|
+
// dampener: assets that have been repeatedly skipped (consecutive_no_ops >=
|
|
1950
|
+
// THRESHOLD) are penalised by FACTOR so they sort after peers with similar
|
|
1951
|
+
// rankScore. The persisted rank_score is left unchanged — this is the whole
|
|
1952
|
+
// point of the dampener (stable assets stay fully retrievable).
|
|
1953
|
+
//
|
|
1954
|
+
// WIRING NOTE (plan §WS-1 step 8 / "consolidation-selection" disambiguation):
|
|
1955
|
+
// "consolidation-selection" in the plan refers to THIS reflect/distill
|
|
1956
|
+
// eligibility ordering — i.e. which assets are chosen for the reflect/distill
|
|
1957
|
+
// LLM pass — NOT to akmConsolidate (the cluster-merge phase at ~line 1994,
|
|
1958
|
+
// which runs earlier and never reads noOpMap). The no-op counter originates
|
|
1959
|
+
// from no-change reflect / quality-rejected distill outcomes; the dampener
|
|
1960
|
+
// suppresses repeated LLM attempts on those same assets without touching their
|
|
1961
|
+
// persisted rank_score (so they remain fully retrievable).
|
|
1962
|
+
//
|
|
1963
|
+
// This is the only ranking path. The eligibilitySource lanes (signal-delta /
|
|
1964
|
+
// proactive / high-salience) survive as labels set above.
|
|
1965
|
+
const effectiveScore = (ref) => {
|
|
1966
|
+
const rankScore = salienceMap.get(ref)?.rankScore ?? 0;
|
|
1967
|
+
const noOps = noOpMap.get(ref) ?? 0;
|
|
1968
|
+
return noOps >= SALIENCE_NO_OP_DAMPEN_THRESHOLD ? rankScore * SALIENCE_NO_OP_DAMPEN_FACTOR : rankScore;
|
|
1969
|
+
};
|
|
1970
|
+
const sorted = [...mergedRefs].sort((a, b) => {
|
|
1971
|
+
const scoreA = effectiveScore(a.ref);
|
|
1972
|
+
const scoreB = effectiveScore(b.ref);
|
|
1973
|
+
if (scoreB !== scoreA)
|
|
1974
|
+
return scoreB - scoreA;
|
|
1975
|
+
// Stable tie-break: deterministic regardless of input ordering.
|
|
1976
|
+
return a.ref < b.ref ? -1 : a.ref > b.ref ? 1 : 0;
|
|
1977
|
+
});
|
|
1978
|
+
// Phase 0: surface coverage gaps from zero-result search queries
|
|
1979
|
+
let coverageGaps = [];
|
|
1980
|
+
try {
|
|
1981
|
+
const dbForGaps = openExistingDatabase();
|
|
1982
|
+
try {
|
|
1983
|
+
coverageGaps = getZeroResultSearches(dbForGaps);
|
|
1984
|
+
}
|
|
1985
|
+
finally {
|
|
1986
|
+
closeDatabase(dbForGaps);
|
|
1987
|
+
}
|
|
1988
|
+
}
|
|
1989
|
+
catch (err) {
|
|
1990
|
+
rethrowIfTestIsolationError(err);
|
|
1991
|
+
// best-effort
|
|
1992
|
+
}
|
|
1993
|
+
const diskCheck = await dropRefsMissingOnDisk({ sorted, options, eventsCtx });
|
|
1994
|
+
const assetMissingOnDisk = diskCheck.assetMissingOnDisk;
|
|
1995
|
+
const actionableRefs = diskCheck.actionableRefs;
|
|
1996
|
+
// Re-split actionableRefs (sorted) into reflect-path vs distill-only-path while
|
|
1997
|
+
// preserving sort order. distillOnlyRefs participate in the sort so --limit
|
|
1998
|
+
// picks them by score, not by arbitrary position.
|
|
1999
|
+
const distillOnlyRefSetForSort = new Set(distillOnlyRefs.map((r) => r.ref));
|
|
2000
|
+
const reflectAndDistillRefsAfterSort = [];
|
|
2001
|
+
const distillOnlyRefsAfterSort = [];
|
|
2002
|
+
for (const r of actionableRefs) {
|
|
2003
|
+
if (distillOnlyRefSetForSort.has(r.ref)) {
|
|
2004
|
+
distillOnlyRefsAfterSort.push(r);
|
|
2005
|
+
}
|
|
2006
|
+
else {
|
|
2007
|
+
reflectAndDistillRefsAfterSort.push(r);
|
|
2008
|
+
}
|
|
2009
|
+
}
|
|
2010
|
+
// ── Phase 5: --limit applies to the post-cooldown actionable set ──────────
|
|
2011
|
+
//
|
|
2012
|
+
// #610 ADDITIVITY: replay-lane refs are budgeted SEPARATELY from the --limit
|
|
2013
|
+
// fresh slice. Without this split, a high-rankScore replay ref could sort above
|
|
2014
|
+
// a fresh ref in the single combined slice and STEAL its slot (violating AC2).
|
|
2015
|
+
// We partition into the replay lane vs the rest, apply --limit to the
|
|
2016
|
+
// non-replay (fresh) refs only, then APPEND up to `replayBudget` replay refs
|
|
2017
|
+
// after the fresh slice. Sort order within each partition is preserved.
|
|
2018
|
+
//
|
|
2019
|
+
// Default replayBudget=0 reduces this to the exact pre-#610 expression: with no
|
|
2020
|
+
// replay refs, `nonReplayLoop === allLoopRefs`, so `baseLoop === old slice` and
|
|
2021
|
+
// `replayLoop.slice(0, 0) === []` — byte-identical.
|
|
2022
|
+
const allLoopRefs = [...reflectAndDistillRefsAfterSort, ...distillOnlyRefsAfterSort];
|
|
2023
|
+
const replayLoop = allLoopRefs.filter((r) => r.eligibilitySource === "replay");
|
|
2024
|
+
const nonReplayLoop = allLoopRefs.filter((r) => r.eligibilitySource !== "replay");
|
|
2025
|
+
const baseLoop = options.limit ? nonReplayLoop.slice(0, options.limit) : nonReplayLoop;
|
|
2026
|
+
const loopRefs = [...baseLoop, ...replayLoop.slice(0, replayBudget)];
|
|
2027
|
+
// Update the returned distillOnlyRefs to the sorted order so callers see the
|
|
2028
|
+
// ranked view (loop stage uses it as a Set so order is irrelevant, but the
|
|
2029
|
+
// shape change keeps downstream consumers consistent).
|
|
2030
|
+
const distillOnlyRefsResult = distillOnlyRefsAfterSort;
|
|
2031
|
+
const totalReflectBlocked = fullySkippedCount + distillOnlyRefs.length;
|
|
2032
|
+
if (totalReflectBlocked > 0) {
|
|
2033
|
+
info(`[improve] ${totalReflectBlocked} of ${preCooldownCount} indexed refs blocked by reflect signal-delta ` +
|
|
2034
|
+
`(${fullySkippedCount} fully skipped, ${distillOnlyRefs.length} routed to distill-only)`);
|
|
2035
|
+
}
|
|
2036
|
+
if (signalAndRetrievalRefs.length > 0) {
|
|
2037
|
+
info(`[improve] ${signalAndRetrievalRefs.length} refs with usage signals (${signalFiltered.length} feedback${replayRefSet.size > 0 ? `, ${replayRefSet.size} replay` : ""})`);
|
|
2038
|
+
}
|
|
2039
|
+
if (validationFailureRefs.size > 0) {
|
|
2040
|
+
info(`[improve] ${validationFailureRefs.size} with validation failures excluded`);
|
|
2041
|
+
}
|
|
2042
|
+
if (assetMissingOnDisk.length > 0) {
|
|
2043
|
+
info(`[improve] ${assetMissingOnDisk.length} candidates dropped — file not on disk`);
|
|
2044
|
+
}
|
|
2045
|
+
const deferredCount = actionableRefs.length - loopRefs.length;
|
|
2046
|
+
info(`[improve] ${actionableRefs.length} actionable; ${loopRefs.length} will be processed` +
|
|
2047
|
+
(options.limit && deferredCount > 0 ? ` (--limit ${options.limit} applied; ${deferredCount} deferred)` : ""));
|
|
2048
|
+
return { loopRefs, actionableRefs, distillOnlyRefs: distillOnlyRefsResult, coverageGaps };
|
|
2049
|
+
}
|
|
2050
|
+
/** The #610 bounded, additive replay-selection lane (eligibility-filter). */
|
|
2051
|
+
function applyReplaySelection(args) {
|
|
2052
|
+
const { scope, options, plannedRefs, eventsCtx, salienceMap, eligibilitySourceByRef } = args;
|
|
2053
|
+
let mergedRefs = args.mergedRefs;
|
|
1700
2054
|
// ── REPLAY SELECTION layer (#610) ─────────────────────────────────────────
|
|
1701
2055
|
// Bounded, ADDITIVE replay budget: up to `replayBudget` top-salience refs are
|
|
1702
2056
|
// revisited even with zero reactive signal (no feedback, no retrieval) and
|
|
@@ -1717,7 +2071,31 @@ export async function runImprovePreparationStage(args) {
|
|
|
1717
2071
|
try {
|
|
1718
2072
|
withStateDb((replayDb) => {
|
|
1719
2073
|
const alreadyInPool = new Set(mergedRefs.map((r) => r.ref));
|
|
1720
|
-
const
|
|
2074
|
+
const storedRankScores = getAllRankScores(replayDb);
|
|
2075
|
+
// Chunk-5 flip F5e — plan reverse map so a stored item_ref row re-keys
|
|
2076
|
+
// back onto its filesystem-facing bare ref (the replay stub must match
|
|
2077
|
+
// a planned entry for its filePath / disk check).
|
|
2078
|
+
const itemRefByPlanned = buildItemRefByRef(plannedRefs);
|
|
2079
|
+
const bareRefByItemRef = new Map();
|
|
2080
|
+
for (const [bareRef, itemRef] of itemRefByPlanned)
|
|
2081
|
+
if (itemRef)
|
|
2082
|
+
bareRefByItemRef.set(itemRef, bareRef);
|
|
2083
|
+
const allRankScores = options.sourceName
|
|
2084
|
+
? (() => {
|
|
2085
|
+
// Source-scope by the bundle prefix, then re-key item_ref rows
|
|
2086
|
+
// onto the short conceptId used by the planned-ref pool.
|
|
2087
|
+
const m = new Map();
|
|
2088
|
+
for (const [ref, score] of storedRankScores) {
|
|
2089
|
+
const boundary = ref.indexOf("//");
|
|
2090
|
+
const prefix = boundary >= 0 ? ref.slice(0, boundary) : undefined;
|
|
2091
|
+
const belongs = prefix === options.sourceName;
|
|
2092
|
+
if (!belongs)
|
|
2093
|
+
continue;
|
|
2094
|
+
m.set(bareRefByItemRef.get(ref) ?? bareImproveRef(ref), score);
|
|
2095
|
+
}
|
|
2096
|
+
return m;
|
|
2097
|
+
})()
|
|
2098
|
+
: storedRankScores;
|
|
1721
2099
|
// Candidate universe = every salience row NOT already in the pool, ordered by
|
|
1722
2100
|
// rank_score desc with a deterministic ref-string tie-break (mirrors the main
|
|
1723
2101
|
// sort). Converged refs (consecutive_no_ops >= dampener threshold) are fully
|
|
@@ -1727,7 +2105,7 @@ export async function runImprovePreparationStage(args) {
|
|
|
1727
2105
|
for (const [ref, rankScore] of allRankScores) {
|
|
1728
2106
|
if (alreadyInPool.has(ref))
|
|
1729
2107
|
continue;
|
|
1730
|
-
const noOps =
|
|
2108
|
+
const noOps = readConsecutiveNoOpsForImproveRef(replayDb, ref, itemRefByPlanned.get(ref));
|
|
1731
2109
|
if (noOps >= SALIENCE_NO_OP_DAMPEN_THRESHOLD) {
|
|
1732
2110
|
convergedSkipped++;
|
|
1733
2111
|
continue;
|
|
@@ -1793,76 +2171,11 @@ export async function runImprovePreparationStage(args) {
|
|
|
1793
2171
|
// best-effort: if DB unavailable, replayRefSet stays empty
|
|
1794
2172
|
}
|
|
1795
2173
|
}
|
|
1796
|
-
|
|
1797
|
-
|
|
1798
|
-
|
|
1799
|
-
|
|
1800
|
-
const
|
|
1801
|
-
try {
|
|
1802
|
-
const noOpDb = eventsCtx?.db ?? (eventsCtx?.dbPath ? openStateDatabase(eventsCtx.dbPath) : null);
|
|
1803
|
-
if (noOpDb) {
|
|
1804
|
-
const ownsNoOpDb = !eventsCtx?.db;
|
|
1805
|
-
try {
|
|
1806
|
-
for (const r of mergedRefs) {
|
|
1807
|
-
noOpMap.set(r.ref, getConsecutiveNoOps(noOpDb, r.ref));
|
|
1808
|
-
}
|
|
1809
|
-
}
|
|
1810
|
-
finally {
|
|
1811
|
-
if (ownsNoOpDb)
|
|
1812
|
-
noOpDb.close();
|
|
1813
|
-
}
|
|
1814
|
-
}
|
|
1815
|
-
}
|
|
1816
|
-
catch {
|
|
1817
|
-
// best-effort: dampener failure never blocks selection
|
|
1818
|
-
}
|
|
1819
|
-
// Sort by effective selection score (desc), with explicit ref-string tie-break
|
|
1820
|
-
// for determinism. The effective score applies the consolidation-selection
|
|
1821
|
-
// dampener: assets that have been repeatedly skipped (consecutive_no_ops >=
|
|
1822
|
-
// THRESHOLD) are penalised by FACTOR so they sort after peers with similar
|
|
1823
|
-
// rankScore. The persisted rank_score is left unchanged — this is the whole
|
|
1824
|
-
// point of the dampener (stable assets stay fully retrievable).
|
|
1825
|
-
//
|
|
1826
|
-
// WIRING NOTE (plan §WS-1 step 8 / "consolidation-selection" disambiguation):
|
|
1827
|
-
// "consolidation-selection" in the plan refers to THIS reflect/distill
|
|
1828
|
-
// eligibility ordering — i.e. which assets are chosen for the reflect/distill
|
|
1829
|
-
// LLM pass — NOT to akmConsolidate (the cluster-merge phase at ~line 1994,
|
|
1830
|
-
// which runs earlier and never reads noOpMap). The no-op counter originates
|
|
1831
|
-
// from no-change reflect / quality-rejected distill outcomes; the dampener
|
|
1832
|
-
// suppresses repeated LLM attempts on those same assets without touching their
|
|
1833
|
-
// persisted rank_score (so they remain fully retrievable).
|
|
1834
|
-
//
|
|
1835
|
-
// This is the ONLY ranking path — negativeOnlyRatio and the legacy
|
|
1836
|
-
// symmetricValence branch are replaced. The three eligibilitySource lanes
|
|
1837
|
-
// (signal-delta / high-retrieval / proactive) survive as labels (set above).
|
|
1838
|
-
const effectiveScore = (ref) => {
|
|
1839
|
-
const rankScore = salienceMap.get(ref)?.rankScore ?? 0;
|
|
1840
|
-
const noOps = noOpMap.get(ref) ?? 0;
|
|
1841
|
-
return noOps >= SALIENCE_NO_OP_DAMPEN_THRESHOLD ? rankScore * SALIENCE_NO_OP_DAMPEN_FACTOR : rankScore;
|
|
1842
|
-
};
|
|
1843
|
-
const sorted = [...mergedRefs].sort((a, b) => {
|
|
1844
|
-
const scoreA = effectiveScore(a.ref);
|
|
1845
|
-
const scoreB = effectiveScore(b.ref);
|
|
1846
|
-
if (scoreB !== scoreA)
|
|
1847
|
-
return scoreB - scoreA;
|
|
1848
|
-
// Stable tie-break: deterministic regardless of input ordering.
|
|
1849
|
-
return a.ref < b.ref ? -1 : a.ref > b.ref ? 1 : 0;
|
|
1850
|
-
});
|
|
1851
|
-
// Phase 0: surface coverage gaps from zero-result search queries
|
|
1852
|
-
let coverageGaps = [];
|
|
1853
|
-
try {
|
|
1854
|
-
const dbForGaps = openExistingDatabase();
|
|
1855
|
-
try {
|
|
1856
|
-
coverageGaps = getZeroResultSearches(dbForGaps);
|
|
1857
|
-
}
|
|
1858
|
-
finally {
|
|
1859
|
-
closeDatabase(dbForGaps);
|
|
1860
|
-
}
|
|
1861
|
-
}
|
|
1862
|
-
catch (err) {
|
|
1863
|
-
rethrowIfTestIsolationError(err);
|
|
1864
|
-
// best-effort
|
|
1865
|
-
}
|
|
2174
|
+
return { mergedRefs, replayRefSet, replayBudget };
|
|
2175
|
+
}
|
|
2176
|
+
/** The final disk-existence guard + its aggregated audit event (eligibility-filter). */
|
|
2177
|
+
async function dropRefsMissingOnDisk(args) {
|
|
2178
|
+
const { sorted, options, eventsCtx } = args;
|
|
1866
2179
|
// actionableRefs is the post-cooldown, post-validation, post-signal, post-sort
|
|
1867
2180
|
// set — i.e. the genuinely processable refs in priority order. Note: this is
|
|
1868
2181
|
// a semantic shift from earlier code where actionableRefs was the pre-cooldown
|
|
@@ -1906,97 +2219,5 @@ export async function runImprovePreparationStage(args) {
|
|
|
1906
2219
|
},
|
|
1907
2220
|
}, eventsCtx);
|
|
1908
2221
|
}
|
|
1909
|
-
|
|
1910
|
-
// Re-split actionableRefs (sorted) into reflect-path vs distill-only-path while
|
|
1911
|
-
// preserving sort order. distillOnlyRefs participate in the sort so --limit
|
|
1912
|
-
// picks them by score, not by arbitrary position.
|
|
1913
|
-
const distillOnlyRefSetForSort = new Set(distillOnlyRefs.map((r) => r.ref));
|
|
1914
|
-
const reflectAndDistillRefsAfterSort = [];
|
|
1915
|
-
const distillOnlyRefsAfterSort = [];
|
|
1916
|
-
for (const r of actionableRefs) {
|
|
1917
|
-
if (distillOnlyRefSetForSort.has(r.ref)) {
|
|
1918
|
-
distillOnlyRefsAfterSort.push(r);
|
|
1919
|
-
}
|
|
1920
|
-
else {
|
|
1921
|
-
reflectAndDistillRefsAfterSort.push(r);
|
|
1922
|
-
}
|
|
1923
|
-
}
|
|
1924
|
-
// ── Phase 5: --limit applies to the post-cooldown actionable set ──────────
|
|
1925
|
-
//
|
|
1926
|
-
// #610 ADDITIVITY: replay-lane refs are budgeted SEPARATELY from the --limit
|
|
1927
|
-
// fresh slice. Without this split, a high-rankScore replay ref could sort above
|
|
1928
|
-
// a fresh ref in the single combined slice and STEAL its slot (violating AC2).
|
|
1929
|
-
// We partition into the replay lane vs the rest, apply --limit to the
|
|
1930
|
-
// non-replay (fresh) refs only, then APPEND up to `replayBudget` replay refs
|
|
1931
|
-
// after the fresh slice. Sort order within each partition is preserved.
|
|
1932
|
-
//
|
|
1933
|
-
// Default replayBudget=0 reduces this to the exact pre-#610 expression: with no
|
|
1934
|
-
// replay refs, `nonReplayLoop === allLoopRefs`, so `baseLoop === old slice` and
|
|
1935
|
-
// `replayLoop.slice(0, 0) === []` — byte-identical.
|
|
1936
|
-
const allLoopRefs = [...reflectAndDistillRefsAfterSort, ...distillOnlyRefsAfterSort];
|
|
1937
|
-
const replayLoop = allLoopRefs.filter((r) => r.eligibilitySource === "replay");
|
|
1938
|
-
const nonReplayLoop = allLoopRefs.filter((r) => r.eligibilitySource !== "replay");
|
|
1939
|
-
const baseLoop = options.limit ? nonReplayLoop.slice(0, options.limit) : nonReplayLoop;
|
|
1940
|
-
const loopRefs = [...baseLoop, ...replayLoop.slice(0, replayBudget)];
|
|
1941
|
-
// Update the returned distillOnlyRefs to the sorted order so callers see the
|
|
1942
|
-
// ranked view (loop stage uses it as a Set so order is irrelevant, but the
|
|
1943
|
-
// shape change keeps downstream consumers consistent).
|
|
1944
|
-
const distillOnlyRefsResult = distillOnlyRefsAfterSort;
|
|
1945
|
-
const totalReflectBlocked = fullySkippedCount + distillOnlyRefs.length;
|
|
1946
|
-
if (totalReflectBlocked > 0) {
|
|
1947
|
-
info(`[improve] ${totalReflectBlocked} of ${preCooldownCount} indexed refs blocked by reflect signal-delta ` +
|
|
1948
|
-
`(${fullySkippedCount} fully skipped, ${distillOnlyRefs.length} routed to distill-only)`);
|
|
1949
|
-
}
|
|
1950
|
-
if (signalAndRetrievalRefs.length > 0) {
|
|
1951
|
-
info(`[improve] ${signalAndRetrievalRefs.length} refs with usage signals (${signalFiltered.length} feedback, ${highRetrievalRefs.length} high-retrieval${replayRefSet.size > 0 ? `, ${replayRefSet.size} replay` : ""})`);
|
|
1952
|
-
}
|
|
1953
|
-
if (validationFailureRefs.size > 0) {
|
|
1954
|
-
info(`[improve] ${validationFailureRefs.size} with validation failures excluded`);
|
|
1955
|
-
}
|
|
1956
|
-
if (assetMissingOnDisk.length > 0) {
|
|
1957
|
-
info(`[improve] ${assetMissingOnDisk.length} candidates dropped — file not on disk`);
|
|
1958
|
-
}
|
|
1959
|
-
const deferredCount = actionableRefs.length - loopRefs.length;
|
|
1960
|
-
info(`[improve] ${actionableRefs.length} actionable; ${loopRefs.length} will be processed` +
|
|
1961
|
-
(options.limit && deferredCount > 0 ? ` (--limit ${options.limit} applied; ${deferredCount} deferred)` : ""));
|
|
1962
|
-
// WS-4: Per-phase threshold auto-tune for the extract phase.
|
|
1963
|
-
// Persists result for the NEXT run's makeGateConfig to read.
|
|
1964
|
-
const extractTuneDbPath = eventsCtx?.dbPath;
|
|
1965
|
-
if (options.autoAccept !== undefined && extractTuneDbPath) {
|
|
1966
|
-
try {
|
|
1967
|
-
maybeAutoTuneThreshold(extractPass.extractGateCfg.phaseThreshold ?? options.autoAccept, options.config ?? loadConfig(), extractTuneDbPath, undefined, "extract");
|
|
1968
|
-
}
|
|
1969
|
-
catch (err) {
|
|
1970
|
-
warn(`[improve] calibration auto-tune (extract) skipped: ${err instanceof Error ? err.message : String(err)}`);
|
|
1971
|
-
}
|
|
1972
|
-
}
|
|
1973
|
-
return {
|
|
1974
|
-
actions,
|
|
1975
|
-
cleanupWarnings,
|
|
1976
|
-
appliedCleanup,
|
|
1977
|
-
memoryIndexHealth,
|
|
1978
|
-
extract: extractResults,
|
|
1979
|
-
actionableRefs,
|
|
1980
|
-
signalBearingSet,
|
|
1981
|
-
validationFailures,
|
|
1982
|
-
schemaRepairs,
|
|
1983
|
-
lintSummary,
|
|
1984
|
-
loopRefs,
|
|
1985
|
-
distillCooledRefs,
|
|
1986
|
-
distillOnlyRefs: distillOnlyRefsResult,
|
|
1987
|
-
coverageGaps,
|
|
1988
|
-
recentErrors,
|
|
1989
|
-
utilityMap,
|
|
1990
|
-
gateAutoAcceptedCount,
|
|
1991
|
-
gateAutoAcceptFailedCount,
|
|
1992
|
-
consolidation: consolidationPass.consolidation,
|
|
1993
|
-
consolidationRan: consolidationPass.consolidationRan,
|
|
1994
|
-
...(proactiveMaintenanceSummary ? { proactiveMaintenance: proactiveMaintenanceSummary } : {}),
|
|
1995
|
-
};
|
|
2222
|
+
return { actionableRefs: existsCheckedActionable, assetMissingOnDisk };
|
|
1996
2223
|
}
|
|
1997
|
-
// TODO(refactor): 13 args including `actions`/`recentErrors` mutation channels. Restructure into immutable plan + mutable context objects — deferred to dedicated refactor with isolated testing.
|
|
1998
|
-
/**
|
|
1999
|
-
* Parameter object for {@link runImproveLoopStage} (WS10). Pure type reshape of
|
|
2000
|
-
* the former inline arg struct — every field, name, and type is preserved so the
|
|
2001
|
-
* function body and all runtime values are byte-identical. No control-flow change.
|
|
2002
|
-
*/
|