akm-cli 0.9.17-alpha.2 → 0.9.17-alpha.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +756 -0
- package/dist/akm +94 -196
- package/dist/cli/shared.js +6 -2
- package/dist/cli.js +22 -9
- package/dist/commands/agent/agent-dispatch.js +1 -1
- package/dist/commands/command/command-execution.js +24 -62
- package/dist/commands/feedback-cli.js +0 -1
- package/dist/commands/health/accept-rate.js +2 -2
- package/dist/commands/health/checks.js +30 -75
- package/dist/commands/health/config-skew.js +38 -0
- package/dist/commands/health/egress.js +54 -0
- package/dist/commands/health/html-report.js +0 -38
- package/dist/commands/health/improve-metrics.js +123 -562
- package/dist/commands/health/plugin-staleness.js +53 -3
- package/dist/commands/health/renderers.js +12 -4
- package/dist/commands/health/report-view-model.js +11 -106
- package/dist/commands/health/types-improve.js +4 -19
- package/dist/commands/health/windows.js +64 -73
- package/dist/commands/health.js +122 -143
- package/dist/commands/improve/consolidate/chunking.js +25 -100
- package/dist/commands/improve/consolidate/sanitize.js +54 -149
- package/dist/commands/improve/consolidate.js +538 -1075
- package/dist/commands/improve/content-hash.js +16 -24
- package/dist/commands/improve/distill/content-repair.js +18 -100
- package/dist/commands/improve/distill-guards.js +20 -81
- package/dist/commands/improve/distill-promotion-policy.js +23 -243
- package/dist/commands/improve/distill.js +608 -1075
- package/dist/commands/improve/eligibility.js +126 -400
- package/dist/commands/improve/execution.js +3 -5
- package/dist/commands/improve/extract.js +487 -1046
- package/dist/commands/improve/feedback-valence.js +0 -25
- package/dist/commands/improve/improve-cli.js +29 -166
- package/dist/commands/improve/improve-result-file.js +10 -66
- package/dist/commands/improve/improve-strategies.js +12 -7
- package/dist/commands/improve/improve-usage-report.js +18 -64
- package/dist/commands/improve/improve.js +443 -1063
- package/dist/commands/improve/ledger.js +114 -0
- package/dist/commands/improve/locks.js +2 -8
- package/dist/commands/improve/loop-stages.js +459 -1172
- package/dist/commands/improve/memory/derived-ref.js +12 -77
- package/dist/commands/improve/memory/memory-belief.js +14 -118
- package/dist/commands/improve/memory/memory-improve.js +4 -3
- package/dist/commands/improve/outcome-loop.js +28 -156
- package/dist/commands/improve/planner.js +5 -10
- package/dist/commands/improve/preparation.js +851 -2339
- package/dist/commands/improve/proactive-maintenance.js +34 -101
- package/dist/commands/improve/reflect-noise.js +104 -280
- package/dist/commands/improve/reflect.js +621 -1367
- package/dist/commands/improve/salience.js +46 -232
- package/dist/commands/improve/session-asset.js +19 -100
- package/dist/commands/improve/stage.js +323 -0
- package/dist/commands/proposal/drain.js +251 -644
- package/dist/commands/proposal/proposal-cli.js +3 -18
- package/dist/commands/proposal/proposal-types.js +20 -41
- package/dist/commands/proposal/proposal.js +1 -2
- package/dist/commands/proposal/propose.js +134 -160
- package/dist/commands/proposal/repository.js +502 -1487
- package/dist/commands/proposal/validators/proposal-quality-validators.js +71 -174
- package/dist/commands/proposal/validators/proposal-validators.js +1 -1
- package/dist/commands/proposal/validators/proposals.js +13 -89
- package/dist/commands/read/curate.js +63 -413
- package/dist/commands/read/search-cli.js +16 -33
- package/dist/commands/read/search.js +17 -23
- package/dist/commands/read/show.js +2 -13
- package/dist/commands/sources/bundle-cli.js +25 -2
- package/dist/commands/sources/bundle-config-ops.js +7 -0
- package/dist/commands/sources/dangerous-env-audit.js +1 -2
- package/dist/commands/sources/info.js +2 -11
- package/dist/commands/sources/installed-stashes.js +197 -746
- package/dist/commands/sources/schema-repair.js +98 -129
- package/dist/commands/sources/source-add.js +62 -12
- package/dist/commands/sources/stash-cli.js +1 -1
- package/dist/commands/tasks/explain.js +10 -13
- package/dist/commands/tasks/tasks-cli.js +9 -8
- package/dist/commands/tasks/tasks.js +326 -930
- package/dist/commands/tasks/validate.js +42 -21
- package/dist/commands/workflow/plan.js +22 -29
- package/dist/commands/workflow-cli.js +4 -4
- package/dist/core/adapter/adapters/akm-adapter.js +0 -1
- package/dist/core/adapter/adapters/akm-lint.js +2 -3
- package/dist/core/adapter/adapters/akm-metadata.js +11 -12
- package/dist/core/adapter/adapters/akm-workflow-adapter.js +1 -1
- package/dist/core/adapter/execution-source.js +17 -29
- package/dist/core/asset/resolve-ref.js +1 -1
- package/dist/core/bundle-id.js +42 -5
- package/dist/core/bundle-rename.js +291 -0
- package/dist/core/config/config-io.js +1 -2
- package/dist/core/config/config-schema.js +1 -33
- package/dist/core/config/config-walker.js +1 -1
- package/dist/core/config/config.js +163 -68
- package/dist/core/config/legacy-source-shape-shim.js +38 -9
- package/dist/core/config/schema/embedding.js +20 -5
- package/dist/core/config/schema/engines.js +5 -0
- package/dist/core/config/schema/execution.js +1 -1
- package/dist/core/config/schema/experimental.js +1 -1
- package/dist/core/config/schema/improve-processes.js +21 -95
- package/dist/core/config/schema/improve.js +4 -42
- package/dist/core/config/schema/scheduler.js +12 -12
- package/dist/core/config/schema/search.js +6 -22
- package/dist/core/env-secret-ref.js +0 -1
- package/dist/core/errors.js +8 -9
- package/dist/core/file-lock.js +76 -173
- package/dist/core/logs-db.js +2 -2
- package/dist/core/paths.js +0 -27
- package/dist/core/redaction.js +109 -2
- package/dist/core/run-lock.js +2 -5
- package/dist/core/spawn-env.js +1 -1
- package/dist/core/state/migrations.js +108 -61
- package/dist/core/state-db-scope.js +2 -4
- package/dist/core/state-db.js +126 -692
- package/dist/core/type-presentation.js +1 -9
- package/dist/core/write-source.js +293 -1012
- package/dist/execution/input-contract.js +1 -1
- package/dist/execution/resolved-request.js +135 -689
- package/dist/execution/source.js +63 -257
- package/dist/execution/target-ref.js +1 -1
- package/dist/indexer/bundle-identity-guard.js +2 -2
- package/dist/indexer/db/graph-db.js +106 -46
- package/dist/indexer/ensure-index.js +44 -85
- package/dist/indexer/graph/graph-extraction.js +340 -562
- package/dist/indexer/graph/graph-related.js +130 -0
- package/dist/indexer/index-rebuild-lock.js +3 -11
- package/dist/indexer/index-writer-lock.js +8 -17
- package/dist/indexer/index-written-assets.js +139 -151
- package/dist/indexer/indexer.js +524 -846
- package/dist/indexer/materialize-embeddings.js +60 -397
- package/dist/indexer/passes/memory-inference.js +81 -90
- package/dist/indexer/passes/metadata.js +132 -200
- package/dist/indexer/read-preflight.js +0 -7
- package/dist/indexer/scan/doc-to-entry.js +1 -3
- package/dist/indexer/scan/drain-dir.js +1 -1
- package/dist/indexer/search/db-search.js +181 -590
- package/dist/indexer/search/fts-query.js +30 -41
- package/dist/indexer/search/ranking.js +28 -154
- package/dist/indexer/search/search-attribution.js +12 -32
- package/dist/indexer/search/search-fields.js +11 -15
- package/dist/indexer/search/search-hit-enrichers.js +54 -85
- package/dist/indexer/search/search-source.js +1 -4
- package/dist/indexer/usage/usage-events.js +2 -7
- package/dist/integrations/agent/engine-fallback.js +23 -40
- package/dist/integrations/agent/engine-resolution.js +93 -183
- package/dist/integrations/agent/execution.js +507 -0
- package/dist/integrations/agent/model-map.js +28 -156
- package/dist/integrations/agent/request-lowering.js +66 -141
- package/dist/integrations/agent/runner-dispatch.js +143 -321
- package/dist/integrations/agent/runner.js +54 -14
- package/dist/integrations/lockfile.js +53 -101
- package/dist/llm/embedders/deterministic.js +2 -3
- package/dist/llm/embedders/profile.js +71 -0
- package/dist/llm/embedders/remote.js +10 -15
- package/dist/llm/graph-extract.js +3 -12
- package/dist/llm/index-passes.js +3 -5
- package/dist/llm/memory-infer.js +1 -2
- package/dist/llm/metadata-enhance.js +1 -2
- package/dist/llm/structured-call.js +5 -24
- package/dist/output/generic-render.js +23 -11
- package/dist/output/html-render.js +13 -10
- package/dist/output/render-registry.js +3 -32
- package/dist/output/shapes/helpers.js +2 -34
- package/dist/output/shapes/passthrough.js +1 -9
- package/dist/{indexer/search/ranking-types.js → output/text/bundle-rename.js} +4 -1
- package/dist/output/text/command-format.js +60 -23
- package/dist/output/text/helpers.js +1 -1
- package/dist/output/text/migrate.js +5 -14
- package/dist/output/text/proposal-format.js +1 -2
- package/dist/output/text/workflow-format.js +0 -32
- package/dist/output/text.js +2 -0
- package/dist/registry/factory.js +4 -19
- package/dist/registry/network.js +66 -220
- package/dist/registry/providers/index.js +0 -2
- package/dist/registry/providers/skills-sh.js +3 -14
- package/dist/registry/providers/static-index.js +24 -26
- package/dist/registry/resolve.js +55 -131
- package/dist/scripts/akm-migrate-node.js +43937 -93313
- package/dist/scripts/akm-migrate.js +43697 -93071
- package/dist/setup/registry-stash-loader.js +4 -13
- package/dist/setup/semantic-assets.js +3 -44
- package/dist/setup/setup.js +1 -1
- package/dist/setup/steps/tasks.js +25 -15
- package/dist/sources/provider-factory.js +17 -18
- package/dist/sources/providers/filesystem.js +2 -3
- package/dist/sources/providers/git-install.js +7 -1
- package/dist/sources/providers/git-provider.js +0 -3
- package/dist/sources/providers/git-stash.js +0 -17
- package/dist/sources/providers/npm.js +2 -4
- package/dist/sources/providers/provider-utils.js +5 -10
- package/dist/sources/providers/website.js +0 -2
- package/dist/sources/snapshot-fetchers/website-ingest.js +1 -1
- package/dist/sources/website-url.js +2 -2
- package/dist/storage/database.js +9 -35
- package/dist/storage/repositories/improve-ledger-repository.js +168 -0
- package/dist/storage/repositories/index-connection.js +34 -70
- package/dist/storage/repositories/index-entries-repository.js +69 -111
- package/dist/storage/repositories/index-entry-mapper.js +1 -2
- package/dist/storage/repositories/index-entry-schema.js +83 -269
- package/dist/storage/repositories/index-fts-repository.js +86 -256
- package/dist/storage/repositories/index-llm-cache-repository.js +17 -0
- package/dist/storage/repositories/index-meta-repository.js +6 -4
- package/dist/storage/repositories/index-schema.js +192 -220
- package/dist/storage/repositories/index-utility-repository.js +8 -29
- package/dist/storage/repositories/index-vec-repository.js +133 -414
- package/dist/storage/repositories/outcome-repository.js +2 -1
- package/dist/storage/repositories/proposals-repository.js +35 -0
- package/dist/storage/repositories/registry-index-cache-repository.js +100 -0
- package/dist/storage/repositories/task-history-repository.js +26 -4
- package/dist/storage/repositories/workflow-runs-repository.js +53 -244
- package/dist/storage/sqlite-migrations.js +136 -0
- package/dist/storage/sqlite-pragmas.js +11 -9
- package/dist/storage/sqlite-transaction.js +170 -0
- package/dist/storage/state-db-integrity.js +34 -27
- package/dist/tasks/activation-config.js +134 -62
- package/dist/tasks/backends/cron.js +129 -277
- package/dist/tasks/backends/exec-utils.js +2 -5
- package/dist/tasks/backends/launchd.js +125 -745
- package/dist/tasks/backends/schtasks.js +101 -620
- package/dist/tasks/prepare/prepare-support.js +5 -15
- package/dist/tasks/prepare/prepare.js +0 -2
- package/dist/tasks/resolve-akm-bin.js +20 -79
- package/dist/tasks/run/attempt-lifecycle.js +0 -1
- package/dist/tasks/scheduler-binding.js +18 -238
- package/dist/tasks/scheduler-invocation.js +52 -52
- package/dist/tasks/scheduler-lock.js +53 -0
- package/dist/tasks/scheduler-sync.js +363 -679
- package/dist/tasks/source/parse-task-source.js +160 -10
- package/dist/tasks/source/task-source-v3-frozen.js +3 -4
- package/dist/tasks/source/task-to-v4.js +2 -2
- package/dist/workflows/authoring/authoring.js +3 -12
- package/dist/workflows/compile.js +211 -0
- package/dist/workflows/concurrency-policy.js +13 -74
- package/dist/workflows/exec/child-invocation.js +3 -17
- package/dist/workflows/exec/child-workflow.js +32 -141
- package/dist/workflows/exec/dispatch-redaction.js +13 -53
- package/dist/workflows/exec/environment.js +98 -0
- package/dist/workflows/exec/exec-unit.js +33 -140
- package/dist/workflows/exec/frozen-judge.js +7 -59
- package/dist/workflows/exec/native-executor.js +82 -341
- package/dist/workflows/exec/param-secrets.js +29 -47
- package/dist/workflows/exec/run-workflow.js +154 -387
- package/dist/workflows/exec/scheduler.js +9 -36
- package/dist/workflows/exec/step-work.js +127 -430
- package/dist/workflows/exec/unit-dispatch.js +11 -63
- package/dist/workflows/exec/unit-writer.js +8 -52
- package/dist/workflows/exec/worktree.js +39 -273
- package/dist/workflows/freeze/child-output-references.js +4 -15
- package/dist/workflows/freeze/environment.js +99 -92
- package/dist/workflows/freeze/freeze.js +172 -0
- package/dist/workflows/freeze/step-values.js +19 -21
- package/dist/workflows/freeze/targets/child-workflow.js +23 -92
- package/dist/workflows/freeze/targets/command.js +10 -33
- package/dist/workflows/freeze/targets/script.js +5 -12
- package/dist/workflows/freeze/targets/shell.js +3 -6
- package/dist/workflows/freeze/targets/task.js +25 -80
- package/dist/workflows/freeze/task-bindings.js +20 -67
- package/dist/workflows/{source-ir/github-yaml.js → github-yaml.js} +88 -206
- package/dist/workflows/ir/params.js +6 -51
- package/dist/workflows/ir/plan-hash.js +2 -34
- package/dist/workflows/parser.js +140 -43
- package/dist/{commands/improve/consolidate/types.js → workflows/plan.js} +2 -1
- package/dist/workflows/renderer.js +36 -69
- package/dist/workflows/resource-limits.js +12 -120
- package/dist/workflows/runtime/agent-identity.js +8 -40
- package/dist/workflows/runtime/run-outputs.js +3 -6
- package/dist/workflows/runtime/run-plan.js +316 -0
- package/dist/workflows/runtime/runs.js +48 -200
- package/dist/workflows/runtime/workflow-asset-loader.js +24 -57
- package/dist/workflows/{source-ir/semantics.js → source-semantics.js} +16 -20
- package/dist/workflows/validate-summary.js +2 -7
- package/docs/integration/bundling-akm.md +49 -42
- package/docs/migration/README.md +1 -0
- package/docs/migration/release-notes/0.9.17.md +41 -0
- package/docs/migration/v0.9.1-to-v0.9.2.md +19 -7
- package/docs/reference/cli.md +182 -125
- package/docs/reference/configuration.md +49 -56
- package/docs/reference/data-and-telemetry.md +19 -20
- package/docs/reference/tasks.md +86 -38
- package/docs/reference/workflow-schema.md +14 -18
- package/docs/reference/workflows.md +6 -9
- package/package.json +1 -1
- package/schemas/akm-config.json +87 -406
- package/dist/commands/health/advisories.js +0 -150
- package/dist/commands/health/metrics.js +0 -329
- package/dist/commands/health/surfaces.js +0 -102
- package/dist/commands/improve/anti-collapse.js +0 -83
- package/dist/commands/improve/collapse-detector.js +0 -432
- package/dist/commands/improve/consolidate/eligibility.js +0 -48
- package/dist/commands/improve/consolidate/merge.js +0 -146
- package/dist/commands/improve/distill/promote-memory.js +0 -329
- package/dist/commands/improve/distill/quality-gate.js +0 -500
- package/dist/commands/improve/memory/memory-contradiction-detect.js +0 -291
- package/dist/commands/improve/proposal-envelope.js +0 -31
- package/dist/commands/improve/run-context.js +0 -123
- package/dist/commands/improve/shared.js +0 -21
- package/dist/commands/improve/source-identity.js +0 -28
- package/dist/commands/improve/triage.js +0 -96
- package/dist/commands/proposal/drain-policies.js +0 -151
- package/dist/commands/sources/update-transaction.js +0 -220
- package/dist/core/action-contributors.js +0 -28
- package/dist/core/config/config-version-shim.js +0 -101
- package/dist/core/config/retired-experimental-keys-shim.js +0 -62
- package/dist/core/fs-txn.js +0 -405
- package/dist/core/lexical-score.js +0 -25
- package/dist/core/maintenance-barrier.js +0 -167
- package/dist/execution/executable-identity.js +0 -105
- package/dist/execution/guarded-source.js +0 -427
- package/dist/indexer/graph/graph-boost.js +0 -427
- package/dist/indexer/graph/graph-dedup.js +0 -95
- package/dist/indexer/search/name-match.js +0 -35
- package/dist/indexer/search/ranking-contributors.js +0 -515
- package/dist/indexer/walk/project-context.js +0 -192
- package/dist/integrations/agent/execution-cascade.js +0 -566
- package/dist/integrations/agent/execution-definitions.js +0 -202
- package/dist/integrations/agent/execution-lowering.js +0 -841
- package/dist/integrations/agent/execution-preparation.js +0 -98
- package/dist/integrations/agent/inline-execution.js +0 -74
- package/dist/registry/create-provider-registry.js +0 -29
- package/dist/registry/pinned-request-helper.js +0 -247
- package/dist/registry/pinned-transport.js +0 -717
- package/dist/sources/providers/index.js +0 -14
- package/dist/storage/engines/sqlite-migrations.js +0 -271
- package/dist/storage/repositories/canaries-repository.js +0 -107
- package/dist/storage/repositories/embedding-salvage-repository.js +0 -184
- package/dist/storage/repositories/registry-cache.js +0 -113
- package/dist/tasks/scheduler-sync-preview.js +0 -52
- package/dist/workflows/freeze/resolve-steps.js +0 -86
- package/dist/workflows/freeze/source-freeze.js +0 -64
- package/dist/workflows/ir/compile.js +0 -321
- package/dist/workflows/ir/environment-v4.js +0 -330
- package/dist/workflows/ir/freeze-v4.js +0 -153
- package/dist/workflows/ir/schema-v4.js +0 -745
- package/dist/workflows/ir/schema.js +0 -354
- package/dist/workflows/program/schema.js +0 -78
- package/dist/workflows/runtime/checkin.js +0 -57
- package/dist/workflows/runtime/plan-classifier.js +0 -196
- package/dist/workflows/runtime/unit-checkin.js +0 -45
- package/dist/workflows/runtime/unit-phases.js +0 -20
- package/dist/workflows/schema.js +0 -4
- package/dist/workflows/source-ir/compile.js +0 -200
- package/dist/workflows/source-ir/program.js +0 -50
- package/dist/workflows/source-ir/result.js +0 -26
- package/dist/workflows/source-ir/schema.js +0 -786
- package/dist/workflows/source-ir/triggers.js +0 -79
- package/dist/workflows/source-ir/uses.js +0 -40
- package/dist/workflows/validator.js +0 -60
|
@@ -1,280 +1,148 @@
|
|
|
1
1
|
// This Source Code Form is subject to the terms of the Mozilla Public
|
|
2
2
|
// License, v. 2.0. If a copy of the MPL was not distributed with this
|
|
3
3
|
// file, You can obtain one at https://mozilla.org/MPL/2.0/.
|
|
4
|
+
/**
|
|
5
|
+
* The improve preparation stage: consolidation and session extraction (which
|
|
6
|
+
* run before the loop), memory cleanup, structural validation, and candidate
|
|
7
|
+
* selection for the reflect/distill loop.
|
|
8
|
+
*
|
|
9
|
+
* Candidate selection reads the improve ledger plus one set of signals: a ref
|
|
10
|
+
* is eligible for a source when feedback newer than its last attempt landed and
|
|
11
|
+
* no ledger window holds it. Refs without recent feedback can still be picked
|
|
12
|
+
* by the fallback lanes (proactive maintenance, high salience, forgetting
|
|
13
|
+
* safety); the survivors are ranked by salience, checked on disk and capped.
|
|
14
|
+
* A plan-only run evaluates the same selectors against read snapshots and
|
|
15
|
+
* writes nothing.
|
|
16
|
+
*/
|
|
4
17
|
import fs from "node:fs";
|
|
5
18
|
import path from "node:path";
|
|
6
|
-
import { makeBundleRef } from "../../core/asset/asset-ref.js";
|
|
7
19
|
import { parseFrontmatter } from "../../core/asset/frontmatter.js";
|
|
8
|
-
import { typeNameFromConceptId } from "../../core/asset/resolve-ref.js";
|
|
9
20
|
import { daysToMs } from "../../core/common.js";
|
|
10
21
|
import { loadConfig } from "../../core/config/config.js";
|
|
11
22
|
import { ConfigError, rethrowIfTestIsolationError } from "../../core/errors.js";
|
|
12
23
|
import { appendEvent, readEvents } from "../../core/events.js";
|
|
13
|
-
import {
|
|
24
|
+
import { withStateDb } from "../../core/state-db.js";
|
|
14
25
|
import { info, warn } from "../../core/warn.js";
|
|
15
26
|
import { countUsageEventsByType } from "../../indexer/usage/usage-events.js";
|
|
16
27
|
import { getAvailableHarnesses } from "../../integrations/session-logs/index.js";
|
|
17
|
-
import { withLlmStage } from "../../llm/usage-telemetry.js";
|
|
18
|
-
import { closeDatabase, openExistingDatabase, openReadonlyExistingDatabase, } from "../../storage/repositories/index-connection.js";
|
|
19
28
|
import { getZeroResultSearches } from "../../storage/repositories/index-entries-repository.js";
|
|
20
29
|
import { getRetrievalCounts } from "../../storage/repositories/index-utility-repository.js";
|
|
21
30
|
import { listStateProposals } from "../../storage/repositories/proposals-repository.js";
|
|
22
31
|
import { akmLint } from "../lint/index.js";
|
|
23
32
|
import { runSchemaRepairPass } from "../sources/schema-repair.js";
|
|
24
33
|
import { isAutonomyLaneAllowed } from "./autonomy-gate.js";
|
|
25
|
-
import { akmConsolidate, inspectConsolidationPool, loadExistingKnowledgeBodyHashes } from "./consolidate.js";
|
|
34
|
+
import { akmConsolidate, inspectConsolidationPool, loadExistingKnowledgeBodyHashes, makeConsolidateResult, } from "./consolidate.js";
|
|
26
35
|
import { computeSafeChunkSize, DEFAULT_CONTEXT_LENGTH_TOKENS } from "./consolidate/chunking.js";
|
|
27
|
-
|
|
28
|
-
import { buildLatestFeedbackTsMap, buildLatestProposalTsMap, buildUtilityMap, dedupeRefs, findAssetFilePath, isDistillCandidateRef, isLessonCandidate, isSignalDeltaEligible, resolveImproveScope, } from "./eligibility.js";
|
|
36
|
+
import { assetTypeOf, buildUtilityMap, dedupeRefs, findAssetFilePath, isDistillCandidateRef, isLessonCandidate, resolveImproveScope, withIndexDb, } from "./eligibility.js";
|
|
29
37
|
import { akmExtract, countNewExtractCandidates } from "./extract.js";
|
|
30
|
-
import { computeValenceScore
|
|
38
|
+
import { computeValenceScore } from "./feedback-valence.js";
|
|
39
|
+
import { isLedgerBlocked, lastAttemptByRef, ledgerRowFor, loadLedgerSnapshot, stateKey, stripBundle, } from "./ledger.js";
|
|
31
40
|
import { applyMemoryCleanup } from "./memory/memory-improve.js";
|
|
32
|
-
import {
|
|
41
|
+
import { getAllAssetOutcomes, getAssetOutcome, getOutcomeScoresByRef, OUTCOME_SCORE_MAX, outcomeScoreToSalience, projectAssetOutcome, updateAssetOutcome, } from "./outcome-loop.js";
|
|
33
42
|
import { projectMemoryCleanup, selectEffectiveImproveRefs } from "./planner.js";
|
|
34
43
|
import { DEFAULT_DUE_DAYS, DEFAULT_MAX_PER_RUN, selectProactiveMaintenanceRefs } from "./proactive-maintenance.js";
|
|
35
44
|
import { buildRankChangeReport, computeSalience, getAllRankScores, getAssetSalience, getLastUseMsByRef, isContentEncodingRow, SALIENCE_NO_OP_DAMPEN_FACTOR, SALIENCE_NO_OP_DAMPEN_THRESHOLD, upsertAssetSalience, } from "./salience.js";
|
|
36
|
-
import {
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
45
|
+
import { attributeStage, errMessage } from "./stage.js";
|
|
46
|
+
/** The candidate's durable state key (salience, outcome, ledger). */
|
|
47
|
+
const keyOf = (r) => stateKey(r.ref, r.itemRef);
|
|
48
|
+
/**
|
|
49
|
+
* Run `fn` against the run's state.db (its long-lived handle when there is
|
|
50
|
+
* one). A plan-only run without a handle reads nothing. Best-effort.
|
|
51
|
+
*/
|
|
52
|
+
function withRunState(eventsCtx, persist, fn) {
|
|
53
|
+
if (!persist && !eventsCtx?.db)
|
|
54
|
+
return undefined;
|
|
55
|
+
try {
|
|
56
|
+
return withStateDb(fn, { path: eventsCtx?.dbPath, borrowed: eventsCtx?.db });
|
|
57
|
+
}
|
|
58
|
+
catch (err) {
|
|
59
|
+
rethrowIfTestIsolationError(err);
|
|
60
|
+
return undefined;
|
|
42
61
|
}
|
|
43
|
-
return undefined;
|
|
44
|
-
}
|
|
45
|
-
function readConsecutiveNoOpsForImproveRef(db, ref, itemRef) {
|
|
46
|
-
return readAssetSalienceForImproveRef(db, ref, itemRef)?.consecutive_no_ops ?? 0;
|
|
47
62
|
}
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
//
|
|
58
|
-
// itemRefByRef is `ref → item_ref | undefined`, built once per pass from the
|
|
59
|
-
// candidate set.
|
|
60
|
-
/** `ref → item_ref | undefined` for a run's candidate set. */
|
|
61
|
-
function buildItemRefByRef(refs) {
|
|
62
|
-
const m = new Map();
|
|
63
|
-
for (const r of refs)
|
|
64
|
-
m.set(r.ref, r.itemRef);
|
|
65
|
-
return m;
|
|
63
|
+
function fileSize(filePath) {
|
|
64
|
+
if (!filePath)
|
|
65
|
+
return undefined;
|
|
66
|
+
try {
|
|
67
|
+
return fs.statSync(filePath).size;
|
|
68
|
+
}
|
|
69
|
+
catch {
|
|
70
|
+
return undefined;
|
|
71
|
+
}
|
|
66
72
|
}
|
|
67
|
-
/**
|
|
68
|
-
function
|
|
69
|
-
|
|
73
|
+
/** `{ key: value }` for each key `source` defines. */
|
|
74
|
+
export function pickDefined(source, keys) {
|
|
75
|
+
const out = {};
|
|
76
|
+
for (const key of keys)
|
|
77
|
+
if (source?.[key] !== undefined)
|
|
78
|
+
out[key] = source[key];
|
|
79
|
+
return out;
|
|
70
80
|
}
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
|
|
81
|
+
export const CONSOLIDATION_CONFIG_KEYS = [
|
|
82
|
+
"enabled",
|
|
83
|
+
"minPoolSize",
|
|
84
|
+
"limit",
|
|
85
|
+
"maxChunkSize",
|
|
86
|
+
"incrementalSince",
|
|
87
|
+
];
|
|
88
|
+
/** Emit an aggregate `improve_skipped` row (never one per ref). */
|
|
89
|
+
export function recordImproveSkip(eventsCtx, ref, metadata) {
|
|
90
|
+
appendEvent({ eventType: "improve_skipped", ref, metadata }, eventsCtx);
|
|
74
91
|
}
|
|
75
|
-
/**
|
|
76
|
-
function
|
|
77
|
-
const
|
|
78
|
-
|
|
92
|
+
/** Per-originator rolling error windows (3 each) shown to later prompts as patterns to avoid. */
|
|
93
|
+
export function pushRecentError(recentErrors, originator, msg) {
|
|
94
|
+
const window = recentErrors[originator] ?? [];
|
|
95
|
+
window.push(msg);
|
|
96
|
+
if (window.length > 3)
|
|
97
|
+
window.shift();
|
|
98
|
+
recentErrors[originator] = window;
|
|
79
99
|
}
|
|
100
|
+
// ── Consolidation ────────────────────────────────────────────────────────────
|
|
80
101
|
/**
|
|
81
|
-
*
|
|
82
|
-
*
|
|
83
|
-
*
|
|
102
|
+
* The consolidation gates and pool, with no model call: the profile toggle,
|
|
103
|
+
* `minPoolSize` (not for a named strategy or ref scope, nor once the pool is
|
|
104
|
+
* over the 100-memory volume trigger), and the ledger delta (every memory
|
|
105
|
+
* judged recently and unchanged since means nothing to do).
|
|
84
106
|
*/
|
|
85
|
-
function evaluateConsolidationEligibility(args) {
|
|
86
|
-
const { options, primaryStashDir, memorySummary, improveProfile, resolvedPlan, eventsCtx } = args;
|
|
87
|
-
const MEMORY_VOLUME_THRESHOLD = options.memoryVolumeConsolidationThreshold ?? 100;
|
|
88
|
-
const hasLlm = resolvedPlan.processes.consolidate.runner !== null;
|
|
89
|
-
const volumeTriggered = typeof memorySummary.eligible === "number" && memorySummary.eligible > MEMORY_VOLUME_THRESHOLD && hasLlm;
|
|
90
|
-
// 0.8.0 pool-delta gate for consolidate: re-eligible iff at least one
|
|
91
|
-
// memory file has been updated since the most recent successful
|
|
92
|
-
// consolidate_completed event. Time-based cooldowns produced the same
|
|
93
|
-
// synchronised-wave failure mode the reflect/distill cooldowns did; the
|
|
94
|
-
// pool-delta gate ties consolidation to actual work-to-do.
|
|
95
|
-
const sourceName = options.sourceName ?? options.writeTarget?.source.name ?? options.config?.defaultBundle ?? "stash";
|
|
96
|
-
const recentConsolidations = readEvents({ type: "consolidate_completed" }, eventsCtx);
|
|
97
|
-
const lastConsolidation = recentConsolidations.events
|
|
98
|
-
.filter((e) => e.metadata?.source === sourceName && Number(e.metadata?.processed) > 0)
|
|
99
|
-
.sort((a, b) => new Date(b.ts ?? 0).getTime() - new Date(a.ts ?? 0).getTime())[0];
|
|
100
|
-
const lastConsolidateTs = typeof lastConsolidation?.metadata?.completedThrough === "string"
|
|
101
|
-
? lastConsolidation.metadata.completedThrough
|
|
102
|
-
: lastConsolidation?.ts;
|
|
103
|
-
// #551 smarter gate: build the set of memory asset paths whose only delta
|
|
104
|
-
// since the last consolidate is their OWN promotion. Those files
|
|
105
|
-
// have not had a full improve cycle to settle, so they offer no merge /
|
|
106
|
-
// contradiction candidates yet — excluding them stops the gate firing on
|
|
107
|
-
// freshly-promoted single-source memories. We read `promoted` events emitted
|
|
108
|
-
// after the last consolidate; each carries the written `assetPath`.
|
|
109
|
-
const promotedSinceConsolidate = (() => {
|
|
110
|
-
const paths = new Set();
|
|
111
|
-
try {
|
|
112
|
-
const promoted = readEvents({
|
|
113
|
-
type: "promoted",
|
|
114
|
-
...(lastConsolidateTs ? { since: lastConsolidateTs } : {}),
|
|
115
|
-
}, eventsCtx).events;
|
|
116
|
-
for (const e of promoted) {
|
|
117
|
-
const ap = e.metadata?.assetPath;
|
|
118
|
-
if (typeof ap === "string" && ap.length > 0)
|
|
119
|
-
paths.add(path.resolve(ap));
|
|
120
|
-
}
|
|
121
|
-
}
|
|
122
|
-
catch {
|
|
123
|
-
// best-effort: if the events query fails, fall back to no exclusions
|
|
124
|
-
// (preserves pre-#551 behaviour rather than over-skipping).
|
|
125
|
-
}
|
|
126
|
-
return paths;
|
|
127
|
-
})();
|
|
128
|
-
// Pool-delta: any memory file with mtime > lastConsolidateTs flags work to do,
|
|
129
|
-
// EXCEPT files whose only post-consolidate change was their own promotion.
|
|
130
|
-
// Using file mtime keeps this query DB-free and matches what the indexer
|
|
131
|
-
// already uses as the canonical `memory.updated_at` proxy.
|
|
132
|
-
//
|
|
133
|
-
// Bootstrap: when no successful consolidate_completed event has ever been
|
|
134
|
-
// recorded, we cannot evaluate the pool-delta — treat as eligible so a
|
|
135
|
-
// fresh stash runs consolidate once before the steady-state gate kicks in.
|
|
136
|
-
//
|
|
137
|
-
// R4: the volume override is bootstrap-only — it exists to force that same
|
|
138
|
-
// "fresh stash, consolidate once" run when the pool is already large enough
|
|
139
|
-
// that waiting for the steady-state gate would be wasteful. Once a
|
|
140
|
-
// consolidate_completed event exists, the pool-delta gate below governs on
|
|
141
|
-
// its own; a large eligible pool no longer bypasses it.
|
|
142
|
-
const memoryUpdatedAfterLastConsolidate = (() => {
|
|
143
|
-
if (!lastConsolidateTs)
|
|
144
|
-
return true; // bootstrap path: never consolidated (volume override included).
|
|
145
|
-
if (!primaryStashDir)
|
|
146
|
-
return false;
|
|
147
|
-
const memoriesDir = path.join(primaryStashDir, "memories");
|
|
148
|
-
if (!fs.existsSync(memoriesDir))
|
|
149
|
-
return false;
|
|
150
|
-
try {
|
|
151
|
-
const pending = [memoriesDir];
|
|
152
|
-
while (pending.length > 0) {
|
|
153
|
-
const current = pending.pop();
|
|
154
|
-
for (const entry of fs.readdirSync(current, { withFileTypes: true })) {
|
|
155
|
-
const filePath = path.join(current, entry.name);
|
|
156
|
-
if (entry.isDirectory()) {
|
|
157
|
-
pending.push(filePath);
|
|
158
|
-
continue;
|
|
159
|
-
}
|
|
160
|
-
if (!entry.isFile() || !entry.name.endsWith(".md"))
|
|
161
|
-
continue;
|
|
162
|
-
if (promotedSinceConsolidate.has(path.resolve(filePath)))
|
|
163
|
-
continue;
|
|
164
|
-
try {
|
|
165
|
-
if (fs.statSync(filePath).mtime.toISOString() > lastConsolidateTs)
|
|
166
|
-
return true;
|
|
167
|
-
}
|
|
168
|
-
catch {
|
|
169
|
-
// Ignore files that disappear during the scan.
|
|
170
|
-
}
|
|
171
|
-
}
|
|
172
|
-
}
|
|
173
|
-
return false;
|
|
174
|
-
}
|
|
175
|
-
catch {
|
|
176
|
-
return false;
|
|
177
|
-
}
|
|
178
|
-
})();
|
|
179
|
-
// R4: no longer `!volumeTriggered && ...` — the volume override only ever
|
|
180
|
-
// applies at bootstrap (see `memoryUpdatedAfterLastConsolidate` above), so
|
|
181
|
-
// the pool-delta result alone determines cooldown post-bootstrap.
|
|
182
|
-
const consolidationOnCooldown = !memoryUpdatedAfterLastConsolidate;
|
|
183
|
-
// Profile gate: if profile explicitly disables consolidate, skip the entire pass.
|
|
184
|
-
const consolidateDisabledByProfile = improveProfile?.processes?.consolidate?.enabled === false;
|
|
185
|
-
// #553 minPoolSize guard: skip consolidation when the eligible memory pool is
|
|
186
|
-
// below a minimum size, rather than spending an LLM pass on a handful of
|
|
187
|
-
// memories. This is an INDEPENDENT skip condition from #551's mtime pool-delta
|
|
188
|
-
// gate — either can skip. Default 0 (disabled) — every built-in strategy
|
|
189
|
-
// used to ship 500, which meant `akm improve --strategy consolidate`, typed
|
|
190
|
-
// by a human, silently did nothing on almost every real install. Evaluated
|
|
191
|
-
// against the eligible-pool count BEFORE entering the LLM loop so a skip
|
|
192
|
-
// costs ZERO LLM calls when an operator opts back into a floor.
|
|
193
|
-
const CONSOLIDATE_DEFAULT_MIN_POOL_SIZE = 0;
|
|
194
|
-
const configuredMinPoolSize = improveProfile?.processes?.consolidate?.minPoolSize;
|
|
195
|
-
const minPoolSize = typeof configuredMinPoolSize === "number" ? configuredMinPoolSize : CONSOLIDATE_DEFAULT_MIN_POOL_SIZE;
|
|
196
|
-
const eligiblePoolSize = typeof memorySummary.eligible === "number" ? memorySummary.eligible : 0;
|
|
197
|
-
const userNamedStrategyOrScope = options.strategy !== undefined || resolveImproveScope(options.scope).mode === "ref";
|
|
198
|
-
// volumeTriggered means the pool already exceeds the volume threshold (100),
|
|
199
|
-
// so a force-triggered run never trips the pool-size guard. The guard only
|
|
200
|
-
// engages when minPoolSize > 0 and the eligible pool is strictly below it.
|
|
201
|
-
const poolBelowMinSize = !volumeTriggered && !userNamedStrategyOrScope && minPoolSize > 0 && eligiblePoolSize < minPoolSize;
|
|
202
|
-
return {
|
|
203
|
-
volumeTriggered,
|
|
204
|
-
consolidationOnCooldown,
|
|
205
|
-
consolidateDisabledByProfile,
|
|
206
|
-
poolBelowMinSize,
|
|
207
|
-
eligiblePoolSize,
|
|
208
|
-
minPoolSize,
|
|
209
|
-
...(lastConsolidateTs ? { lastConsolidationTs: lastConsolidateTs } : {}),
|
|
210
|
-
};
|
|
211
|
-
}
|
|
212
|
-
/** Build the no-dispatch consolidation projection consumed by dry and live. */
|
|
213
107
|
function planConsolidationPass(args) {
|
|
214
|
-
const { options, primaryStashDir, memorySummary,
|
|
215
|
-
const processConfig = improveProfile?.processes?.consolidate;
|
|
216
|
-
const
|
|
217
|
-
|
|
218
|
-
|
|
219
|
-
|
|
220
|
-
|
|
221
|
-
resolvedPlan,
|
|
222
|
-
eventsCtx,
|
|
223
|
-
});
|
|
224
|
-
const effectiveOptions = {
|
|
225
|
-
...options.consolidateOptions,
|
|
226
|
-
config: options.config,
|
|
227
|
-
stashDir: options.stashDir,
|
|
228
|
-
writeTarget: options.writeTarget,
|
|
229
|
-
limit: processConfig?.limit,
|
|
230
|
-
incrementalSince: processConfig?.incrementalSince,
|
|
231
|
-
neighborsPerChanged: processConfig?.neighborsPerChanged,
|
|
232
|
-
maxChunkSize: processConfig?.maxChunkSize,
|
|
233
|
-
};
|
|
234
|
-
const poolWarnings = [];
|
|
235
|
-
// Same hash set the live run's pre-filter uses (R2-1), so the preview's
|
|
236
|
-
// candidate pool and eligibility gate agree with what the run will act on.
|
|
237
|
-
// Reuse the caller's set when given one (runConsolidationPass) instead of
|
|
238
|
-
// walking knowledge/ again here.
|
|
108
|
+
const { options, primaryStashDir, memorySummary, resolvedPlan } = args;
|
|
109
|
+
const processConfig = args.improveProfile?.processes?.consolidate;
|
|
110
|
+
const volumeTriggered = memorySummary.eligible > 100 && resolvedPlan.processes.consolidate.runner !== null;
|
|
111
|
+
const minPoolSize = typeof processConfig?.minPoolSize === "number" ? processConfig.minPoolSize : 0;
|
|
112
|
+
const eligiblePoolSize = typeof memorySummary.eligible === "number" ? memorySummary.eligible : 0;
|
|
113
|
+
const userNamed = options.strategy !== undefined || resolveImproveScope(options.scope).mode === "ref";
|
|
114
|
+
const poolBelowMinSize = !volumeTriggered && !userNamed && minPoolSize > 0 && eligiblePoolSize < minPoolSize;
|
|
239
115
|
const pool = primaryStashDir
|
|
240
|
-
? inspectConsolidationPool(
|
|
241
|
-
|
|
242
|
-
|
|
243
|
-
|
|
244
|
-
|
|
245
|
-
|
|
246
|
-
|
|
247
|
-
|
|
116
|
+
? inspectConsolidationPool({
|
|
117
|
+
config: options.config,
|
|
118
|
+
stashDir: options.stashDir,
|
|
119
|
+
writeTarget: options.writeTarget,
|
|
120
|
+
target: options.target,
|
|
121
|
+
limit: processConfig?.limit,
|
|
122
|
+
incrementalSince: processConfig?.incrementalSince,
|
|
123
|
+
neighborsPerChanged: processConfig?.neighborsPerChanged,
|
|
124
|
+
maxChunkSize: processConfig?.maxChunkSize,
|
|
125
|
+
}, primaryStashDir, [], args.existingKnowledgeBodyHashes ?? loadExistingKnowledgeBodyHashes(primaryStashDir), { readOnly: args.eventsCtx?.readOnly === true })
|
|
126
|
+
: { poolSize: 0, candidatePoolSize: 0, judgedUnchanged: 0 };
|
|
127
|
+
// A credential-unavailable engine still resolved its context length.
|
|
128
|
+
const unavailable = resolvedPlan.engineUnavailable.find((item) => item.process === "consolidate");
|
|
248
129
|
const chunkSize = computeSafeChunkSize(resolvedPlan.processes.consolidate.runner?.connection.contextLength ??
|
|
249
|
-
|
|
130
|
+
unavailable?.contextLength ??
|
|
250
131
|
DEFAULT_CONTEXT_LENGTH_TOKENS, 500, processConfig?.maxChunkSize);
|
|
251
|
-
const profilePassed =
|
|
252
|
-
const
|
|
253
|
-
const
|
|
254
|
-
const
|
|
255
|
-
const
|
|
256
|
-
const reason = !profilePassed
|
|
257
|
-
? "disabled by improve profile"
|
|
258
|
-
: !minimumPoolPassed
|
|
259
|
-
? `pool ${eligibility.eligiblePoolSize} is below minPoolSize ${eligibility.minPoolSize}`
|
|
260
|
-
: !deltaPassed
|
|
261
|
-
? "no memory updates since the last completed consolidation"
|
|
262
|
-
: !nonEmptyPool
|
|
263
|
-
? "candidate pool is empty after narrowing"
|
|
264
|
-
: "all consolidation gates pass";
|
|
132
|
+
const profilePassed = processConfig?.enabled !== false;
|
|
133
|
+
const deltaPassed = pool.candidatePoolSize > 0 || pool.judgedUnchanged === 0;
|
|
134
|
+
const wouldRun = profilePassed && !poolBelowMinSize && deltaPassed && pool.candidatePoolSize > 0;
|
|
135
|
+
const belowMin = `pool ${eligiblePoolSize} is below minPoolSize ${minPoolSize}`;
|
|
136
|
+
const unchanged = "every memory was judged recently and is unchanged since";
|
|
265
137
|
return {
|
|
266
|
-
|
|
138
|
+
poolBelowMinSize,
|
|
139
|
+
eligiblePoolSize,
|
|
140
|
+
minPoolSize,
|
|
267
141
|
plan: {
|
|
268
|
-
configured:
|
|
269
|
-
...(processConfig?.enabled !== undefined ? { enabled: processConfig.enabled } : {}),
|
|
270
|
-
...(processConfig?.minPoolSize !== undefined ? { minPoolSize: processConfig.minPoolSize } : {}),
|
|
271
|
-
...(processConfig?.limit !== undefined ? { limit: processConfig.limit } : {}),
|
|
272
|
-
...(processConfig?.maxChunkSize !== undefined ? { maxChunkSize: processConfig.maxChunkSize } : {}),
|
|
273
|
-
...(processConfig?.incrementalSince !== undefined ? { incrementalSince: processConfig.incrementalSince } : {}),
|
|
274
|
-
},
|
|
142
|
+
configured: pickDefined(processConfig, CONSOLIDATION_CONFIG_KEYS),
|
|
275
143
|
effective: {
|
|
276
144
|
enabled: profilePassed,
|
|
277
|
-
minPoolSize
|
|
145
|
+
minPoolSize,
|
|
278
146
|
...(processConfig?.limit !== undefined ? { limit: processConfig.limit } : {}),
|
|
279
147
|
chunkSize,
|
|
280
148
|
},
|
|
@@ -286,338 +154,182 @@ function planConsolidationPass(args) {
|
|
|
286
154
|
reason: profilePassed ? "consolidation enabled" : "disabled by improve profile",
|
|
287
155
|
},
|
|
288
156
|
minimumPool: {
|
|
289
|
-
passed:
|
|
290
|
-
reason:
|
|
291
|
-
? `pool satisfies minPoolSize ${eligibility.minPoolSize}`
|
|
292
|
-
: `pool ${eligibility.eligiblePoolSize} is below minPoolSize ${eligibility.minPoolSize}`,
|
|
157
|
+
passed: !poolBelowMinSize,
|
|
158
|
+
reason: poolBelowMinSize ? belowMin : `pool satisfies minPoolSize ${minPoolSize}`,
|
|
293
159
|
},
|
|
294
160
|
delta: {
|
|
295
161
|
passed: deltaPassed,
|
|
296
162
|
reason: !deltaPassed
|
|
297
|
-
?
|
|
298
|
-
:
|
|
299
|
-
?
|
|
300
|
-
: "no
|
|
163
|
+
? unchanged
|
|
164
|
+
: pool.judgedUnchanged > 0
|
|
165
|
+
? `${pool.judgedUnchanged} recently judged, unchanged memories skipped`
|
|
166
|
+
: "no memory was judged recently",
|
|
301
167
|
},
|
|
302
168
|
},
|
|
303
169
|
wouldRun,
|
|
304
|
-
reason
|
|
170
|
+
reason: !profilePassed
|
|
171
|
+
? "disabled by improve profile"
|
|
172
|
+
: poolBelowMinSize
|
|
173
|
+
? belowMin
|
|
174
|
+
: !deltaPassed
|
|
175
|
+
? unchanged
|
|
176
|
+
: pool.candidatePoolSize === 0
|
|
177
|
+
? "candidate pool is empty after narrowing"
|
|
178
|
+
: "all consolidation gates pass",
|
|
305
179
|
estimatedChunks: wouldRun ? Math.ceil(pool.candidatePoolSize / chunkSize) : 0,
|
|
306
180
|
},
|
|
307
181
|
};
|
|
308
182
|
}
|
|
309
|
-
|
|
310
|
-
const { options, primaryStashDir,
|
|
311
|
-
|
|
312
|
-
const consolidationConfig = baseConfig;
|
|
313
|
-
// Computed once here and reused by both the pool preview below and the
|
|
314
|
-
// akmConsolidate call further down (R2-1/R3-1) — knowledge/ can hold
|
|
315
|
-
// thousands of files, so walking it twice per run would double that cost.
|
|
183
|
+
async function runConsolidationPass(args) {
|
|
184
|
+
const { options, primaryStashDir, improveProfile, resolvedPlan, eventsCtx } = args;
|
|
185
|
+
// Walked once and shared with the live pass.
|
|
316
186
|
const existingKnowledgeBodyHashes = primaryStashDir ? loadExistingKnowledgeBodyHashes(primaryStashDir) : undefined;
|
|
317
|
-
const planned = planConsolidationPass({
|
|
318
|
-
|
|
319
|
-
|
|
320
|
-
|
|
321
|
-
improveProfile,
|
|
322
|
-
resolvedPlan,
|
|
323
|
-
eventsCtx,
|
|
324
|
-
existingKnowledgeBodyHashes,
|
|
325
|
-
});
|
|
326
|
-
const { volumeTriggered, consolidationOnCooldown, consolidateDisabledByProfile, poolBelowMinSize, eligiblePoolSize, minPoolSize, lastConsolidationTs, } = planned.eligibility;
|
|
327
|
-
let consolidation = {
|
|
328
|
-
schemaVersion: 1,
|
|
329
|
-
ok: true,
|
|
330
|
-
shape: "consolidate-result",
|
|
331
|
-
dryRun: false,
|
|
332
|
-
previewOnly: false,
|
|
333
|
-
target: "",
|
|
334
|
-
processed: 0,
|
|
335
|
-
merged: 0,
|
|
336
|
-
deleted: 0,
|
|
337
|
-
promoted: [],
|
|
338
|
-
contradicted: 0,
|
|
339
|
-
warnings: [],
|
|
340
|
-
durationMs: 0,
|
|
341
|
-
};
|
|
342
|
-
if (consolidateDisabledByProfile) {
|
|
187
|
+
const planned = planConsolidationPass({ ...args, existingKnowledgeBodyHashes });
|
|
188
|
+
const processConfig = improveProfile?.processes?.consolidate;
|
|
189
|
+
let consolidation = makeConsolidateResult({ target: "", durationMs: 0 });
|
|
190
|
+
if (!planned.plan.gates.profile.passed) {
|
|
343
191
|
info("[improve] consolidation skipped (disabled by improve profile)");
|
|
344
192
|
}
|
|
345
|
-
else if (poolBelowMinSize) {
|
|
346
|
-
|
|
347
|
-
|
|
348
|
-
|
|
349
|
-
|
|
350
|
-
|
|
351
|
-
|
|
352
|
-
|
|
353
|
-
|
|
354
|
-
|
|
355
|
-
|
|
356
|
-
},
|
|
357
|
-
}, eventsCtx);
|
|
358
|
-
info(`[improve] consolidation skipped (pool ${eligiblePoolSize} < minPoolSize ${minPoolSize})`);
|
|
193
|
+
else if (planned.poolBelowMinSize) {
|
|
194
|
+
recordImproveSkip(eventsCtx, "memories/_consolidation", {
|
|
195
|
+
reason: "pool_below_min_size",
|
|
196
|
+
poolSize: planned.eligiblePoolSize,
|
|
197
|
+
minPoolSize: planned.minPoolSize,
|
|
198
|
+
});
|
|
199
|
+
info(`[improve] consolidation skipped (pool ${planned.eligiblePoolSize} < minPoolSize ${planned.minPoolSize})`);
|
|
200
|
+
}
|
|
201
|
+
else if (!planned.plan.gates.delta.passed) {
|
|
202
|
+
recordImproveSkip(eventsCtx, "memories/_consolidation", { reason: "consolidation_no_memory_updates" });
|
|
203
|
+
info("[improve] consolidation skipped (every memory was judged recently and is unchanged)");
|
|
359
204
|
}
|
|
360
|
-
else
|
|
361
|
-
|
|
362
|
-
|
|
363
|
-
...options.
|
|
364
|
-
config:
|
|
205
|
+
else {
|
|
206
|
+
consolidation = await attributeStage(resolvedPlan, "consolidate", () => akmConsolidate({
|
|
207
|
+
target: options.target,
|
|
208
|
+
...(options.writeTarget ? { writeTarget: options.writeTarget } : {}),
|
|
209
|
+
config: options.config ?? loadConfig(),
|
|
365
210
|
dryRun: options.dryRun ?? false,
|
|
366
211
|
stashDir: options.stashDir,
|
|
367
|
-
// Active profile for this improve run — lets consolidate's secondary
|
|
368
|
-
// process-config reads honor `--profile <name>` instead of `default`.
|
|
369
212
|
improveProfile,
|
|
370
213
|
llmRunner: resolvedPlan.processes.consolidate.runner,
|
|
371
|
-
autoTriggered: volumeTriggered,
|
|
372
|
-
// Reuse the hash set computed above instead of a second knowledge/
|
|
373
|
-
// walk inside akmConsolidateInner (R2-1/R3-1).
|
|
374
214
|
existingKnowledgeBodyHashes,
|
|
375
|
-
// Tie consolidate proposals back to this improve invocation so
|
|
376
|
-
// accept-rate-per-run aggregation works. Mirrors reflect/propose/extract.
|
|
377
215
|
sourceRun: `consolidate-${Date.now()}`,
|
|
378
|
-
|
|
379
|
-
|
|
380
|
-
|
|
381
|
-
|
|
382
|
-
|
|
383
|
-
|
|
384
|
-
|
|
385
|
-
maxChunkSize: improveProfile?.processes?.consolidate?.maxChunkSize,
|
|
386
|
-
// WS-3a: forward budget signal for graceful abort on timeout, and pass
|
|
387
|
-
// the profile's p90 estimate for cold-start budget reduction.
|
|
388
|
-
signal: budgetSignal,
|
|
389
|
-
p90ChunkSecondsDefault: improveProfile?.processes?.consolidate?.p90ChunkSecondsDefault,
|
|
390
|
-
// WS-5: pass total run budget so perfTelemetry.estimatedBudgetFractionUsed
|
|
391
|
-
// can flag when consolidation alone exceeded the budget.
|
|
392
|
-
runBudgetMs,
|
|
393
|
-
}), { engine: resolvedPlan.processes.consolidate.runner?.engine, process: "consolidate" });
|
|
394
|
-
const sourceName = options.sourceName ?? options.writeTarget?.source.name ?? baseConfig.defaultBundle ?? "stash";
|
|
395
|
-
const complete = (consolidation.failedChunks ?? 0) === 0 &&
|
|
396
|
-
(consolidation.failedChunkMemories ?? 0) === 0 &&
|
|
397
|
-
(consolidation.failedPromotions ?? 0) === 0 &&
|
|
398
|
-
(consolidation.deferredMemories ?? 0) === 0;
|
|
399
|
-
// R4: advisory ops (merge/delete/contradict) are never auto-applied — see
|
|
400
|
-
// consolidate.ts — so a run that plans some is still a completed pass over
|
|
401
|
-
// the pool, not an incomplete one. Gating the event on zero advisory ops
|
|
402
|
-
// meant it was never emitted in practice, which kept the pool-delta gate
|
|
403
|
-
// permanently bootstrapped. Record the unapplied count for reporting
|
|
404
|
-
// instead of withholding the event.
|
|
405
|
-
const advisoryOpsUnapplied = consolidation.planned?.filter((op) => op.op !== "promote").length ?? 0;
|
|
406
|
-
if (consolidation.ok && !consolidation.dryRun && complete && consolidation.processed > 0) {
|
|
407
|
-
appendEvent({
|
|
408
|
-
eventType: "consolidate_completed",
|
|
409
|
-
ref: makeBundleRef(sourceName, "memories/_consolidation"),
|
|
410
|
-
metadata: {
|
|
411
|
-
processed: consolidation.processed,
|
|
412
|
-
source: sourceName,
|
|
413
|
-
completedThrough: consolidationStartedAt,
|
|
414
|
-
merged: consolidation.merged,
|
|
415
|
-
deleted: consolidation.deleted,
|
|
416
|
-
contradicted: consolidation.contradicted,
|
|
417
|
-
failedChunks: consolidation.failedChunks ?? 0,
|
|
418
|
-
durationMs: consolidation.durationMs,
|
|
419
|
-
advisoryOpsUnapplied,
|
|
420
|
-
},
|
|
421
|
-
}, eventsCtx);
|
|
422
|
-
}
|
|
423
|
-
}
|
|
424
|
-
else {
|
|
425
|
-
appendEvent({
|
|
426
|
-
eventType: "improve_skipped",
|
|
427
|
-
ref: "memories/_consolidation",
|
|
428
|
-
metadata: {
|
|
429
|
-
reason: "consolidation_no_memory_updates",
|
|
430
|
-
lastEventTs: lastConsolidationTs ?? null,
|
|
431
|
-
},
|
|
432
|
-
}, eventsCtx);
|
|
433
|
-
info("[improve] consolidation skipped (no memory updates since last run)");
|
|
216
|
+
limit: processConfig?.limit,
|
|
217
|
+
incrementalSince: processConfig?.incrementalSince,
|
|
218
|
+
neighborsPerChanged: processConfig?.neighborsPerChanged,
|
|
219
|
+
maxChunkSize: processConfig?.maxChunkSize,
|
|
220
|
+
signal: args.budgetSignal,
|
|
221
|
+
p90ChunkSecondsDefault: processConfig?.p90ChunkSecondsDefault,
|
|
222
|
+
}));
|
|
434
223
|
}
|
|
435
|
-
|
|
436
|
-
// detector (loop-stages.ts, gated on this flag). `processed` counts memories the LLM JUDGED,
|
|
437
|
-
// not files consolidation WROTE — R4's advisory gate (consolidate.ts) means merge/delete/
|
|
438
|
-
// contradict ops are never auto-applied, and the only op that does execute, promote, calls
|
|
439
|
-
// emitProposal → createProposal, which persists to the `proposals` table in state.db, not to
|
|
440
|
-
// any file under the stash (src/commands/proposal/repository.ts). So `processed > 0` is the
|
|
441
|
-
// right gate here: the detector needs one snapshot per cycle where consolidate did work,
|
|
442
|
-
// regardless of whether that work produced a write. It would be the wrong gate for anything
|
|
443
|
-
// that needs to know whether a stash file changed, since promote/merge/delete/contradict never
|
|
444
|
-
// write one.
|
|
445
|
-
const consolidationRan = !consolidateDisabledByProfile &&
|
|
446
|
-
!poolBelowMinSize &&
|
|
447
|
-
!consolidationOnCooldown &&
|
|
448
|
-
!consolidation.previewOnly &&
|
|
449
|
-
consolidation.processed > 0;
|
|
450
|
-
return { consolidation, consolidationRan, plan: planned.plan };
|
|
224
|
+
return { consolidation, plan: planned.plan };
|
|
451
225
|
}
|
|
452
|
-
/**
|
|
453
|
-
|
|
454
|
-
|
|
455
|
-
* selector independently.
|
|
456
|
-
*/
|
|
457
|
-
function inspectExtractPass(args) {
|
|
458
|
-
const { options, improveProfile, resolvedPlan, eventsCtx, readOnly } = args;
|
|
226
|
+
/** The extract gates, evaluated once for both the live pass and the dry-run report. */
|
|
227
|
+
function inspectExtractPass(args, readOnly) {
|
|
228
|
+
const { options, improveProfile, resolvedPlan, eventsCtx } = args;
|
|
459
229
|
const enabled = resolvedPlan.processes.extract.enabled;
|
|
460
230
|
const hasRunner = resolvedPlan.processes.extract.runner?.engine !== undefined;
|
|
461
|
-
const availableHarnesses = (options.extractHarnesses ?? getAvailableHarnesses()).filter((
|
|
462
|
-
const
|
|
463
|
-
const minNewSessions = typeof
|
|
231
|
+
const availableHarnesses = (options.extractHarnesses ?? getAvailableHarnesses()).filter((h) => h.isAvailable());
|
|
232
|
+
const configured = improveProfile.processes?.extract?.minNewSessions;
|
|
233
|
+
const minNewSessions = typeof configured === "number" ? configured : 0;
|
|
464
234
|
let newCandidateCount;
|
|
465
235
|
if (enabled && hasRunner && availableHarnesses.length > 0 && minNewSessions > 0) {
|
|
466
|
-
const
|
|
467
|
-
newCandidateCount =
|
|
236
|
+
const defaultSince = improveProfile.processes?.extract?.defaultSince;
|
|
237
|
+
newCandidateCount = (options.extractCandidateCountFn ?? countNewExtractCandidates)(options.config ?? loadConfig(), {
|
|
468
238
|
harnesses: availableHarnesses,
|
|
469
239
|
improveProfile,
|
|
470
|
-
...(
|
|
471
|
-
? { since: improveProfile.processes.extract.defaultSince }
|
|
472
|
-
: {}),
|
|
240
|
+
...(defaultSince ? { since: defaultSince } : {}),
|
|
473
241
|
...(eventsCtx?.db ? { stateDb: eventsCtx.db } : {}),
|
|
474
242
|
...(!readOnly && eventsCtx?.dbPath ? { stateDbPath: eventsCtx.dbPath } : {}),
|
|
475
243
|
...(readOnly ? { readOnly: true } : {}),
|
|
476
244
|
});
|
|
477
245
|
}
|
|
478
246
|
const belowMinNewSessions = minNewSessions > 0 && newCandidateCount !== undefined && newCandidateCount < minNewSessions;
|
|
479
|
-
const
|
|
480
|
-
const reason = !enabled
|
|
481
|
-
? "disabled"
|
|
482
|
-
: !hasRunner
|
|
483
|
-
? "enabled but no runner is resolved"
|
|
484
|
-
: availableHarnesses.length === 0
|
|
485
|
-
? "enabled but no session-log harness is available"
|
|
486
|
-
: belowMinNewSessions
|
|
487
|
-
? `${newCandidateCount ?? 0} new sessions is below minNewSessions ${minNewSessions}`
|
|
488
|
-
: minNewSessions > 0
|
|
489
|
-
? `${newCandidateCount ?? 0} new sessions satisfies minNewSessions ${minNewSessions}`
|
|
490
|
-
: `enabled with ${availableHarnesses.length} available session-log harness(es); minNewSessions is disabled`;
|
|
247
|
+
const count = `${newCandidateCount ?? 0} new sessions`;
|
|
491
248
|
return {
|
|
492
249
|
availableHarnesses,
|
|
493
250
|
minNewSessions,
|
|
494
251
|
...(newCandidateCount !== undefined ? { newCandidateCount } : {}),
|
|
495
252
|
belowMinNewSessions,
|
|
496
|
-
wouldRun,
|
|
497
|
-
reason
|
|
253
|
+
wouldRun: enabled && hasRunner && availableHarnesses.length > 0 && !belowMinNewSessions,
|
|
254
|
+
reason: !enabled
|
|
255
|
+
? "disabled"
|
|
256
|
+
: !hasRunner
|
|
257
|
+
? "enabled but no runner is resolved"
|
|
258
|
+
: availableHarnesses.length === 0
|
|
259
|
+
? "enabled but no session-log harness is available"
|
|
260
|
+
: belowMinNewSessions
|
|
261
|
+
? `${count} is below minNewSessions ${minNewSessions}`
|
|
262
|
+
: minNewSessions > 0
|
|
263
|
+
? `${count} satisfies minNewSessions ${minNewSessions}`
|
|
264
|
+
: `enabled with ${availableHarnesses.length} available session-log harness(es); minNewSessions is disabled`,
|
|
498
265
|
};
|
|
499
266
|
}
|
|
500
267
|
/**
|
|
501
|
-
*
|
|
502
|
-
*
|
|
503
|
-
*
|
|
504
|
-
* results + any warnings collected along the way.
|
|
268
|
+
* One `akmExtract` per available harness under the strategy's frozen plan. A
|
|
269
|
+
* harness that throws is a warning; the `minNewSessions` gate skips the whole
|
|
270
|
+
* pass with no model call.
|
|
505
271
|
*/
|
|
506
|
-
async function runSessionExtractPass(args) {
|
|
507
|
-
const { options, primaryStashDir,
|
|
272
|
+
async function runSessionExtractPass(args, plan) {
|
|
273
|
+
const { options, primaryStashDir, resolvedPlan, eventsCtx } = args;
|
|
508
274
|
const warnings = [];
|
|
509
|
-
|
|
510
|
-
|
|
511
|
-
|
|
512
|
-
|
|
513
|
-
|
|
514
|
-
|
|
515
|
-
|
|
516
|
-
|
|
517
|
-
|
|
518
|
-
|
|
519
|
-
|
|
520
|
-
|
|
521
|
-
|
|
522
|
-
|
|
523
|
-
|
|
524
|
-
|
|
525
|
-
|
|
526
|
-
|
|
527
|
-
|
|
528
|
-
|
|
529
|
-
|
|
530
|
-
|
|
531
|
-
// already done upstream; here we elide every akmExtract/processSession call)
|
|
532
|
-
// when the NEW (unseen, in-window) candidate-session pool is below a minimum.
|
|
533
|
-
// 22% of improve runs produce zero memory-inference writes because extract
|
|
534
|
-
// finds no new sessions, yet still burns the full extract pipeline. Default 0
|
|
535
|
-
// (disabled) preserves existing always-run behaviour; only opted-in profiles
|
|
536
|
-
// (e.g. a user-enabled `frequent` strategy) set it. Evaluated BEFORE any LLM
|
|
537
|
-
// call so a skip costs zero LLM work AND writes nothing. A skipped extract
|
|
538
|
-
// never flags work for the NEXT run's consolidation mtime-gate (the
|
|
539
|
-
// downstream trigger #554 asks us to suppress).
|
|
540
|
-
const plan = args.plan ?? inspectExtractPass({ options, improveProfile, resolvedPlan, eventsCtx, readOnly: false });
|
|
541
|
-
// #593/#594: the ACTIVE resolved improve profile is the single source of
|
|
542
|
-
// truth for whether extract runs. (Previously this also ANDed in the legacy
|
|
543
|
-
// `session_extraction` feature flag, which only reads
|
|
544
|
-
// a retired global feature path; the selected strategy is authoritative.)
|
|
545
|
-
// `akmExtract` re-checks the same active profile internally via `improveProfile`.
|
|
546
|
-
if (resolvedPlan.processes.extract.enabled) {
|
|
547
|
-
const extractRunner = resolvedPlan.processes.extract.runner;
|
|
548
|
-
if (!extractRunner?.engine) {
|
|
549
|
-
throw new ConfigError("Resolved improve plan has no runner for enabled extract process.", "LLM_NOT_CONFIGURED");
|
|
550
|
-
}
|
|
551
|
-
const extractPlan = Object.freeze({
|
|
552
|
-
strategy: resolvedPlan.strategy.name,
|
|
553
|
-
engine: extractRunner.engine,
|
|
554
|
-
enabled: true,
|
|
555
|
-
process: resolvedPlan.processes.extract.config,
|
|
556
|
-
runner: extractRunner,
|
|
557
|
-
timeoutMs: extractRunner.timeoutMs === undefined ? 600_000 : extractRunner.timeoutMs,
|
|
558
|
-
embeddingConfig: Object.freeze(structuredClone(extractConfig.embedding)),
|
|
559
|
-
...(resolvedPlan.processes.extract.notices?.length ? { notices: resolvedPlan.processes.extract.notices } : {}),
|
|
275
|
+
if (!resolvedPlan.processes.extract.enabled)
|
|
276
|
+
return { warnings };
|
|
277
|
+
const runner = resolvedPlan.processes.extract.runner;
|
|
278
|
+
if (!runner?.engine) {
|
|
279
|
+
throw new ConfigError("Resolved improve plan has no runner for enabled extract process.", "LLM_NOT_CONFIGURED");
|
|
280
|
+
}
|
|
281
|
+
const config = options.config ?? loadConfig();
|
|
282
|
+
const extractPlan = Object.freeze({
|
|
283
|
+
strategy: resolvedPlan.strategy.name,
|
|
284
|
+
engine: runner.engine,
|
|
285
|
+
enabled: true,
|
|
286
|
+
process: resolvedPlan.processes.extract.config,
|
|
287
|
+
runner,
|
|
288
|
+
timeoutMs: runner.timeoutMs === undefined ? 600_000 : runner.timeoutMs,
|
|
289
|
+
embeddingConfig: Object.freeze(structuredClone(config.embedding)),
|
|
290
|
+
...(resolvedPlan.processes.extract.notices?.length ? { notices: resolvedPlan.processes.extract.notices } : {}),
|
|
291
|
+
});
|
|
292
|
+
if (plan.belowMinNewSessions) {
|
|
293
|
+
recordImproveSkip(eventsCtx, "memories/_extract", {
|
|
294
|
+
reason: "below_min_new_sessions",
|
|
295
|
+
newSessions: plan.newCandidateCount ?? 0,
|
|
296
|
+
minNewSessions: plan.minNewSessions,
|
|
560
297
|
});
|
|
561
|
-
|
|
562
|
-
|
|
563
|
-
|
|
564
|
-
|
|
565
|
-
|
|
566
|
-
|
|
567
|
-
|
|
568
|
-
|
|
569
|
-
|
|
570
|
-
|
|
571
|
-
|
|
572
|
-
|
|
573
|
-
|
|
574
|
-
|
|
298
|
+
info(`[improve] extract skipped (new sessions ${plan.newCandidateCount ?? 0} < minNewSessions ${plan.minNewSessions})`);
|
|
299
|
+
}
|
|
300
|
+
if (!plan.wouldRun)
|
|
301
|
+
return { warnings };
|
|
302
|
+
const extractResults = [];
|
|
303
|
+
for (const harness of plan.availableHarnesses) {
|
|
304
|
+
try {
|
|
305
|
+
extractResults.push(await attributeStage(resolvedPlan, "extract", () => akmExtract({
|
|
306
|
+
type: harness.name,
|
|
307
|
+
...(primaryStashDir !== undefined ? { stashDir: primaryStashDir } : {}),
|
|
308
|
+
config,
|
|
309
|
+
resolvedPlan: extractPlan,
|
|
310
|
+
dryRun: options.dryRun ?? false,
|
|
311
|
+
signal: args.budgetSignal,
|
|
312
|
+
...(options.extractHarnesses ? { harnesses: options.extractHarnesses } : {}),
|
|
313
|
+
...(eventsCtx?.dbPath ? { stateDbPath: eventsCtx.dbPath } : {}),
|
|
314
|
+
eventsCtx,
|
|
315
|
+
})));
|
|
575
316
|
}
|
|
576
|
-
|
|
577
|
-
|
|
578
|
-
for (const h of availableHarnesses) {
|
|
579
|
-
try {
|
|
580
|
-
const result = await withLlmStage("session-extraction", () => akmExtract({
|
|
581
|
-
type: h.name,
|
|
582
|
-
...(primaryStashDir !== undefined ? { stashDir: primaryStashDir } : {}),
|
|
583
|
-
config: extractConfig,
|
|
584
|
-
resolvedPlan: extractPlan,
|
|
585
|
-
dryRun: options.dryRun ?? false,
|
|
586
|
-
signal: budgetSignal,
|
|
587
|
-
...(options.extractHarnesses ? { harnesses: options.extractHarnesses } : {}),
|
|
588
|
-
// C2: pin extract's skip-tracking state.db open to the boundary path.
|
|
589
|
-
...(eventsCtx?.dbPath ? { stateDbPath: eventsCtx.dbPath } : {}),
|
|
590
|
-
// R25: extract's event emits reuse the run's events context
|
|
591
|
-
// (fast path when it carries the long-lived handle).
|
|
592
|
-
eventsCtx,
|
|
593
|
-
}), { engine: resolvedPlan.processes.extract.runner?.engine, process: "extract" });
|
|
594
|
-
extractResults.push(result);
|
|
595
|
-
}
|
|
596
|
-
catch (err) {
|
|
597
|
-
const msg = err instanceof Error ? err.message : String(err);
|
|
598
|
-
warnings.push(`extract(${h.name}) failed: ${msg}`);
|
|
599
|
-
}
|
|
600
|
-
}
|
|
601
|
-
if (extractResults.length === 0) {
|
|
602
|
-
// All harnesses threw — clear so the envelope's `extract` field is
|
|
603
|
-
// absent rather than misleadingly empty.
|
|
604
|
-
extractResults = undefined;
|
|
605
|
-
}
|
|
317
|
+
catch (err) {
|
|
318
|
+
warnings.push(`extract(${harness.name}) failed: ${errMessage(err)}`);
|
|
606
319
|
}
|
|
607
320
|
}
|
|
608
|
-
|
|
609
|
-
|
|
610
|
-
warnings,
|
|
611
|
-
};
|
|
321
|
+
// Every harness threw: no `extract` field rather than a misleadingly empty one.
|
|
322
|
+
return { ...(extractResults.length > 0 ? { extractResults } : {}), warnings };
|
|
612
323
|
}
|
|
324
|
+
// ── Validation ───────────────────────────────────────────────────────────────
|
|
613
325
|
/**
|
|
614
|
-
*
|
|
615
|
-
*
|
|
616
|
-
*
|
|
326
|
+
* Structural validation (file on disk, lesson description) with optional LLM
|
|
327
|
+
* schema repair. A repair is advisory: a ref leaves the failure set only when a
|
|
328
|
+
* fresh read of the live asset passes.
|
|
617
329
|
*/
|
|
618
330
|
export async function runValidationAndRepairPass(args) {
|
|
619
|
-
const { postCleanupRefs, options,
|
|
620
|
-
const
|
|
331
|
+
const { postCleanupRefs, options, resolvedPlan, repairValidationFailures } = args;
|
|
332
|
+
const validate = async (candidate) => {
|
|
621
333
|
try {
|
|
622
334
|
const filePath = candidate.filePath && fs.existsSync(candidate.filePath)
|
|
623
335
|
? candidate.filePath
|
|
@@ -626,10 +338,8 @@ export async function runValidationAndRepairPass(args) {
|
|
|
626
338
|
return "file not found on disk";
|
|
627
339
|
if (path.extname(filePath).toLowerCase() !== ".md")
|
|
628
340
|
return undefined;
|
|
629
|
-
if (isLessonCandidate(candidate.ref)) {
|
|
630
|
-
|
|
631
|
-
if (!fm.description)
|
|
632
|
-
return "missing description";
|
|
341
|
+
if (isLessonCandidate(candidate.ref) && !parseFrontmatter(fs.readFileSync(filePath, "utf8")).data.description) {
|
|
342
|
+
return "missing description";
|
|
633
343
|
}
|
|
634
344
|
return undefined;
|
|
635
345
|
}
|
|
@@ -639,7 +349,7 @@ export async function runValidationAndRepairPass(args) {
|
|
|
639
349
|
};
|
|
640
350
|
const validationFailures = [];
|
|
641
351
|
for (const candidate of postCleanupRefs) {
|
|
642
|
-
const reason = await
|
|
352
|
+
const reason = await validate(candidate);
|
|
643
353
|
if (reason)
|
|
644
354
|
validationFailures.push({ ref: candidate.ref, reason });
|
|
645
355
|
}
|
|
@@ -649,166 +359,122 @@ export async function runValidationAndRepairPass(args) {
|
|
|
649
359
|
info(` ${f.ref}: ${f.reason}`);
|
|
650
360
|
}
|
|
651
361
|
let schemaRepairs = [];
|
|
652
|
-
const
|
|
653
|
-
|
|
654
|
-
if (repairValidationFailures && validationFailures.length > 0) {
|
|
655
|
-
const
|
|
656
|
-
|
|
657
|
-
|
|
658
|
-
|
|
659
|
-
|
|
660
|
-
|
|
661
|
-
|
|
662
|
-
|
|
663
|
-
|
|
664
|
-
|
|
665
|
-
|
|
666
|
-
|
|
667
|
-
|
|
668
|
-
|
|
669
|
-
|
|
670
|
-
isLessonCandidateFn: isLessonCandidate,
|
|
671
|
-
}), { engine: resolvedPlan.processes.validation.runner?.engine, process: "validation" });
|
|
672
|
-
schemaRepairs = result.repairs;
|
|
673
|
-
// A repair result is advisory. Only a fresh structural read of the live
|
|
674
|
-
// asset can remove it from the failure set; queued content is not live.
|
|
675
|
-
const failedRefs = new Set(validationFailures.map((failure) => failure.ref));
|
|
676
|
-
const candidatesByRef = new Map(postCleanupRefs.map((candidate) => [candidate.ref, candidate]));
|
|
677
|
-
for (const ref of failedRefs) {
|
|
678
|
-
const candidate = candidatesByRef.get(ref);
|
|
679
|
-
if (candidate && !(await validateCandidate(candidate)))
|
|
680
|
-
repairedRefs.add(ref);
|
|
681
|
-
}
|
|
362
|
+
const repaired = new Set();
|
|
363
|
+
const runner = resolvedPlan.processes.validation.runner;
|
|
364
|
+
if (repairValidationFailures && validationFailures.length > 0 && runner) {
|
|
365
|
+
const result = await attributeStage(resolvedPlan, "validation", () => (args.schemaRepairFn ?? runSchemaRepairPass)(validationFailures, {
|
|
366
|
+
startMs: args.startMs,
|
|
367
|
+
budgetMs: args.budgetMs,
|
|
368
|
+
llmRunner: runner,
|
|
369
|
+
// The resolved source path, not the raw `--stash-dir` flag.
|
|
370
|
+
stashDir: args.primaryStashDir,
|
|
371
|
+
findFilePath: findAssetFilePath,
|
|
372
|
+
isLessonCandidateFn: isLessonCandidate,
|
|
373
|
+
}));
|
|
374
|
+
schemaRepairs = result.repairs;
|
|
375
|
+
const byRef = new Map(postCleanupRefs.map((candidate) => [candidate.ref, candidate]));
|
|
376
|
+
for (const { ref } of validationFailures) {
|
|
377
|
+
const candidate = byRef.get(ref);
|
|
378
|
+
if (candidate && !(await validate(candidate)))
|
|
379
|
+
repaired.add(ref);
|
|
682
380
|
}
|
|
683
381
|
}
|
|
684
|
-
const validationFailureRefs = new Set(validationFailures.filter((f) => !
|
|
685
|
-
if (
|
|
686
|
-
info(`[improve] schema repair fixed ${
|
|
382
|
+
const validationFailureRefs = new Set(validationFailures.filter((f) => !repaired.has(f.ref)).map((f) => f.ref));
|
|
383
|
+
if (repaired.size > 0) {
|
|
384
|
+
info(`[improve] schema repair fixed ${repaired.size}/${validationFailures.length} validation failures; ${validationFailureRefs.size} remain`);
|
|
687
385
|
}
|
|
688
386
|
return { validationFailures, validationFailureRefs, schemaRepairs };
|
|
689
387
|
}
|
|
690
|
-
|
|
691
|
-
|
|
692
|
-
|
|
693
|
-
|
|
694
|
-
|
|
695
|
-
|
|
696
|
-
const
|
|
697
|
-
|
|
698
|
-
if (memoryBudget.warning)
|
|
699
|
-
cleanupWarnings.push(memoryBudget.warning);
|
|
700
|
-
// Consolidation intentionally precedes extract so current-run promotions
|
|
701
|
-
// cannot force the pool-delta gate open (#551).
|
|
388
|
+
export async function runImprovePreparationStage(args) {
|
|
389
|
+
const { scope, options, plannedRefs, memoryCleanupPlan, primaryStashDir, eventsCtx, resolvedPlan } = args;
|
|
390
|
+
const planOnly = args.planOnly ?? options.dryRun === true;
|
|
391
|
+
const persist = !planOnly;
|
|
392
|
+
const actions = [];
|
|
393
|
+
const cleanupWarnings = [...(args.initialCleanupWarnings ?? [])];
|
|
394
|
+
const memoryIndexHealth = assessMemoryIndex(primaryStashDir, cleanupWarnings);
|
|
395
|
+
// Consolidation precedes extract, so it only judges memories from earlier runs.
|
|
702
396
|
const consolidationPass = planOnly
|
|
703
|
-
?
|
|
704
|
-
|
|
705
|
-
|
|
706
|
-
|
|
707
|
-
|
|
708
|
-
|
|
709
|
-
|
|
710
|
-
|
|
711
|
-
|
|
712
|
-
|
|
713
|
-
|
|
714
|
-
|
|
715
|
-
|
|
716
|
-
|
|
717
|
-
dryRun: true,
|
|
718
|
-
previewOnly: true,
|
|
719
|
-
target: options.target ?? options.stashDir ?? "",
|
|
720
|
-
processed: 0,
|
|
721
|
-
merged: 0,
|
|
722
|
-
deleted: 0,
|
|
723
|
-
promoted: [],
|
|
724
|
-
contradicted: 0,
|
|
725
|
-
warnings: [],
|
|
726
|
-
durationMs: 0,
|
|
727
|
-
},
|
|
728
|
-
consolidationRan: false,
|
|
729
|
-
plan: planned.plan,
|
|
730
|
-
};
|
|
731
|
-
})()
|
|
732
|
-
: await runConsolidationPass({
|
|
733
|
-
options,
|
|
734
|
-
primaryStashDir,
|
|
735
|
-
memorySummary,
|
|
736
|
-
improveProfile,
|
|
737
|
-
resolvedPlan,
|
|
738
|
-
eventsCtx,
|
|
739
|
-
budgetSignal,
|
|
740
|
-
runBudgetMs: budgetMs,
|
|
741
|
-
});
|
|
742
|
-
const extractPlan = inspectExtractPass({ options, improveProfile, resolvedPlan, eventsCtx, readOnly: planOnly });
|
|
743
|
-
const extractPass = planOnly
|
|
744
|
-
? { extractResults: undefined, warnings: [] }
|
|
745
|
-
: await runSessionExtractPass({
|
|
746
|
-
options,
|
|
747
|
-
primaryStashDir,
|
|
748
|
-
improveProfile,
|
|
749
|
-
resolvedPlan,
|
|
750
|
-
eventsCtx,
|
|
751
|
-
budgetSignal,
|
|
752
|
-
plan: extractPlan,
|
|
753
|
-
});
|
|
754
|
-
if (extractPass.warnings.length > 0)
|
|
755
|
-
cleanupWarnings.push(...extractPass.warnings);
|
|
756
|
-
if (!planOnly) {
|
|
397
|
+
? {
|
|
398
|
+
consolidation: makeConsolidateResult({
|
|
399
|
+
dryRun: true,
|
|
400
|
+
previewOnly: true,
|
|
401
|
+
target: options.target ?? options.stashDir ?? "",
|
|
402
|
+
durationMs: 0,
|
|
403
|
+
}),
|
|
404
|
+
plan: planConsolidationPass(args).plan,
|
|
405
|
+
}
|
|
406
|
+
: await runConsolidationPass(args);
|
|
407
|
+
const extractPlan = inspectExtractPass(args, planOnly);
|
|
408
|
+
const extractPass = planOnly ? { warnings: [] } : await runSessionExtractPass(args, extractPlan);
|
|
409
|
+
cleanupWarnings.push(...extractPass.warnings);
|
|
410
|
+
if (persist) {
|
|
757
411
|
appendEvent({
|
|
758
412
|
eventType: "improve_invoked",
|
|
759
413
|
ref: scope.mode === "ref" ? scope.value : `improve:${scope.mode}:${scope.value ?? "all"}`,
|
|
760
|
-
metadata: {
|
|
414
|
+
metadata: {
|
|
415
|
+
strategy: args.strategyName,
|
|
416
|
+
scope,
|
|
417
|
+
dryRun: options.dryRun ?? false,
|
|
418
|
+
eligibleCount: plannedRefs.length,
|
|
419
|
+
},
|
|
761
420
|
}, eventsCtx);
|
|
762
421
|
}
|
|
763
|
-
|
|
764
|
-
const
|
|
765
|
-
|
|
766
|
-
|
|
767
|
-
|
|
768
|
-
|
|
769
|
-
|
|
770
|
-
|
|
771
|
-
|
|
772
|
-
|
|
773
|
-
|
|
774
|
-
|
|
422
|
+
// Memory cleanup: archive redundant derived memories (autonomy-gated).
|
|
423
|
+
const allowCleanup = isAutonomyLaneAllowed("memoryCleanup", options.config ?? loadConfig());
|
|
424
|
+
let appliedCleanup;
|
|
425
|
+
if (persist) {
|
|
426
|
+
try {
|
|
427
|
+
appliedCleanup =
|
|
428
|
+
primaryStashDir && memoryCleanupPlan && allowCleanup
|
|
429
|
+
? applyMemoryCleanup(primaryStashDir, memoryCleanupPlan)
|
|
430
|
+
: undefined;
|
|
431
|
+
}
|
|
432
|
+
catch (err) {
|
|
433
|
+
cleanupWarnings.push(`applyMemoryCleanup failed: ${errMessage(err)}`);
|
|
775
434
|
}
|
|
776
|
-
|
|
777
|
-
|
|
778
|
-
|
|
435
|
+
}
|
|
436
|
+
const cleanup = planOnly
|
|
437
|
+
? projectMemoryCleanup({
|
|
438
|
+
mode: "estimate",
|
|
439
|
+
plannedRefs,
|
|
440
|
+
candidateRefs: memoryCleanupPlan?.pruneCandidates.map((candidate) => candidate.ref) ?? [],
|
|
441
|
+
allowApply: allowCleanup,
|
|
442
|
+
})
|
|
443
|
+
: projectMemoryCleanup({
|
|
444
|
+
mode: "execution",
|
|
779
445
|
plannedRefs,
|
|
780
|
-
|
|
781
|
-
|
|
782
|
-
allowApply: allowCleanupApply,
|
|
446
|
+
archivedRefs: appliedCleanup?.archived.map((record) => record.ref) ?? [],
|
|
447
|
+
allowApply: allowCleanup,
|
|
783
448
|
});
|
|
784
|
-
|
|
785
|
-
|
|
786
|
-
|
|
787
|
-
|
|
449
|
+
if (appliedCleanup) {
|
|
450
|
+
for (const candidate of memoryCleanupPlan?.pruneCandidates ?? []) {
|
|
451
|
+
if (!appliedCleanup.archived.some((record) => record.ref === candidate.ref))
|
|
452
|
+
continue;
|
|
453
|
+
actions.push({
|
|
454
|
+
ref: candidate.ref,
|
|
455
|
+
mode: "memory-prune",
|
|
456
|
+
result: { ok: true, pruned: true, reason: candidate.reason },
|
|
457
|
+
});
|
|
458
|
+
}
|
|
459
|
+
if ((appliedCleanup.archived.length > 0 || appliedCleanup.beliefStateTransitions.length > 0) && primaryStashDir) {
|
|
460
|
+
try {
|
|
461
|
+
await args.reindexFn({ stashDir: primaryStashDir, signal: args.budgetSignal });
|
|
462
|
+
}
|
|
463
|
+
catch (err) {
|
|
464
|
+
cleanupWarnings.push(`reindex after cleanup failed: ${errMessage(err)}`);
|
|
465
|
+
}
|
|
466
|
+
}
|
|
467
|
+
}
|
|
468
|
+
const { postCleanupRefs } = cleanup;
|
|
469
|
+
const { validationFailures, validationFailureRefs, schemaRepairs } = await runValidationAndRepairPass({
|
|
470
|
+
postCleanupRefs,
|
|
788
471
|
options,
|
|
789
|
-
startMs,
|
|
790
|
-
budgetMs,
|
|
472
|
+
startMs: args.startMs,
|
|
473
|
+
budgetMs: args.budgetMs,
|
|
791
474
|
primaryStashDir,
|
|
792
475
|
resolvedPlan,
|
|
793
|
-
repairValidationFailures:
|
|
476
|
+
repairValidationFailures: persist && resolvedPlan.processes.validation.enabled && options.repairValidationFailures !== false,
|
|
794
477
|
});
|
|
795
|
-
return {
|
|
796
|
-
memoryIndexHealth: memoryBudget.memoryIndexHealth,
|
|
797
|
-
consolidationPass,
|
|
798
|
-
extractPlan,
|
|
799
|
-
extractResults: extractPass.extractResults,
|
|
800
|
-
appliedCleanup: cleanup.appliedCleanup,
|
|
801
|
-
postCleanupRefs: cleanup.postCleanupRefs,
|
|
802
|
-
cleanupGate: cleanup.gate,
|
|
803
|
-
...validation,
|
|
804
|
-
};
|
|
805
|
-
}
|
|
806
|
-
export async function runImprovePreparationStage(args) {
|
|
807
|
-
const { scope, options, primaryStashDir, eventsCtx, initialCleanupWarnings, improveProfile, resolvedPlan, planOnly = options.dryRun === true, } = args;
|
|
808
|
-
const actions = [];
|
|
809
|
-
const cleanupWarnings = initialCleanupWarnings ? [...initialCleanupWarnings] : [];
|
|
810
|
-
const { memoryIndexHealth, consolidationPass, extractPlan, extractResults, appliedCleanup, postCleanupRefs, cleanupGate, validationFailures, validationFailureRefs, schemaRepairs, } = await runPreparationPrelude({ ...args, planOnly, actions, cleanupWarnings });
|
|
811
|
-
// Phase 0.5 — structural hygiene pass
|
|
812
478
|
let lintSummary;
|
|
813
479
|
if (primaryStashDir) {
|
|
814
480
|
try {
|
|
@@ -816,1796 +482,642 @@ export async function runImprovePreparationStage(args) {
|
|
|
816
482
|
lintSummary = { fixed: lintResult.summary.fixed, flagged: lintResult.summary.flagged };
|
|
817
483
|
}
|
|
818
484
|
catch {
|
|
819
|
-
// lint
|
|
485
|
+
// lint never blocks improve
|
|
820
486
|
}
|
|
821
487
|
}
|
|
822
|
-
|
|
823
|
-
const
|
|
824
|
-
const
|
|
825
|
-
|
|
826
|
-
|
|
827
|
-
|
|
828
|
-
eventsCtx,
|
|
829
|
-
improveProfile,
|
|
830
|
-
resolvedPlan,
|
|
831
|
-
postCleanupRefs,
|
|
832
|
-
validationFailureRefs,
|
|
833
|
-
snapshot,
|
|
834
|
-
persist: !planOnly,
|
|
835
|
-
});
|
|
836
|
-
const eligibilitySourceByRef = stampEligibilitySource({
|
|
837
|
-
scope,
|
|
838
|
-
processableRefs: gathered.processableRefs,
|
|
839
|
-
mergedRefs: gathered.mergedRefs,
|
|
840
|
-
signalFiltered: gathered.signalFiltered,
|
|
841
|
-
proactiveRefs: gathered.proactiveRefs,
|
|
842
|
-
highSalienceRefs: gathered.highSalienceRefs,
|
|
843
|
-
});
|
|
844
|
-
// Shared admission boundary for every synthetic fallback lane. Cleanup and
|
|
845
|
-
// structural validation are exclusive selectors: no later rank/replay state
|
|
846
|
-
// may re-create a candidate they removed. Keep the exact surviving objects
|
|
847
|
-
// so any admitted fallback preserves its index-resolved file/item provenance.
|
|
848
|
-
const fallbackEligibleRefs = postCleanupRefs.filter((candidate) => !validationFailureRefs.has(candidate.ref));
|
|
849
|
-
const scored = scoreSalience({
|
|
850
|
-
scope,
|
|
851
|
-
options,
|
|
852
|
-
primaryStashDir,
|
|
853
|
-
eventsCtx,
|
|
854
|
-
mergedRefs: gathered.mergedRefs,
|
|
855
|
-
eligibilitySourceByRef,
|
|
856
|
-
feedbackSummary: gathered.feedbackSummary,
|
|
857
|
-
retrievalCounts: gathered.retrievalCounts,
|
|
858
|
-
signalFiltered: gathered.signalFiltered,
|
|
859
|
-
proactiveRefs: gathered.proactiveRefs,
|
|
860
|
-
highSalienceRefs: gathered.highSalienceRefs,
|
|
861
|
-
forgettingEligibleRefs: fallbackEligibleRefs,
|
|
862
|
-
persist: !planOnly,
|
|
863
|
-
});
|
|
864
|
-
// Replay is additive to the signal lanes, but it must not bypass selectors
|
|
865
|
-
// that have already removed a ref. Use the exact surviving objects so a
|
|
866
|
-
// replay admission preserves the index-resolved file/item provenance while
|
|
867
|
-
// excluding cleanup-pruned and structurally-invalid candidates.
|
|
868
|
-
const filtered = await filterEligibility({
|
|
869
|
-
scope,
|
|
870
|
-
options,
|
|
871
|
-
replayEligibleRefs: fallbackEligibleRefs,
|
|
872
|
-
eventsCtx,
|
|
873
|
-
mergedRefs: scored.mergedRefs,
|
|
874
|
-
salienceMap: scored.salienceMap,
|
|
875
|
-
eligibilitySourceByRef,
|
|
876
|
-
distillOnlyRefs: gathered.distillOnlyRefs,
|
|
877
|
-
validationFailureRefs,
|
|
878
|
-
summary: {
|
|
879
|
-
signalAndRetrievalRefs: gathered.signalAndRetrievalRefs,
|
|
880
|
-
signalFiltered: gathered.signalFiltered,
|
|
881
|
-
},
|
|
882
|
-
persist: !planOnly,
|
|
883
|
-
});
|
|
884
|
-
const preDiskRefSet = new Set(filtered.preDiskRefs.map((candidate) => candidate.ref));
|
|
885
|
-
const terminalSignalSkippedRefs = fallbackEligibleRefs.filter((candidate) => !preDiskRefSet.has(candidate.ref));
|
|
886
|
-
recordSignalSkipObservability({
|
|
887
|
-
actions,
|
|
888
|
-
terminalSignalSkippedRefs,
|
|
889
|
-
distillCooledRefs: gathered.distillCooledRefs,
|
|
890
|
-
eventsCtx,
|
|
891
|
-
persist: !planOnly,
|
|
892
|
-
});
|
|
893
|
-
// Gate counts are an exclusive, sequential accounting of the raw pool.
|
|
894
|
-
// Replay, proactive maintenance, high-salience, and forgetting-safety are
|
|
895
|
-
// legitimate signal-gate fallback lanes, so a ref admitted by any of them
|
|
896
|
-
// was not removed by the signal gate. Derive this count from the actual
|
|
897
|
-
// pre-disk survivor set instead of the earlier lane-rescue snapshot; the
|
|
898
|
-
// latter is intentionally assembled before replay and is also broader than
|
|
899
|
-
// the effective pool when --require-feedback-signal suppresses fallbacks.
|
|
900
|
-
const signalRemoved = terminalSignalSkippedRefs.length;
|
|
901
|
-
const totalReflectBlocked = terminalSignalSkippedRefs.length + gathered.distillOnlyRefs.length;
|
|
902
|
-
if (totalReflectBlocked > 0) {
|
|
903
|
-
info(`[improve] ${totalReflectBlocked} of ${gathered.preCooldownCount} indexed refs blocked by reflect signal-delta ` +
|
|
904
|
-
`(${terminalSignalSkippedRefs.length} fully skipped, ${gathered.distillOnlyRefs.length} routed to distill-only)`);
|
|
488
|
+
// Schema-repair errors get their own window; they are never shown to reflect.
|
|
489
|
+
const recentErrors = {};
|
|
490
|
+
for (const repair of schemaRepairs) {
|
|
491
|
+
if (repair.outcome !== "error")
|
|
492
|
+
continue;
|
|
493
|
+
pushRecentError(recentErrors, "schema-repair", repair.error ?? `schema repair error: ${repair.reason}`);
|
|
905
494
|
}
|
|
906
|
-
const
|
|
907
|
-
cleanupGate,
|
|
908
|
-
{
|
|
909
|
-
name: "validation",
|
|
910
|
-
removed: validationFailureRefs.size,
|
|
911
|
-
reason: "structural validation failures",
|
|
912
|
-
},
|
|
913
|
-
{
|
|
914
|
-
name: "signal",
|
|
915
|
-
removed: signalRemoved,
|
|
916
|
-
reason: "no fresh signal and no fallback lane selected the ref",
|
|
917
|
-
},
|
|
918
|
-
{
|
|
919
|
-
name: "disk",
|
|
920
|
-
removed: filtered.missingDiskCount,
|
|
921
|
-
reason: "backing asset is absent on disk",
|
|
922
|
-
},
|
|
923
|
-
{
|
|
924
|
-
name: "limit",
|
|
925
|
-
removed: filtered.limitRemoved,
|
|
926
|
-
reason: "deferred by the effective run limit",
|
|
927
|
-
},
|
|
928
|
-
];
|
|
495
|
+
const selection = await selectLoopCandidates(args, postCleanupRefs, validationFailureRefs, actions, persist);
|
|
929
496
|
return {
|
|
930
497
|
actions,
|
|
931
498
|
cleanupWarnings,
|
|
932
499
|
appliedCleanup,
|
|
933
500
|
memoryIndexHealth,
|
|
934
|
-
extract: extractResults,
|
|
935
|
-
actionableRefs:
|
|
936
|
-
signalBearingSet:
|
|
501
|
+
extract: extractPass.extractResults,
|
|
502
|
+
actionableRefs: selection.actionableRefs,
|
|
503
|
+
signalBearingSet: selection.signalBearingSet,
|
|
937
504
|
validationFailures,
|
|
938
505
|
schemaRepairs,
|
|
939
506
|
lintSummary,
|
|
940
|
-
loopRefs:
|
|
941
|
-
distillCooledRefs:
|
|
942
|
-
distillOnlyRefs:
|
|
943
|
-
coverageGaps:
|
|
507
|
+
loopRefs: selection.loopRefs,
|
|
508
|
+
distillCooledRefs: selection.distillCooledRefs,
|
|
509
|
+
distillOnlyRefs: selection.distillOnlyRefs,
|
|
510
|
+
coverageGaps: selection.coverageGaps,
|
|
944
511
|
recentErrors,
|
|
945
|
-
utilityMap: scored.utilityMap,
|
|
946
512
|
consolidation: consolidationPass.consolidation,
|
|
947
|
-
|
|
948
|
-
|
|
513
|
+
...(selection.proactive.proactiveMaintenanceSummary
|
|
514
|
+
? { proactiveMaintenance: selection.proactive.proactiveMaintenanceSummary }
|
|
515
|
+
: {}),
|
|
949
516
|
planning: {
|
|
950
|
-
gates:
|
|
951
|
-
|
|
952
|
-
|
|
517
|
+
gates: [
|
|
518
|
+
cleanup.gate,
|
|
519
|
+
{ name: "validation", removed: validationFailureRefs.size, reason: "structural validation failures" },
|
|
520
|
+
...selection.gates,
|
|
521
|
+
],
|
|
522
|
+
...(selection.proactive.proactivePlan ? { proactive: selection.proactive.proactivePlan } : {}),
|
|
953
523
|
consolidation: consolidationPass.plan,
|
|
954
524
|
extract: { wouldRun: extractPlan.wouldRun, reason: extractPlan.reason },
|
|
955
525
|
},
|
|
956
526
|
};
|
|
957
527
|
}
|
|
958
|
-
|
|
959
|
-
|
|
960
|
-
|
|
961
|
-
|
|
962
|
-
|
|
963
|
-
|
|
964
|
-
|
|
965
|
-
// inside the salience passes (pure fn; no separate pass)
|
|
966
|
-
// standards-context → does not exist in preparation.ts (assembly lives in
|
|
967
|
-
// extract.ts — recorded in the ledger, no empty pass)
|
|
968
|
-
// eligibility-filter → filterEligibility (+ replay/disk-check sub-passes)
|
|
969
|
-
// Every pass takes an args object and returns its results; the orchestrator
|
|
970
|
-
// folds them. Shared-by-reference structures (the ImproveEligibleRef objects,
|
|
971
|
-
// eligibilitySourceByRef, salienceMap, actions, recentErrors) keep their
|
|
972
|
-
// identity — attribution stamps must travel with the ref objects into the
|
|
973
|
-
// loop stage exactly as before.
|
|
974
|
-
/** Phase 0 — MEMORY.md budget check (200-line cap; warn at 180). */
|
|
975
|
-
function assessMemoryIndexBudget(primaryStashDir) {
|
|
976
|
-
let warning;
|
|
977
|
-
// Phase 0 — MEMORY.md budget check (200-line cap; warn at 180)
|
|
978
|
-
let memoryIndexHealth;
|
|
979
|
-
if (primaryStashDir) {
|
|
980
|
-
const memoryMdPath = path.join(primaryStashDir, "memories", "MEMORY.md");
|
|
981
|
-
if (fs.existsSync(memoryMdPath)) {
|
|
982
|
-
try {
|
|
983
|
-
const lines = fs.readFileSync(memoryMdPath, "utf8").split("\n").length;
|
|
984
|
-
const overBudget = lines >= 180;
|
|
985
|
-
memoryIndexHealth = { lineCount: lines, overBudget };
|
|
986
|
-
if (overBudget) {
|
|
987
|
-
warning = `MEMORY.md has ${lines} lines (budget: 200). Consolidation strongly recommended.`;
|
|
988
|
-
}
|
|
989
|
-
}
|
|
990
|
-
catch {
|
|
991
|
-
// best-effort
|
|
992
|
-
}
|
|
993
|
-
}
|
|
994
|
-
}
|
|
995
|
-
return { memoryIndexHealth, warning };
|
|
996
|
-
}
|
|
997
|
-
/**
|
|
998
|
-
* Memory-cleanup apply + prune-action recording + the post-cleanup reindex.
|
|
999
|
-
* Returns the surviving ref set and the prune actions/warnings for the
|
|
1000
|
-
* orchestrator to fold (same order as the old inline pushes).
|
|
1001
|
-
*/
|
|
1002
|
-
async function applyCleanupPass(args) {
|
|
1003
|
-
const { primaryStashDir, memoryCleanupPlan, plannedRefs, reindexFn, budgetSignal, allowApply } = args;
|
|
1004
|
-
const pruneActions = [];
|
|
1005
|
-
const warnings = [];
|
|
1006
|
-
let appliedCleanup;
|
|
528
|
+
/** MEMORY.md line budget: warn at 180 of 200 lines. */
|
|
529
|
+
function assessMemoryIndex(primaryStashDir, warnings) {
|
|
530
|
+
if (!primaryStashDir)
|
|
531
|
+
return undefined;
|
|
532
|
+
const memoryMdPath = path.join(primaryStashDir, "memories", "MEMORY.md");
|
|
533
|
+
if (!fs.existsSync(memoryMdPath))
|
|
534
|
+
return undefined;
|
|
1007
535
|
try {
|
|
1008
|
-
|
|
1009
|
-
|
|
1010
|
-
|
|
1011
|
-
: undefined;
|
|
1012
|
-
}
|
|
1013
|
-
catch (err) {
|
|
1014
|
-
warnings.push(`applyMemoryCleanup failed: ${err instanceof Error ? err.message : String(err)}`);
|
|
1015
|
-
}
|
|
1016
|
-
const projection = projectMemoryCleanup({
|
|
1017
|
-
mode: "execution",
|
|
1018
|
-
plannedRefs,
|
|
1019
|
-
archivedRefs: appliedCleanup?.archived.map((record) => record.ref) ?? [],
|
|
1020
|
-
allowApply,
|
|
1021
|
-
});
|
|
1022
|
-
// ── Phase 1: validation pass + schema repair (run on full postCleanupRefs) ──
|
|
1023
|
-
// Identifies refs whose on-disk asset has structural problems. Validation
|
|
1024
|
-
// failures are excluded from every downstream bucket. Run early so the
|
|
1025
|
-
// cooldown partition operates on a clean set.
|
|
1026
|
-
if (appliedCleanup) {
|
|
1027
|
-
for (const candidate of memoryCleanupPlan?.pruneCandidates ?? []) {
|
|
1028
|
-
const archived = appliedCleanup.archived.find((record) => record.ref === candidate.ref);
|
|
1029
|
-
if (!archived)
|
|
1030
|
-
continue;
|
|
1031
|
-
pruneActions.push({
|
|
1032
|
-
ref: candidate.ref,
|
|
1033
|
-
mode: "memory-prune",
|
|
1034
|
-
result: { ok: true, pruned: true, reason: candidate.reason },
|
|
1035
|
-
});
|
|
1036
|
-
}
|
|
1037
|
-
if ((appliedCleanup.archived.length > 0 || appliedCleanup.beliefStateTransitions.length > 0) && primaryStashDir) {
|
|
1038
|
-
try {
|
|
1039
|
-
await reindexFn({ stashDir: primaryStashDir, signal: budgetSignal });
|
|
1040
|
-
}
|
|
1041
|
-
catch (err) {
|
|
1042
|
-
warnings.push(`reindex after cleanup failed: ${err instanceof Error ? err.message : String(err)}`);
|
|
1043
|
-
}
|
|
536
|
+
const lineCount = fs.readFileSync(memoryMdPath, "utf8").split("\n").length;
|
|
537
|
+
if (lineCount >= 180) {
|
|
538
|
+
warnings.push(`MEMORY.md has ${lineCount} lines (budget: 200). Consolidation strongly recommended.`);
|
|
1044
539
|
}
|
|
540
|
+
return { lineCount, overBudget: lineCount >= 180 };
|
|
1045
541
|
}
|
|
1046
|
-
|
|
1047
|
-
|
|
1048
|
-
/** Seed the per-originator rolling error windows from schema-repair errors. */
|
|
1049
|
-
function seedRecentErrorWindows(schemaRepairs) {
|
|
1050
|
-
// O-5 / #378: Per-originator rolling error windows.
|
|
1051
|
-
// Reflexion (arXiv:2303.11366) warns that cross-task verbal critique
|
|
1052
|
-
// contamination degrades below single-shot baseline. Each originator key
|
|
1053
|
-
// ("schema-repair", "reflect") maintains its own rolling window so that
|
|
1054
|
-
// schema-repair failures are not injected as avoidPatterns into reflect calls.
|
|
1055
|
-
const recentErrors = {};
|
|
1056
|
-
const RECENT_ERRORS_CAP = 3;
|
|
1057
|
-
// Helper: push an error onto an originator's rolling window.
|
|
1058
|
-
function pushRecentError(originator, msg) {
|
|
1059
|
-
if (!recentErrors[originator])
|
|
1060
|
-
recentErrors[originator] = [];
|
|
1061
|
-
recentErrors[originator].push(msg);
|
|
1062
|
-
if (recentErrors[originator].length > RECENT_ERRORS_CAP)
|
|
1063
|
-
recentErrors[originator].shift();
|
|
1064
|
-
}
|
|
1065
|
-
// Seed schema-repair originator window from any schema-repair errors.
|
|
1066
|
-
for (const repair of schemaRepairs) {
|
|
1067
|
-
if (repair.outcome === "error") {
|
|
1068
|
-
const errMsg = repair.error ?? `schema repair error: ${repair.reason}`;
|
|
1069
|
-
pushRecentError("schema-repair", errMsg);
|
|
1070
|
-
}
|
|
542
|
+
catch {
|
|
543
|
+
return undefined;
|
|
1071
544
|
}
|
|
1072
|
-
return recentErrors;
|
|
1073
545
|
}
|
|
1074
|
-
|
|
546
|
+
// ── Candidate selection ──────────────────────────────────────────────────────
|
|
547
|
+
const FEEDBACK_SIGNAL_WINDOW_DAYS = 30;
|
|
548
|
+
/** Feedback that counts as a signal carries a signal or a note (a bare `akm feedback` does not). */
|
|
549
|
+
function isSignalEvent(metadata) {
|
|
550
|
+
const meta = metadata;
|
|
551
|
+
return meta !== undefined && (typeof meta.signal === "string" || typeof meta.note === "string");
|
|
552
|
+
}
|
|
553
|
+
/** One read of the feedback events and the ledger's reflect/distill rows. */
|
|
1075
554
|
export function buildSnapshotManifest(args) {
|
|
1076
|
-
const {
|
|
1077
|
-
// ── Phase 2: signal-delta eligibility sets built EARLY ────────────────────
|
|
1078
|
-
// 0.8.0 replaces the flat time-based cooldowns (which produced synchronised
|
|
1079
|
-
// waves whenever many refs cooled at the same instant — see the 2026-05-26
|
|
1080
|
-
// 54-ref simultaneous-reflect incident) with a *signal-delta* gate:
|
|
1081
|
-
//
|
|
1082
|
-
// reflectEligible(ref) ≡ latestFeedbackTs(ref) > lastReflectProposalTs(ref)
|
|
1083
|
-
// distillEligible(ref) ≡ latestFeedbackTs(ref) > lastDistillProposalTs(ref)
|
|
1084
|
-
//
|
|
1085
|
-
// i.e. a ref is re-eligible iff new feedback has landed since the last
|
|
1086
|
-
// proposal was generated for it. Stable content with no new signal stays
|
|
1087
|
-
// out of the queue regardless of clock time; a sudden burst of feedback
|
|
1088
|
-
// surfaces only the refs that the burst actually touches.
|
|
1089
|
-
//
|
|
1090
|
-
// The 30-day FEEDBACK_SIGNAL_WINDOW_DAYS bound still applies — only feedback
|
|
1091
|
-
// events newer than that count as "current signal". Ancient one-off
|
|
1092
|
-
// negatives don't permanently lock a ref into every run.
|
|
1093
|
-
const FEEDBACK_SIGNAL_WINDOW_DAYS = 30;
|
|
555
|
+
const { eventsCtx, stashDir } = args;
|
|
1094
556
|
const feedbackSinceCutoff = new Date(Date.now() - daysToMs(FEEDBACK_SIGNAL_WINDOW_DAYS)).toISOString();
|
|
1095
|
-
|
|
1096
|
-
|
|
1097
|
-
|
|
1098
|
-
const
|
|
1099
|
-
|
|
1100
|
-
|
|
1101
|
-
|
|
1102
|
-
|
|
1103
|
-
|
|
1104
|
-
|
|
1105
|
-
|
|
1106
|
-
|
|
1107
|
-
|
|
1108
|
-
|
|
1109
|
-
|
|
1110
|
-
|
|
1111
|
-
|
|
1112
|
-
|
|
1113
|
-
|
|
1114
|
-
|
|
1115
|
-
|
|
1116
|
-
|
|
1117
|
-
scope,
|
|
1118
|
-
options,
|
|
1119
|
-
postCleanupRefs,
|
|
1120
|
-
validationFailureRefs: args.validationFailureRefs,
|
|
1121
|
-
snapshot: args.snapshot,
|
|
1122
|
-
});
|
|
1123
|
-
const { distillCooledRefs, preCooldownCount, eligibleRefs, distillOnlyRefs, noFeedbackPool } = partition;
|
|
1124
|
-
// ── Phase 4: signal/feedback/utility/sort on the reduced set ──────────────
|
|
1125
|
-
// Everything from here works on (eligibleRefs ∪ distillOnlyRefs) plus the
|
|
1126
|
-
// deferred noFeedbackPool that may be rescued by the proactive-maintenance
|
|
1127
|
-
// (Layer 2) or high-salience (Layer 3) fallbacks below. The fully-skipped
|
|
1128
|
-
// bucket is retained as partition metadata only; terminal skip observability
|
|
1129
|
-
// is delayed until every fallback lane has finalized. We deliberately avoid
|
|
1130
|
-
// spending DB/CPU on refs that the signal-delta gate rejected with feedback
|
|
1131
|
-
// already on record.
|
|
1132
|
-
const processableRefs = [...eligibleRefs, ...distillOnlyRefs];
|
|
1133
|
-
const feedbackSummary = buildFeedbackSummaryMap({
|
|
1134
|
-
processableRefs,
|
|
1135
|
-
noFeedbackPool,
|
|
1136
|
-
eventsCtx,
|
|
1137
|
-
feedbackSinceCutoff,
|
|
1138
|
-
});
|
|
1139
|
-
const signalFiltered = processableRefs.filter((candidate) => feedbackSummary.get(candidate.ref)?.hasSignal === true);
|
|
1140
|
-
const signalBearingSet = new Set(signalFiltered.map((r) => r.ref));
|
|
1141
|
-
// Zero-feedback candidates for the proactive/high-salience fallbacks:
|
|
1142
|
-
// processableRefs without a recent signal, plus the deferred noFeedbackPool.
|
|
1143
|
-
// Dedupe by ref (the two sources are disjoint by construction, but guard
|
|
1144
|
-
// against overlap defensively).
|
|
1145
|
-
const noFeedbackSeen = new Set();
|
|
1146
|
-
const noFeedbackCandidates = [];
|
|
1147
|
-
for (const r of [...processableRefs.filter((r) => !signalBearingSet.has(r.ref)), ...noFeedbackPool]) {
|
|
1148
|
-
if (noFeedbackSeen.has(r.ref))
|
|
1149
|
-
continue;
|
|
1150
|
-
noFeedbackSeen.add(r.ref);
|
|
1151
|
-
noFeedbackCandidates.push(r);
|
|
557
|
+
const candidates = args.postCleanupRefs.filter((r) => !args.validationFailureRefs.has(r.ref));
|
|
558
|
+
const refByKey = new Map(candidates.map((r) => [keyOf(r), r.ref]));
|
|
559
|
+
const latestFeedbackTs = new Map();
|
|
560
|
+
const feedback = new Map(candidates.map((r) => [r.ref, { hasSignal: false, positive: 0, negative: 0 }]));
|
|
561
|
+
if (candidates.length > 0) {
|
|
562
|
+
for (const e of readEvents({ type: "feedback" }, eventsCtx).events) {
|
|
563
|
+
const ref = e.ref ? refByKey.get(e.ref) : undefined;
|
|
564
|
+
const entry = ref ? feedback.get(ref) : undefined;
|
|
565
|
+
if (!ref || !entry)
|
|
566
|
+
continue;
|
|
567
|
+
const ts = e.ts ?? "";
|
|
568
|
+
if (ts >= feedbackSinceCutoff && isSignalEvent(e.metadata)) {
|
|
569
|
+
entry.hasSignal = true;
|
|
570
|
+
if (ts > (latestFeedbackTs.get(ref) ?? ""))
|
|
571
|
+
latestFeedbackTs.set(ref, ts);
|
|
572
|
+
}
|
|
573
|
+
const signal = e.metadata?.signal;
|
|
574
|
+
if (signal === "positive")
|
|
575
|
+
entry.positive++;
|
|
576
|
+
else if (signal === "negative")
|
|
577
|
+
entry.negative++;
|
|
578
|
+
}
|
|
1152
579
|
}
|
|
1153
|
-
const
|
|
1154
|
-
|
|
1155
|
-
|
|
1156
|
-
signalFiltered,
|
|
1157
|
-
noFeedbackCandidates,
|
|
1158
|
-
eventsCtx,
|
|
1159
|
-
persist,
|
|
1160
|
-
});
|
|
1161
|
-
// `--require-feedback-signal` is a hard policy boundary, not merely a final
|
|
1162
|
-
// list filter. Do not run or report fallback selectors that the invocation
|
|
1163
|
-
// explicitly disabled (and do not emit their live selection events).
|
|
1164
|
-
const allowFallbacks = options.requireFeedbackSignal !== true;
|
|
1165
|
-
const proactive = allowFallbacks
|
|
1166
|
-
? selectProactiveMaintenanceLane({
|
|
1167
|
-
scope,
|
|
1168
|
-
improveProfile,
|
|
1169
|
-
resolvedPlan,
|
|
1170
|
-
eventsCtx,
|
|
1171
|
-
noFeedbackCandidates,
|
|
1172
|
-
lastReflectProposalTs,
|
|
1173
|
-
lastDistillProposalTs,
|
|
1174
|
-
retrievalCounts,
|
|
1175
|
-
lastUseMsForProactive,
|
|
1176
|
-
persist,
|
|
1177
|
-
})
|
|
1178
|
-
: { proactiveRefs: [] };
|
|
1179
|
-
const proactiveRefs = proactive.proactiveRefs;
|
|
1180
|
-
const proactiveMaintenanceSummary = proactive.proactiveMaintenanceSummary;
|
|
1181
|
-
const highSalienceRefs = allowFallbacks
|
|
1182
|
-
? selectHighSalienceLane({
|
|
1183
|
-
options,
|
|
1184
|
-
improveProfile,
|
|
1185
|
-
eventsCtx,
|
|
1186
|
-
noFeedbackCandidates,
|
|
1187
|
-
proactiveRefs,
|
|
1188
|
-
lastReflectProposalTs,
|
|
1189
|
-
persist,
|
|
1190
|
-
})
|
|
1191
|
-
: [];
|
|
1192
|
-
// If the user explicitly scoped to a single ref, always act on it —
|
|
1193
|
-
// skip the signal/retrieval filter entirely. The filter exists to avoid
|
|
1194
|
-
// noisy "improve everything" runs; it should not gate an intentional
|
|
1195
|
-
// per-ref invocation where the user's explicit choice is the signal.
|
|
1196
|
-
//
|
|
1197
|
-
// For type/all scope: only process refs with usage signals (recent feedback
|
|
1198
|
-
// or a proactive/high-salience rescue). A stash with no signals has 0
|
|
1199
|
-
// eligible refs — usage is the gate. Run `akm feedback <ref> --positive` or
|
|
1200
|
-
// retrieve assets to bring them into the eligible pool.
|
|
1201
|
-
// Layer-2 proactive refs join the eligible set alongside feedback-signal
|
|
1202
|
-
// refs. The three sources are disjoint by construction (proactive draws from
|
|
1203
|
-
// noFeedbackCandidates, and high-salience draws from the remainder), but
|
|
1204
|
-
// dedupe defensively so a ref can never enter the loop twice.
|
|
1205
|
-
// `requireFeedbackSignal` still suppresses all fallback sources for callers
|
|
1206
|
-
// that want feedback-only runs.
|
|
1207
|
-
const signalAndRetrievalRefs = dedupeRefs([...signalFiltered, ...proactiveRefs, ...highSalienceRefs]);
|
|
1208
|
-
const mergedRefs = scope.mode === "ref" ? processableRefs : options.requireFeedbackSignal ? signalFiltered : signalAndRetrievalRefs;
|
|
580
|
+
const ledger = stashDir
|
|
581
|
+
? loadLedgerSnapshot({ eventsCtx, ...(args.readOnly ? { readOnly: true } : {}) }, stashDir, ["reflect", "distill"])
|
|
582
|
+
: new Map();
|
|
1209
583
|
return {
|
|
1210
|
-
|
|
1211
|
-
|
|
1212
|
-
|
|
1213
|
-
|
|
1214
|
-
|
|
1215
|
-
|
|
1216
|
-
|
|
1217
|
-
proactiveRefs,
|
|
1218
|
-
proactiveMaintenanceSummary,
|
|
1219
|
-
proactivePlan: proactive.proactivePlan,
|
|
1220
|
-
highSalienceRefs,
|
|
1221
|
-
signalAndRetrievalRefs,
|
|
1222
|
-
mergedRefs,
|
|
1223
|
-
processableRefs,
|
|
584
|
+
feedbackSinceCutoff,
|
|
585
|
+
nowIso: new Date().toISOString(),
|
|
586
|
+
latestFeedbackTs,
|
|
587
|
+
ledger,
|
|
588
|
+
lastReflectAttemptAt: lastAttemptByRef(ledger, "reflect", candidates),
|
|
589
|
+
lastDistillAttemptAt: lastAttemptByRef(ledger, "distill", candidates),
|
|
590
|
+
feedback,
|
|
1224
591
|
};
|
|
1225
592
|
}
|
|
1226
593
|
/**
|
|
1227
|
-
*
|
|
1228
|
-
*
|
|
1229
|
-
*
|
|
1230
|
-
*
|
|
1231
|
-
*
|
|
1232
|
-
*
|
|
1233
|
-
*
|
|
1234
|
-
* the sort, disk-check, and loop stages down to the reflect/distill event
|
|
1235
|
-
* emit sites and createProposal calls. See EligibilitySource for the lane
|
|
1236
|
-
* vocabulary.
|
|
1237
|
-
*
|
|
1238
|
-
* Precedence (prefer the most specific reactive signal):
|
|
1239
|
-
* scope > signal-delta > proactive > high-salience
|
|
1240
|
-
* A ref with real feedback is attributed to feedback even if it was also due
|
|
1241
|
-
* for proactive maintenance or had high encoding salience. We apply lanes
|
|
1242
|
-
* weakest-first so the strongest overwrites; the explicit --scope <ref> bypass
|
|
1243
|
-
* wins outright (user intent).
|
|
1244
|
-
*/
|
|
1245
|
-
function stampEligibilitySource(args) {
|
|
1246
|
-
const { scope, processableRefs, mergedRefs, signalFiltered, proactiveRefs, highSalienceRefs } = args;
|
|
1247
|
-
const eligibilitySourceByRef = new Map();
|
|
1248
|
-
for (const r of highSalienceRefs)
|
|
1249
|
-
eligibilitySourceByRef.set(r.ref, "high-salience");
|
|
1250
|
-
for (const r of proactiveRefs)
|
|
1251
|
-
eligibilitySourceByRef.set(r.ref, "proactive");
|
|
1252
|
-
for (const r of signalFiltered)
|
|
1253
|
-
eligibilitySourceByRef.set(r.ref, "signal-delta");
|
|
1254
|
-
if (scope.mode === "ref") {
|
|
1255
|
-
// O-2 (#365): explicit --scope <ref> bypass — every ref in processableRefs
|
|
1256
|
-
// arrived via the scopeRefBypass branch, so attribute the whole set to scope.
|
|
1257
|
-
for (const r of processableRefs)
|
|
1258
|
-
eligibilitySourceByRef.set(r.ref, "scope");
|
|
1259
|
-
}
|
|
1260
|
-
for (const r of mergedRefs) {
|
|
1261
|
-
// "unknown" is a genuine fallback, never a silent alias for signal-delta:
|
|
1262
|
-
// only refs we truly cannot attribute land here (none in practice, since
|
|
1263
|
-
// mergedRefs is always a subset of the four lanes above).
|
|
1264
|
-
r.eligibilitySource = eligibilitySourceByRef.get(r.ref) ?? "unknown";
|
|
1265
|
-
}
|
|
1266
|
-
return eligibilitySourceByRef;
|
|
1267
|
-
}
|
|
1268
|
-
/**
|
|
1269
|
-
* The signal-delta partition of postCleanupRefs into the four buckets (pass:
|
|
1270
|
-
* candidate-gather, phase 3). The 2026-05-26 54-ref incident semantics move
|
|
1271
|
-
* VERBATIM — see the phase-2/3 comments inside.
|
|
594
|
+
* Partition the post-cleanup refs against the ledger:
|
|
595
|
+
* - eligibleRefs: reflect's signal delta passes (distill may still be cooled);
|
|
596
|
+
* - distillOnlyRefs: only distill's passes, on a distill candidate;
|
|
597
|
+
* - noFeedbackPool: no recent feedback and no reflect window, left to the
|
|
598
|
+
* fallback lanes;
|
|
599
|
+
* - fullySkippedCount: feedback on record but nothing new, or a live window.
|
|
600
|
+
* An explicit `--scope <ref>` bypasses every gate.
|
|
1272
601
|
*/
|
|
1273
602
|
export function partitionBySignalDelta(args) {
|
|
1274
|
-
const {
|
|
1275
|
-
const { latestFeedbackTs,
|
|
1276
|
-
//
|
|
1277
|
-
|
|
1278
|
-
|
|
1279
|
-
|
|
1280
|
-
|
|
1281
|
-
|
|
1282
|
-
|
|
1283
|
-
|
|
1284
|
-
|
|
1285
|
-
|
|
1286
|
-
|
|
1287
|
-
|
|
1288
|
-
|
|
1289
|
-
|
|
1290
|
-
|
|
1291
|
-
|
|
1292
|
-
// below so never-rated assets can still be improved.
|
|
1293
|
-
// Only refs those lanes decline are fully skipped.
|
|
1294
|
-
// fullySkippedCount — has stale feedback but no signal delta → genuine
|
|
1295
|
-
// skip candidate, excluded from sort. Final skip
|
|
1296
|
-
// observability is emitted only after fallbacks.
|
|
1297
|
-
const eligibleRefs = [];
|
|
1298
|
-
const distillOnlyRefs = [];
|
|
1299
|
-
// Zero-(recent-)feedback refs deferred to the proactive/high-salience fallbacks.
|
|
1300
|
-
const noFeedbackPool = [];
|
|
1301
|
-
let fullySkippedCount = 0;
|
|
1302
|
-
// O-2 (#365): explicit --scope <ref> bypasses every gate (user intent wins).
|
|
1303
|
-
const scopeRefBypass = scope.mode === "ref";
|
|
603
|
+
const { postCleanupRefs, validationFailureRefs } = args;
|
|
604
|
+
const { latestFeedbackTs, ledger, nowIso } = args.snapshot;
|
|
605
|
+
// Newer feedback lifts a revisit window, never a rejection.
|
|
606
|
+
const deltaPasses = (candidate, source) => {
|
|
607
|
+
const feedbackAt = latestFeedbackTs.get(candidate.ref);
|
|
608
|
+
if (!feedbackAt)
|
|
609
|
+
return false;
|
|
610
|
+
const row = ledgerRowFor(ledger, source, candidate.ref, candidate.itemRef);
|
|
611
|
+
return feedbackAt > (row?.lastAttemptAt ?? "") && !isLedgerBlocked(row, nowIso, feedbackAt);
|
|
612
|
+
};
|
|
613
|
+
const out = {
|
|
614
|
+
distillCooledRefs: new Set(),
|
|
615
|
+
preCooldownCount: postCleanupRefs.length,
|
|
616
|
+
eligibleRefs: [],
|
|
617
|
+
distillOnlyRefs: [],
|
|
618
|
+
noFeedbackPool: [],
|
|
619
|
+
fullySkippedCount: 0,
|
|
620
|
+
};
|
|
1304
621
|
for (const r of postCleanupRefs) {
|
|
1305
622
|
if (validationFailureRefs.has(r.ref))
|
|
1306
623
|
continue;
|
|
1307
|
-
if (
|
|
1308
|
-
eligibleRefs.push(r);
|
|
624
|
+
if (args.scope.mode === "ref") {
|
|
625
|
+
out.eligibleRefs.push(r);
|
|
1309
626
|
continue;
|
|
1310
627
|
}
|
|
1311
|
-
const reflectOk =
|
|
1312
|
-
const distillOk =
|
|
1313
|
-
const isDistillCandidate = isDistillCandidateRef(r.ref, options.stashDir);
|
|
628
|
+
const reflectOk = deltaPasses(r, "reflect");
|
|
629
|
+
const distillOk = deltaPasses(r, "distill");
|
|
1314
630
|
if (reflectOk) {
|
|
1315
|
-
if (!distillOk
|
|
1316
|
-
|
|
1317
|
-
|
|
1318
|
-
// has finalized the invocation's terminal skipped set.
|
|
1319
|
-
distillCooledRefs.add(r.ref);
|
|
1320
|
-
}
|
|
1321
|
-
else if (!distillOk) {
|
|
1322
|
-
// Not a distill candidate AND distill gate doesn't pass — just mark
|
|
1323
|
-
// distillCooled so the loop's distill section is a no-op.
|
|
1324
|
-
distillCooledRefs.add(r.ref);
|
|
1325
|
-
}
|
|
1326
|
-
eligibleRefs.push(r);
|
|
631
|
+
if (!distillOk)
|
|
632
|
+
out.distillCooledRefs.add(r.ref);
|
|
633
|
+
out.eligibleRefs.push(r);
|
|
1327
634
|
}
|
|
1328
|
-
else if (distillOk &&
|
|
1329
|
-
|
|
1330
|
-
distillOnlyRefs.push(r);
|
|
635
|
+
else if (distillOk && isDistillCandidateRef(r.ref, args.options.stashDir)) {
|
|
636
|
+
out.distillOnlyRefs.push(r);
|
|
1331
637
|
}
|
|
1332
|
-
else if (!latestFeedbackTs.has(r.ref)
|
|
1333
|
-
|
|
1334
|
-
|
|
1335
|
-
// and high-salience fallbacks below: a never-rated asset is exactly what
|
|
1336
|
-
// those lanes are meant to rescue. Refs those lanes decline are skipped there.
|
|
1337
|
-
noFeedbackPool.push(r);
|
|
638
|
+
else if (!latestFeedbackTs.has(r.ref) &&
|
|
639
|
+
!isLedgerBlocked(ledgerRowFor(ledger, "reflect", r.ref, r.itemRef), nowIso)) {
|
|
640
|
+
out.noFeedbackPool.push(r);
|
|
1338
641
|
}
|
|
1339
642
|
else {
|
|
1340
|
-
|
|
1341
|
-
// genuinely a fully-skipped candidate. Count it as partition metadata;
|
|
1342
|
-
// final observability waits until replay and every other fallback lane
|
|
1343
|
-
// has had a chance to rescue it.
|
|
1344
|
-
fullySkippedCount++;
|
|
643
|
+
out.fullySkippedCount++;
|
|
1345
644
|
}
|
|
1346
645
|
}
|
|
1347
|
-
return
|
|
1348
|
-
distillCooledRefs,
|
|
1349
|
-
preCooldownCount,
|
|
1350
|
-
eligibleRefs,
|
|
1351
|
-
distillOnlyRefs,
|
|
1352
|
-
noFeedbackPool,
|
|
1353
|
-
fullySkippedCount,
|
|
1354
|
-
};
|
|
646
|
+
return out;
|
|
1355
647
|
}
|
|
1356
648
|
/**
|
|
1357
|
-
*
|
|
1358
|
-
*
|
|
1359
|
-
*
|
|
1360
|
-
* terminally skipped work.
|
|
649
|
+
* Pick the loop's refs: signal delta, the fallback lanes (unless
|
|
650
|
+
* `--require-feedback-signal`), lane attribution, salience and forgetting
|
|
651
|
+
* safety, the no-op-dampened ranking, the disk check and the limit.
|
|
1361
652
|
*/
|
|
1362
|
-
function
|
|
1363
|
-
const {
|
|
1364
|
-
|
|
653
|
+
async function selectLoopCandidates(args, postCleanupRefs, validationFailureRefs, actions, persist) {
|
|
654
|
+
const { scope, options, primaryStashDir, eventsCtx, improveProfile } = args;
|
|
655
|
+
const snapshot = buildSnapshotManifest({
|
|
656
|
+
postCleanupRefs,
|
|
657
|
+
validationFailureRefs,
|
|
658
|
+
eventsCtx,
|
|
659
|
+
stashDir: primaryStashDir ?? options.stashDir,
|
|
660
|
+
readOnly: !persist,
|
|
661
|
+
});
|
|
662
|
+
const partition = partitionBySignalDelta({ scope, options, postCleanupRefs, validationFailureRefs, snapshot });
|
|
663
|
+
const processableRefs = [...partition.eligibleRefs, ...partition.distillOnlyRefs];
|
|
664
|
+
const signalFiltered = processableRefs.filter((c) => snapshot.feedback.get(c.ref)?.hasSignal === true);
|
|
665
|
+
const signalBearingSet = new Set(signalFiltered.map((r) => r.ref));
|
|
666
|
+
const noFeedbackCandidates = dedupeRefs([
|
|
667
|
+
...processableRefs.filter((r) => !signalBearingSet.has(r.ref)),
|
|
668
|
+
...partition.noFeedbackPool,
|
|
669
|
+
]);
|
|
670
|
+
const retrieval = fetchRetrievalSignals(options, signalFiltered, noFeedbackCandidates, eventsCtx, persist);
|
|
671
|
+
const allowFallbacks = options.requireFeedbackSignal !== true;
|
|
672
|
+
const proactive = allowFallbacks
|
|
673
|
+
? selectProactiveMaintenanceLane(args, noFeedbackCandidates, snapshot, retrieval, persist)
|
|
674
|
+
: { proactiveRefs: [] };
|
|
675
|
+
const highSalienceRefs = allowFallbacks
|
|
676
|
+
? selectHighSalienceLane(options, improveProfile, eventsCtx, noFeedbackCandidates.filter((r) => !proactive.proactiveRefs.some((p) => p.ref === r.ref)), snapshot.lastReflectAttemptAt, persist)
|
|
677
|
+
: [];
|
|
678
|
+
// An explicit ref scope always acts on its ref; otherwise usage signals gate the pool.
|
|
679
|
+
const signalAndRetrievalRefs = dedupeRefs([...signalFiltered, ...proactive.proactiveRefs, ...highSalienceRefs]);
|
|
680
|
+
let mergedRefs = scope.mode === "ref" ? processableRefs : options.requireFeedbackSignal ? signalFiltered : signalAndRetrievalRefs;
|
|
681
|
+
// Lane attribution, weakest first so the strongest wins: high-salience <
|
|
682
|
+
// proactive < signal-delta, and an explicit ref scope over everything.
|
|
683
|
+
const sourceByRef = new Map();
|
|
684
|
+
for (const r of highSalienceRefs)
|
|
685
|
+
sourceByRef.set(r.ref, "high-salience");
|
|
686
|
+
for (const r of proactive.proactiveRefs)
|
|
687
|
+
sourceByRef.set(r.ref, "proactive");
|
|
688
|
+
for (const r of signalFiltered)
|
|
689
|
+
sourceByRef.set(r.ref, "signal-delta");
|
|
690
|
+
if (scope.mode === "ref")
|
|
691
|
+
for (const r of processableRefs)
|
|
692
|
+
sourceByRef.set(r.ref, "scope");
|
|
693
|
+
for (const r of mergedRefs)
|
|
694
|
+
r.eligibilitySource = sourceByRef.get(r.ref) ?? "unknown";
|
|
695
|
+
// Forgetting safety may only reuse this plan's own surviving objects, and
|
|
696
|
+
// never a ref whose reflect window is still open.
|
|
697
|
+
const fallbackEligible = postCleanupRefs.filter((c) => !validationFailureRefs.has(c.ref));
|
|
698
|
+
const forgettingEligible = fallbackEligible.filter((c) => !isLedgerBlocked(ledgerRowFor(snapshot.ledger, "reflect", c.ref, c.itemRef), snapshot.nowIso));
|
|
699
|
+
const scored = scoreSalience(args, mergedRefs, snapshot.feedback, retrieval.retrievalCounts, persist);
|
|
700
|
+
mergedRefs = applyForgettingSafety({
|
|
701
|
+
pendingForgettingRefs: scored.pendingForgettingRefs,
|
|
702
|
+
scope,
|
|
703
|
+
mergedRefs,
|
|
704
|
+
eligibleRefs: forgettingEligible,
|
|
705
|
+
allowFallbacks,
|
|
706
|
+
eligibilitySourceByRef: sourceByRef,
|
|
707
|
+
highSalienceRefs,
|
|
708
|
+
proactiveRefs: proactive.proactiveRefs,
|
|
709
|
+
signalFiltered,
|
|
710
|
+
});
|
|
711
|
+
// Rank by salience; a ref skipped as a no-op repeatedly sorts lower (its stored rank is untouched).
|
|
712
|
+
const noOps = new Map();
|
|
713
|
+
withRunState(eventsCtx, persist, (db) => {
|
|
714
|
+
for (const r of mergedRefs)
|
|
715
|
+
noOps.set(r.ref, getAssetSalience(db, keyOf(r))?.consecutive_no_ops ?? 0);
|
|
716
|
+
});
|
|
717
|
+
const effectiveScore = (ref) => {
|
|
718
|
+
const rank = scored.salienceMap.get(ref)?.rankScore ?? 0;
|
|
719
|
+
return (noOps.get(ref) ?? 0) >= SALIENCE_NO_OP_DAMPEN_THRESHOLD ? rank * SALIENCE_NO_OP_DAMPEN_FACTOR : rank;
|
|
720
|
+
};
|
|
721
|
+
const sorted = [...mergedRefs].sort((a, b) => effectiveScore(b.ref) - effectiveScore(a.ref) || (a.ref < b.ref ? -1 : a.ref > b.ref ? 1 : 0));
|
|
722
|
+
const coverageGaps = withIndexDb(!persist, getZeroResultSearches) ?? [];
|
|
723
|
+
const { actionableRefs, missing } = await dropRefsMissingOnDisk(sorted, options, eventsCtx, persist);
|
|
724
|
+
const selection = selectEffectiveImproveRefs({
|
|
725
|
+
rankedRefs: actionableRefs,
|
|
726
|
+
distillOnlyRefs: partition.distillOnlyRefs,
|
|
727
|
+
limit: options.limit,
|
|
728
|
+
});
|
|
729
|
+
if (signalAndRetrievalRefs.length > 0) {
|
|
730
|
+
info(`[improve] ${signalAndRetrievalRefs.length} refs with usage signals (${signalFiltered.length} feedback)`);
|
|
731
|
+
}
|
|
732
|
+
if (validationFailureRefs.size > 0)
|
|
733
|
+
info(`[improve] ${validationFailureRefs.size} with validation failures excluded`);
|
|
734
|
+
if (persist && missing.length > 0)
|
|
735
|
+
info(`[improve] ${missing.length} candidates dropped — file not on disk`);
|
|
736
|
+
const deferred = actionableRefs.length - selection.loopRefs.length;
|
|
737
|
+
info(`[improve] ${actionableRefs.length} actionable; ${selection.loopRefs.length} will be processed` +
|
|
738
|
+
(options.limit && deferred > 0 ? ` (--limit ${options.limit} applied; ${deferred} deferred)` : ""));
|
|
739
|
+
// Skip observability waits until every fallback lane has finalized the
|
|
740
|
+
// survivors, so a rescued ref is never also reported skipped.
|
|
741
|
+
const survivors = new Set(sorted.map((c) => c.ref));
|
|
742
|
+
const signalSkipped = fallbackEligible.filter((c) => !survivors.has(c.ref));
|
|
743
|
+
for (const ref of partition.distillCooledRefs) {
|
|
1365
744
|
actions.push({ ref, mode: "distill-skipped", result: { ok: true, reason: "distill signal-delta" } });
|
|
1366
|
-
if (persist)
|
|
1367
|
-
|
|
1368
|
-
eventType: "improve_skipped",
|
|
1369
|
-
ref,
|
|
1370
|
-
metadata: { reason: "distill_no_new_signal" },
|
|
1371
|
-
}, eventsCtx);
|
|
1372
|
-
}
|
|
745
|
+
if (persist)
|
|
746
|
+
recordImproveSkip(eventsCtx, ref, { reason: "distill_no_new_signal" });
|
|
1373
747
|
}
|
|
1374
|
-
for (const candidate of
|
|
748
|
+
for (const candidate of signalSkipped) {
|
|
1375
749
|
actions.push({
|
|
1376
750
|
ref: candidate.ref,
|
|
1377
751
|
mode: "distill-skipped",
|
|
1378
752
|
result: { ok: true, reason: "no new signal since last proposal" },
|
|
1379
753
|
});
|
|
1380
754
|
}
|
|
1381
|
-
|
|
1382
|
-
|
|
1383
|
-
|
|
1384
|
-
|
|
1385
|
-
|
|
1386
|
-
|
|
1387
|
-
|
|
1388
|
-
reason: "no_new_signal",
|
|
1389
|
-
count: terminalSignalSkippedRefs.length,
|
|
1390
|
-
},
|
|
1391
|
-
}, eventsCtx);
|
|
755
|
+
if (persist && signalSkipped.length > 0) {
|
|
756
|
+
recordImproveSkip(eventsCtx, undefined, { reason: "no_new_signal", count: signalSkipped.length });
|
|
757
|
+
}
|
|
758
|
+
const blocked = signalSkipped.length + partition.distillOnlyRefs.length;
|
|
759
|
+
if (blocked > 0) {
|
|
760
|
+
info(`[improve] ${blocked} of ${partition.preCooldownCount} indexed refs blocked by reflect signal-delta ` +
|
|
761
|
+
`(${signalSkipped.length} fully skipped, ${partition.distillOnlyRefs.length} routed to distill-only)`);
|
|
1392
762
|
}
|
|
763
|
+
const gates = [
|
|
764
|
+
{
|
|
765
|
+
name: "signal",
|
|
766
|
+
removed: signalSkipped.length,
|
|
767
|
+
reason: "no fresh signal since the last attempt (or an improve-ledger window) and no fallback lane selected the ref",
|
|
768
|
+
},
|
|
769
|
+
{ name: "disk", removed: missing.length, reason: "backing asset is absent on disk" },
|
|
770
|
+
{ name: "limit", removed: selection.limitRemoved, reason: "deferred by the effective run limit" },
|
|
771
|
+
];
|
|
772
|
+
return {
|
|
773
|
+
actionableRefs,
|
|
774
|
+
loopRefs: selection.loopRefs,
|
|
775
|
+
distillOnlyRefs: selection.distillOnlyRefs,
|
|
776
|
+
distillCooledRefs: partition.distillCooledRefs,
|
|
777
|
+
signalBearingSet,
|
|
778
|
+
coverageGaps,
|
|
779
|
+
gates,
|
|
780
|
+
proactive,
|
|
781
|
+
};
|
|
1393
782
|
}
|
|
1394
|
-
/**
|
|
1395
|
-
function
|
|
1396
|
-
const {
|
|
1397
|
-
|
|
1398
|
-
|
|
1399
|
-
|
|
1400
|
-
|
|
1401
|
-
|
|
1402
|
-
// Phase 2 above for the signal-delta gate; we reuse them here.)
|
|
1403
|
-
// Pre-compute feedback summary per ref in a SINGLE bulk read so we don't
|
|
1404
|
-
// open state.db once per asset (which caused 5000+ accumulated FDs and a
|
|
1405
|
-
// 2-hour runaway on a 13K-asset stash). Pattern mirrors buildLatestFeedbackTsMap
|
|
1406
|
-
// above: one readEvents() call fetches ALL feedback events, then we aggregate
|
|
1407
|
-
// in-memory by ref — O(1) DB opens regardless of candidate set size.
|
|
1408
|
-
// Cover processableRefs *and* the deferred noFeedbackPool so utility/feedback
|
|
1409
|
-
// ratios are available for any noFeedbackPool ref the fallback lanes rescue below.
|
|
1410
|
-
//
|
|
1411
|
-
// Behavioral note: positive/negative COUNTS are all-time (same as the old
|
|
1412
|
-
// per-ref readEvents call which had no `since` filter); hasSignal is bounded
|
|
1413
|
-
// to feedbackSinceCutoff (same as the old inline `(e.ts ?? "") >= cutoff` guard).
|
|
1414
|
-
const feedbackSummary = new Map();
|
|
1415
|
-
{
|
|
1416
|
-
const feedbackCandidates = [...processableRefs, ...noFeedbackPool];
|
|
1417
|
-
const feedbackCandidateSet = new Set(feedbackCandidates.map((r) => r.ref));
|
|
1418
|
-
// Map each candidate's single durable event key back to its display ref.
|
|
1419
|
-
const feedbackRefByDurableKey = new Map(feedbackCandidates.flatMap((r) => improveStateReadRefs(r.ref, r.itemRef).map((key) => [key, r.ref])));
|
|
1420
|
-
if (feedbackCandidateSet.size > 0) {
|
|
1421
|
-
// Fetch ALL feedback events in one query (no ref filter, no since filter =
|
|
1422
|
-
// single full table scan). Filtering per-ref in memory avoids N sequential
|
|
1423
|
-
// state.db opens — the dominant FD-leak path on large stashes.
|
|
1424
|
-
const { events: allFeedbackEvents } = readEvents({ type: "feedback" }, eventsCtx);
|
|
1425
|
-
for (const e of allFeedbackEvents) {
|
|
1426
|
-
const ref = e.ref ? feedbackRefByDurableKey.get(e.ref) : undefined;
|
|
1427
|
-
if (!ref)
|
|
1428
|
-
continue;
|
|
1429
|
-
const entry = feedbackSummary.get(ref) ?? { hasSignal: false, positive: 0, negative: 0 };
|
|
1430
|
-
const meta = e.metadata;
|
|
1431
|
-
// hasSignal: only count feedback events within the 30-day window.
|
|
1432
|
-
if (!entry.hasSignal &&
|
|
1433
|
-
(e.ts ?? "") >= feedbackSinceCutoff &&
|
|
1434
|
-
meta !== undefined &&
|
|
1435
|
-
(typeof meta.signal === "string" || typeof meta.note === "string")) {
|
|
1436
|
-
entry.hasSignal = true;
|
|
1437
|
-
}
|
|
1438
|
-
// positive/negative: all-time counts (no since filter, matching prior behaviour).
|
|
1439
|
-
if (meta?.signal === "positive")
|
|
1440
|
-
entry.positive++;
|
|
1441
|
-
else if (meta?.signal === "negative")
|
|
1442
|
-
entry.negative++;
|
|
1443
|
-
feedbackSummary.set(ref, entry);
|
|
1444
|
-
}
|
|
1445
|
-
// Ensure every candidate has an entry (even refs with zero feedback events).
|
|
1446
|
-
for (const ref of feedbackCandidateSet) {
|
|
1447
|
-
if (!feedbackSummary.has(ref)) {
|
|
1448
|
-
feedbackSummary.set(ref, { hasSignal: false, positive: 0, negative: 0 });
|
|
1449
|
-
}
|
|
783
|
+
/** Retrieval counts for every candidate, and last-use times for the zero-feedback pool. */
|
|
784
|
+
function fetchRetrievalSignals(options, signalFiltered, noFeedbackCandidates, eventsCtx, persist) {
|
|
785
|
+
const out = { retrievalCounts: new Map(), lastUseMs: new Map() };
|
|
786
|
+
withIndexDb(!persist, (indexDb) => {
|
|
787
|
+
// usage_events live in state.db, entries in index.db.
|
|
788
|
+
withRunState(eventsCtx, persist, (stateDb) => {
|
|
789
|
+
if (countUsageEventsByType(stateDb, "show") === 0) {
|
|
790
|
+
warn("Warning: show events not yet in usage_events — zero-feedback fallback will match only search-retrieved assets.");
|
|
1450
791
|
}
|
|
1451
|
-
|
|
1452
|
-
|
|
1453
|
-
|
|
792
|
+
const refs = [...new Set([...signalFiltered, ...noFeedbackCandidates].map((r) => r.ref))];
|
|
793
|
+
out.retrievalCounts = getRetrievalCounts(indexDb, stateDb, refs, { sourceName: options.sourceName });
|
|
794
|
+
});
|
|
795
|
+
out.lastUseMs = getLastUseMsByRef(indexDb, noFeedbackCandidates);
|
|
796
|
+
});
|
|
797
|
+
return out;
|
|
1454
798
|
}
|
|
1455
|
-
/**
|
|
1456
|
-
|
|
1457
|
-
|
|
1458
|
-
|
|
1459
|
-
|
|
1460
|
-
|
|
1461
|
-
|
|
1462
|
-
|
|
1463
|
-
let lastUseMsForProactive = new Map();
|
|
1464
|
-
let dbForRetrieval;
|
|
1465
|
-
try {
|
|
1466
|
-
dbForRetrieval = persist
|
|
1467
|
-
? openExistingDatabase()
|
|
1468
|
-
: openReadonlyExistingDatabase(undefined, { isolatedSnapshot: true });
|
|
1469
|
-
if (!dbForRetrieval)
|
|
1470
|
-
return { retrievalCounts, lastUseMsForProactive };
|
|
1471
|
-
// usage_events lives in state.db (Chunk-8 WI-8.3); entries stay in index.db,
|
|
1472
|
-
// so the retrieval-count reads take both handles.
|
|
1473
|
-
const dbForRetrievalIndex = dbForRetrieval;
|
|
1474
|
-
if (persist || eventsCtx?.db) {
|
|
1475
|
-
withStateDb((stateDb) => {
|
|
1476
|
-
const showEventCount = countUsageEventsByType(stateDb, "show");
|
|
1477
|
-
if (showEventCount === 0) {
|
|
1478
|
-
warn("Warning: show events not yet in usage_events — zero-feedback fallback will match only search-retrieved assets.");
|
|
1479
|
-
}
|
|
1480
|
-
// Fetch retrieval counts for ALL candidates — not only the zero-feedback pool.
|
|
1481
|
-
// Previously only noFeedbackCandidates were looked up, so feedback-bearing refs
|
|
1482
|
-
// had retrievalFreq=0 in computeSalience(), collapsing their retrievalSalience
|
|
1483
|
-
// to 0 regardless of actual use. Two assets of the same type — one
|
|
1484
|
-
// heavily-retrieved, one never-touched — would receive identical rankScores.
|
|
1485
|
-
// Fix (WS-1 blocker 3): union the feedback pool into the lookup.
|
|
1486
|
-
const allCandidateRefs = [...new Set([...signalFiltered, ...noFeedbackCandidates].map((r) => r.ref))];
|
|
1487
|
-
retrievalCounts = getRetrievalCounts(dbForRetrievalIndex, stateDb, allCandidateRefs, {
|
|
1488
|
-
sourceName: options.sourceName,
|
|
1489
|
-
});
|
|
1490
|
-
}, { path: eventsCtx?.dbPath, borrowed: eventsCtx?.db });
|
|
1491
|
-
}
|
|
1492
|
-
lastUseMsForProactive = getLastUseMsByRef(dbForRetrieval, noFeedbackCandidates);
|
|
799
|
+
/**
|
|
800
|
+
* Proactive maintenance (default off, whole-stash/type runs): revisit stable
|
|
801
|
+
* assets on a schedule. The due gate doubles as the rotation cooldown: a
|
|
802
|
+
* freshly reflected asset waits `dueDays` before it is picked again.
|
|
803
|
+
*/
|
|
804
|
+
function selectProactiveMaintenanceLane(args, candidates, snapshot, retrieval, persist) {
|
|
805
|
+
if (args.scope.mode === "ref" || !args.resolvedPlan.processes.proactiveMaintenance.enabled) {
|
|
806
|
+
return { proactiveRefs: [] };
|
|
1493
807
|
}
|
|
1494
|
-
|
|
1495
|
-
|
|
1496
|
-
|
|
808
|
+
const pmCfg = args.improveProfile.processes?.proactiveMaintenance;
|
|
809
|
+
const dueDays = pmCfg?.dueDays ?? DEFAULT_DUE_DAYS;
|
|
810
|
+
const maxPerRun = pmCfg?.maxPerRun ?? pmCfg?.limit ?? DEFAULT_MAX_PER_RUN;
|
|
811
|
+
const selection = selectProactiveMaintenanceRefs({
|
|
812
|
+
candidates,
|
|
813
|
+
lastReflectTs: snapshot.lastReflectAttemptAt,
|
|
814
|
+
lastDistillTs: snapshot.lastDistillAttemptAt,
|
|
815
|
+
retrievalCounts: retrieval.retrievalCounts,
|
|
816
|
+
lastUseMs: retrieval.lastUseMs,
|
|
817
|
+
sizeBytesOf: (r) => fileSize(r.filePath),
|
|
818
|
+
dueDays,
|
|
819
|
+
maxPerRun,
|
|
820
|
+
});
|
|
821
|
+
const summary = {
|
|
822
|
+
selected: selection.selected.length,
|
|
823
|
+
dueTotal: selection.dueTotal,
|
|
824
|
+
neverReflected: selection.neverReflected,
|
|
825
|
+
};
|
|
826
|
+
if (persist) {
|
|
827
|
+
appendEvent({
|
|
828
|
+
eventType: "proactive_selected",
|
|
829
|
+
ref: undefined,
|
|
830
|
+
metadata: { count: summary.selected, dueTotal: summary.dueTotal, neverReflected: summary.neverReflected },
|
|
831
|
+
}, args.eventsCtx);
|
|
1497
832
|
}
|
|
1498
|
-
|
|
1499
|
-
|
|
1500
|
-
|
|
833
|
+
if (summary.selected > 0) {
|
|
834
|
+
info(`[improve] proactive maintenance selected ${summary.selected}/${summary.dueTotal} due refs ` +
|
|
835
|
+
`(${summary.neverReflected} never reflected, dueDays=${dueDays}, maxPerRun=${maxPerRun})`);
|
|
1501
836
|
}
|
|
1502
|
-
|
|
1503
|
-
|
|
1504
|
-
|
|
1505
|
-
|
|
1506
|
-
|
|
1507
|
-
|
|
1508
|
-
// The signal-delta gate only surfaces assets with fresh feedback. It never
|
|
1509
|
-
// revisits a stable, high-value asset on a schedule, so on a quiet stash
|
|
1510
|
-
// useful assets drift stale and are never refreshed. When the
|
|
1511
|
-
// `proactiveMaintenance` process is enabled (DEFAULT OFF)
|
|
1512
|
-
// and the run is whole-stash / type scope, this selector ranks the eligible
|
|
1513
|
-
// population by a composite maintenance priority, gates on staleness ("due"),
|
|
1514
|
-
// bounds to top-N, and folds the winners into the SAME candidate set the other
|
|
1515
|
-
// sources feed — so they flow through the existing #580 empty-diff /
|
|
1516
|
-
// cosmetic suppression and additive-distill gates. It adds no new mutation
|
|
1517
|
-
// logic of its own. The due gate doubles as the rotation cooldown: a freshly
|
|
1518
|
-
// reflected asset is excluded until it ages back past `dueDays`, so successive
|
|
1519
|
-
// runs rotate through the due pool rather than re-selecting the same heads.
|
|
1520
|
-
let proactiveRefs = [];
|
|
1521
|
-
let proactiveMaintenanceSummary;
|
|
1522
|
-
let proactivePlan;
|
|
1523
|
-
const proactiveEnabled = scope.mode !== "ref" && resolvedPlan.processes.proactiveMaintenance.enabled;
|
|
1524
|
-
if (proactiveEnabled) {
|
|
1525
|
-
const pmCfg = improveProfile.processes?.proactiveMaintenance;
|
|
1526
|
-
const dueDays = pmCfg?.dueDays ?? DEFAULT_DUE_DAYS;
|
|
1527
|
-
const maxPerRun = pmCfg?.maxPerRun ?? pmCfg?.limit ?? DEFAULT_MAX_PER_RUN;
|
|
1528
|
-
// Candidate population: the zero-feedback / non-signal pool — exactly the
|
|
1529
|
-
// assets the signal-delta gate would NOT pick this run.
|
|
1530
|
-
const pmCandidates = noFeedbackCandidates;
|
|
1531
|
-
const selection = selectProactiveMaintenanceRefs({
|
|
1532
|
-
candidates: pmCandidates,
|
|
1533
|
-
lastReflectTs: lastReflectProposalTs,
|
|
1534
|
-
lastDistillTs: lastDistillProposalTs,
|
|
1535
|
-
retrievalCounts,
|
|
1536
|
-
// WS-1: wire lastUseMs so the recency decay term is genuine (plan §step 2).
|
|
1537
|
-
lastUseMs: lastUseMsForProactive,
|
|
1538
|
-
sizeBytesOf: (r) => {
|
|
1539
|
-
const fp = r.filePath;
|
|
1540
|
-
if (!fp)
|
|
1541
|
-
return undefined;
|
|
1542
|
-
try {
|
|
1543
|
-
return fs.statSync(fp).size;
|
|
1544
|
-
}
|
|
1545
|
-
catch {
|
|
1546
|
-
return undefined;
|
|
1547
|
-
}
|
|
1548
|
-
},
|
|
1549
|
-
dueDays,
|
|
1550
|
-
maxPerRun,
|
|
1551
|
-
});
|
|
1552
|
-
proactiveRefs = selection.selected;
|
|
1553
|
-
proactiveMaintenanceSummary = {
|
|
1554
|
-
selected: selection.selected.length,
|
|
1555
|
-
dueTotal: selection.dueTotal,
|
|
1556
|
-
neverReflected: selection.neverReflected,
|
|
1557
|
-
selectedRefs: selection.selected.map((entry) => entry.ref),
|
|
1558
|
-
};
|
|
1559
|
-
proactivePlan = {
|
|
1560
|
-
configured: {
|
|
1561
|
-
...(pmCfg?.dueDays !== undefined ? { dueDays: pmCfg.dueDays } : {}),
|
|
1562
|
-
...(pmCfg?.maxPerRun !== undefined ? { maxPerRun: pmCfg.maxPerRun } : {}),
|
|
1563
|
-
...(pmCfg?.limit !== undefined ? { limit: pmCfg.limit } : {}),
|
|
1564
|
-
},
|
|
837
|
+
const selectedRefs = selection.selected.map((entry) => entry.ref);
|
|
838
|
+
return {
|
|
839
|
+
proactiveRefs: selection.selected,
|
|
840
|
+
proactiveMaintenanceSummary: { ...summary, selectedRefs },
|
|
841
|
+
proactivePlan: {
|
|
842
|
+
configured: pickDefined(pmCfg, ["dueDays", "maxPerRun", "limit"]),
|
|
1565
843
|
effective: { dueDays, maxPerRun },
|
|
1566
|
-
candidatePool:
|
|
1567
|
-
|
|
1568
|
-
|
|
1569
|
-
|
|
1570
|
-
|
|
1571
|
-
};
|
|
1572
|
-
// Aggregated observability event (never per-ref — avoids the event flood the
|
|
1573
|
-
// Layer-1 work eliminated). Mirrors the `no_new_signal` aggregation pattern.
|
|
1574
|
-
if (persist) {
|
|
1575
|
-
appendEvent({
|
|
1576
|
-
eventType: "proactive_selected",
|
|
1577
|
-
ref: undefined,
|
|
1578
|
-
metadata: {
|
|
1579
|
-
count: selection.selected.length,
|
|
1580
|
-
dueTotal: selection.dueTotal,
|
|
1581
|
-
neverReflected: selection.neverReflected,
|
|
1582
|
-
},
|
|
1583
|
-
}, eventsCtx);
|
|
1584
|
-
}
|
|
1585
|
-
if (selection.selected.length > 0) {
|
|
1586
|
-
info(`[improve] proactive maintenance selected ${selection.selected.length}/${selection.dueTotal} due refs ` +
|
|
1587
|
-
`(${selection.neverReflected} never reflected, dueDays=${dueDays}, maxPerRun=${maxPerRun})`);
|
|
1588
|
-
}
|
|
1589
|
-
}
|
|
1590
|
-
return { proactiveRefs, proactiveMaintenanceSummary, proactivePlan };
|
|
844
|
+
candidatePool: candidates.length,
|
|
845
|
+
...summary,
|
|
846
|
+
selectedRefs,
|
|
847
|
+
},
|
|
848
|
+
};
|
|
1591
849
|
}
|
|
1592
|
-
/**
|
|
1593
|
-
|
|
1594
|
-
|
|
1595
|
-
|
|
1596
|
-
|
|
1597
|
-
|
|
1598
|
-
|
|
1599
|
-
|
|
1600
|
-
|
|
1601
|
-
|
|
1602
|
-
|
|
1603
|
-
|
|
1604
|
-
|
|
1605
|
-
|
|
1606
|
-
|
|
1607
|
-
|
|
1608
|
-
|
|
1609
|
-
|
|
1610
|
-
|
|
1611
|
-
|
|
1612
|
-
|
|
1613
|
-
|
|
1614
|
-
|
|
1615
|
-
|
|
1616
|
-
// lesson 0.75 from DEFAULT_TYPE_ENCODING_WEIGHTS) for every distill-unscored
|
|
1617
|
-
// asset — i.e. "high-salience" degenerates into "is a skill/agent/command/
|
|
1618
|
-
// lesson", which selected the lore-writer type-stub agent on every run. Only
|
|
1619
|
-
// content-scored assets earn the high-salience rescue; type-stub rows must
|
|
1620
|
-
// earn retrieval/feedback signal via the other lanes. This PRESERVES #608's
|
|
1621
|
-
// intent — distilled assets (the lane's real targets) keep their real content
|
|
1622
|
-
// score and still qualify — while cutting the type-stub waste. See §5 F1 of
|
|
1623
|
-
// #608/#644.
|
|
1624
|
-
const highSalienceRefs = [];
|
|
1625
|
-
const salienceCfg = (options.config ?? loadConfig()).improve?.salience;
|
|
1626
|
-
const salienceThreshold = salienceCfg?.salienceThreshold ?? 0.75;
|
|
1627
|
-
const proactiveSelectedSet = new Set(proactiveRefs.map((r) => r.ref));
|
|
1628
|
-
try {
|
|
1629
|
-
if (!persist && !eventsCtx?.db)
|
|
1630
|
-
return highSalienceRefs;
|
|
1631
|
-
withStateDb((dbForHighSalience) => {
|
|
1632
|
-
// Derive the cap from the resolved reflect limit (mirrors improve.ts's
|
|
1633
|
-
// options.limit resolution) so an unbounded whole-stash run does not
|
|
1634
|
-
// collapse the lane to exactly 1 ref via the bare `?? 10` fallback.
|
|
1635
|
-
const effectiveLimit = options.limit ?? improveProfile?.processes?.reflect?.limit ?? improveProfile.limit ?? 10;
|
|
1636
|
-
const highSalienceCap = Math.max(1, Math.floor(effectiveLimit * 0.1));
|
|
1637
|
-
const candidates = noFeedbackCandidates.filter((r) => !proactiveSelectedSet.has(r.ref));
|
|
1638
|
-
// Collect ALL qualifying candidates, then take the top-N BY SCORE — the
|
|
1639
|
-
// previous first-N-in-scan-order break meant a higher-salience candidate
|
|
1640
|
-
// found later in the scan lost its slot to an earlier lower-scoring one.
|
|
1641
|
-
const qualifying = [];
|
|
1642
|
-
for (const r of candidates) {
|
|
1643
|
-
const row = readAssetSalienceForImproveRef(dbForHighSalience, r.ref, r.itemRef);
|
|
1644
|
-
if (row &&
|
|
1645
|
-
isContentEncodingRow(row) &&
|
|
1646
|
-
row.encoding_salience >= salienceThreshold &&
|
|
1647
|
-
!lastReflectProposalTs.has(r.ref)) {
|
|
1648
|
-
qualifying.push({ ref: r, score: row.encoding_salience });
|
|
1649
|
-
}
|
|
1650
|
-
}
|
|
1651
|
-
qualifying.sort((a, b) => b.score - a.score);
|
|
1652
|
-
for (const q of qualifying.slice(0, highSalienceCap)) {
|
|
1653
|
-
highSalienceRefs.push(q.ref);
|
|
1654
|
-
}
|
|
1655
|
-
}, { path: eventsCtx?.dbPath, borrowed: eventsCtx?.db });
|
|
1656
|
-
}
|
|
1657
|
-
catch (err) {
|
|
1658
|
-
rethrowIfTestIsolationError(err);
|
|
1659
|
-
// best-effort: if DB unavailable, highSalienceRefs stays empty
|
|
1660
|
-
}
|
|
1661
|
-
if (highSalienceRefs.length > 0) {
|
|
1662
|
-
info(`[improve] high-salience lane admitted ${highSalienceRefs.length} content-scored ref(s) ` +
|
|
1663
|
-
`(threshold=${salienceThreshold}, requires content-derived encoding_source)`);
|
|
850
|
+
/**
|
|
851
|
+
* High salience: zero-feedback refs whose content-derived encoding score (not
|
|
852
|
+
* a per-type stub) reaches `salienceThreshold` and that were never reflected,
|
|
853
|
+
* top-N by score, capped at 10% of the effective limit.
|
|
854
|
+
*/
|
|
855
|
+
function selectHighSalienceLane(options, improveProfile, eventsCtx, candidates, lastReflectAttemptAt, persist) {
|
|
856
|
+
const threshold = (options.config ?? loadConfig()).improve?.salience?.salienceThreshold ?? 0.75;
|
|
857
|
+
const effectiveLimit = options.limit ?? improveProfile?.processes?.reflect?.limit ?? improveProfile.limit ?? 10;
|
|
858
|
+
const selected = withRunState(eventsCtx, persist, (db) => candidates
|
|
859
|
+
.flatMap((r) => {
|
|
860
|
+
const row = getAssetSalience(db, keyOf(r));
|
|
861
|
+
return row &&
|
|
862
|
+
isContentEncodingRow(row) &&
|
|
863
|
+
row.encoding_salience >= threshold &&
|
|
864
|
+
!lastReflectAttemptAt.has(r.ref)
|
|
865
|
+
? [{ ref: r, score: row.encoding_salience }]
|
|
866
|
+
: [];
|
|
867
|
+
})
|
|
868
|
+
.sort((a, b) => b.score - a.score)
|
|
869
|
+
.slice(0, Math.max(1, Math.floor(effectiveLimit * 0.1)))
|
|
870
|
+
.map((q) => q.ref)) ?? [];
|
|
871
|
+
if (selected.length > 0) {
|
|
872
|
+
info(`[improve] high-salience lane admitted ${selected.length} content-scored ref(s) ` +
|
|
873
|
+
`(threshold=${threshold}, requires content-derived encoding_source)`);
|
|
1664
874
|
}
|
|
1665
|
-
return
|
|
875
|
+
return selected;
|
|
1666
876
|
}
|
|
1667
877
|
/**
|
|
1668
|
-
*
|
|
1669
|
-
*
|
|
1670
|
-
* and the
|
|
1671
|
-
*
|
|
1672
|
-
* Mutates the shared eligibilitySourceByRef map and ref objects in place —
|
|
1673
|
-
* attribution identity is load-bearing (see the candidate-gather comments).
|
|
878
|
+
* Score the merged refs: update `asset_outcome` (projected on a plan-only
|
|
879
|
+
* run), compute each salience vector (keeping a stored content-derived
|
|
880
|
+
* encoding score), then persist and compare the stash-wide ranking. A ref that
|
|
881
|
+
* falls from the top 200 to below 500 becomes a forgetting-safety candidate.
|
|
1674
882
|
*/
|
|
1675
|
-
function scoreSalience(args) {
|
|
1676
|
-
const {
|
|
1677
|
-
const mergedRefs = args.mergedRefs;
|
|
1678
|
-
// Chunk-5 flip F5e — resolve each candidate's durable item_ref ONCE for this
|
|
1679
|
-
// pass (the write/read key source for the outcome + salience state writers).
|
|
1680
|
-
const itemRefByRef = buildItemRefByRef(mergedRefs);
|
|
1681
|
-
// WS-1 — Unified salience vector (S1 seam).
|
|
1682
|
-
//
|
|
1683
|
-
// WS-1 converges utility, valence, and proactive-maintenance signals into one
|
|
1684
|
-
// `computeSalience()` call per ref, with
|
|
1685
|
-
// three independently-stored sub-scores and one documented rankScore projection.
|
|
1686
|
-
//
|
|
1687
|
-
// Fetch last-use timestamps from the index DB for the full merged set so the
|
|
1688
|
-
// recency term in retrievalSalience is genuinely decayable (plan §WS-1 step 2).
|
|
1689
|
-
// This reuses the index DB opened earlier for retrieval counts; a separate
|
|
1690
|
-
// lightweight open is used here to avoid holding the connection longer than needed.
|
|
1691
|
-
let lastUseMsByRef = new Map();
|
|
1692
|
-
// Health and outcome reporting consume the utility projection.
|
|
883
|
+
function scoreSalience(args, mergedRefs, feedback, retrievalCounts, persist) {
|
|
884
|
+
const { options, eventsCtx } = args;
|
|
1693
885
|
const utilityMap = buildUtilityMap(mergedRefs, !persist);
|
|
1694
|
-
|
|
1695
|
-
|
|
1696
|
-
dbForSalience = persist
|
|
1697
|
-
? openExistingDatabase()
|
|
1698
|
-
: openReadonlyExistingDatabase(undefined, { isolatedSnapshot: true });
|
|
1699
|
-
if (dbForSalience) {
|
|
1700
|
-
lastUseMsByRef = getLastUseMsByRef(dbForSalience, mergedRefs);
|
|
1701
|
-
}
|
|
1702
|
-
}
|
|
1703
|
-
catch (err) {
|
|
1704
|
-
rethrowIfTestIsolationError(err);
|
|
1705
|
-
// best-effort: if DB unavailable, recency term stays at floor (lastUseMs=0)
|
|
1706
|
-
}
|
|
1707
|
-
finally {
|
|
1708
|
-
if (dbForSalience)
|
|
1709
|
-
closeDatabase(dbForSalience);
|
|
1710
|
-
}
|
|
1711
|
-
const outcomeSalienceByRef = updateOutcomeScores({
|
|
1712
|
-
mergedRefs,
|
|
1713
|
-
itemRefByRef,
|
|
1714
|
-
feedbackSummary,
|
|
1715
|
-
retrievalCounts,
|
|
1716
|
-
lastUseMsByRef,
|
|
1717
|
-
utilityMap,
|
|
1718
|
-
primaryStashDir,
|
|
1719
|
-
eventsCtx,
|
|
1720
|
-
persist,
|
|
1721
|
-
});
|
|
1722
|
-
const { salienceMap, nowForSalience } = computeSalienceVectors({
|
|
886
|
+
const lastUseMsByRef = withIndexDb(!persist, (db) => getLastUseMsByRef(db, mergedRefs)) ?? new Map();
|
|
887
|
+
const outcomeSalience = updateOutcomeScores({
|
|
1723
888
|
mergedRefs,
|
|
1724
|
-
|
|
1725
|
-
options,
|
|
1726
|
-
eventsCtx,
|
|
889
|
+
feedback,
|
|
1727
890
|
retrievalCounts,
|
|
1728
891
|
lastUseMsByRef,
|
|
1729
892
|
utilityMap,
|
|
1730
|
-
|
|
1731
|
-
persist,
|
|
1732
|
-
});
|
|
1733
|
-
const pendingForgettingRefs = persistSalienceAndReportRanks({
|
|
1734
|
-
salienceMap,
|
|
1735
|
-
itemRefByRef,
|
|
1736
|
-
utilityMap,
|
|
1737
|
-
feedbackSummary,
|
|
1738
|
-
options,
|
|
893
|
+
primaryStashDir: args.primaryStashDir,
|
|
1739
894
|
eventsCtx,
|
|
1740
|
-
nowForSalience,
|
|
1741
895
|
persist,
|
|
1742
896
|
});
|
|
1743
|
-
const
|
|
1744
|
-
|
|
1745
|
-
|
|
1746
|
-
mergedRefs
|
|
1747
|
-
|
|
1748
|
-
|
|
1749
|
-
|
|
1750
|
-
highSalienceRefs,
|
|
1751
|
-
proactiveRefs,
|
|
1752
|
-
signalFiltered,
|
|
1753
|
-
});
|
|
1754
|
-
return { mergedRefs: finalMergedRefs, utilityMap, lastUseMsByRef, salienceMap, nowForSalience };
|
|
1755
|
-
}
|
|
1756
|
-
/** WS-2 — update asset_outcome for the merged set; returns outcomeSalience by ref. */
|
|
1757
|
-
function updateOutcomeScores(args) {
|
|
1758
|
-
const { mergedRefs, itemRefByRef, feedbackSummary, retrievalCounts, lastUseMsByRef, utilityMap, primaryStashDir, eventsCtx, persist, } = args;
|
|
1759
|
-
// ── WS-2 Outcome loop ─────────────────────────────────────────────────────
|
|
1760
|
-
//
|
|
1761
|
-
// Update asset_outcome for every ref in the merged set BEFORE computing the
|
|
1762
|
-
// salience vector so the updated outcome_score feeds outcomeSalience this run.
|
|
1763
|
-
//
|
|
1764
|
-
// Inputs per ref:
|
|
1765
|
-
// - currentRetrievalCount: from retrievalCounts (index DB)
|
|
1766
|
-
// - lastRetrievedAt: from lastUseMsByRef (utility_scores.last_used_at)
|
|
1767
|
-
// - negativeFeedbackCount: cumulative negatives from feedbackSummary
|
|
1768
|
-
// - acceptedChangeCount: accepted proposals for this ref (state.db)
|
|
1769
|
-
// - valence: net valence from computeValenceScore(feedbackSummary.get(ref))
|
|
1770
|
-
// - utilityScore: from utilityMap (for warm-start seed on new rows)
|
|
1771
|
-
//
|
|
1772
|
-
// Best-effort: outcome failures never block the salience or ranking pass.
|
|
1773
|
-
const outcomeSalienceByRef = new Map();
|
|
1774
|
-
// Missing state.db is itself a complete snapshot: no prior outcome rows and
|
|
1775
|
-
// no accepted proposals. Project the same warm-start values a live run would
|
|
1776
|
-
// insert, without creating the database merely to represent empty tables.
|
|
1777
|
-
if (!persist && !eventsCtx?.db) {
|
|
1778
|
-
const projectedScores = new Map();
|
|
1779
|
-
const nowForOutcome = Date.now();
|
|
1780
|
-
for (const ref of mergedRefs) {
|
|
1781
|
-
const feedback = feedbackSummary.get(ref.ref) ?? { positive: 0, negative: 0 };
|
|
1782
|
-
const result = projectAssetOutcome(undefined, {
|
|
1783
|
-
ref: outcomeWriteKey(ref.ref, itemRefByRef),
|
|
1784
|
-
currentRetrievalCount: retrievalCounts.get(ref.ref) ?? 0,
|
|
1785
|
-
lastRetrievedAt: lastUseMsByRef.get(ref.ref) ?? 0,
|
|
1786
|
-
acceptedChangeCount: 0,
|
|
1787
|
-
negativeFeedbackCount: feedback.negative,
|
|
1788
|
-
valence: computeValenceScore(feedback).valence,
|
|
1789
|
-
utilityScore: utilityMap.get(ref.ref),
|
|
1790
|
-
now: nowForOutcome,
|
|
1791
|
-
});
|
|
1792
|
-
projectedScores.set(ref.ref, result.outcomeScore);
|
|
1793
|
-
}
|
|
1794
|
-
const maxOutcomeScore = Math.min(OUTCOME_SCORE_MAX, Math.max(0, ...projectedScores.values()));
|
|
1795
|
-
for (const [ref, score] of projectedScores) {
|
|
1796
|
-
outcomeSalienceByRef.set(ref, outcomeScoreToSalience(score, maxOutcomeScore));
|
|
897
|
+
const outcomeWeightEnabled = (options.config ?? loadConfig()).improve?.salience?.outcomeWeightEnabled !== false;
|
|
898
|
+
const storedEncoding = new Map();
|
|
899
|
+
withRunState(eventsCtx, persist, (db) => {
|
|
900
|
+
for (const r of mergedRefs) {
|
|
901
|
+
const row = getAssetSalience(db, keyOf(r));
|
|
902
|
+
if (row && isContentEncodingRow(row))
|
|
903
|
+
storedEncoding.set(r.ref, row.encoding_salience);
|
|
1797
904
|
}
|
|
1798
|
-
|
|
1799
|
-
|
|
1800
|
-
try {
|
|
1801
|
-
withStateDb((outcomeDb) => {
|
|
1802
|
-
// Count accepted proposals per ref in one pass (avoid N separate queries).
|
|
1803
|
-
// Scoped to primaryStashDir when available so multi-stash installs don't
|
|
1804
|
-
// inflate counts with proposals from other stashes.
|
|
1805
|
-
const acceptedCountByRef = new Map();
|
|
1806
|
-
try {
|
|
1807
|
-
// #858/#859: listStateProposals() now skips-and-warns on individual
|
|
1808
|
-
// unparseable rows (including legacy pre-#578 rows with no
|
|
1809
|
-
// persisted `changes`, which it tolerates directly) instead of
|
|
1810
|
-
// throwing, so this no longer silently zeroes out every ref's
|
|
1811
|
-
// count on a single bad row. The outer try/catch stays as a
|
|
1812
|
-
// defense-in-depth fallback for unexpected failures (e.g. a query
|
|
1813
|
-
// error), not the primary safeguard it used to be.
|
|
1814
|
-
const acceptedProposals = listStateProposals(outcomeDb, {
|
|
1815
|
-
status: "accepted",
|
|
1816
|
-
...(primaryStashDir ? { stashDir: primaryStashDir } : {}),
|
|
1817
|
-
});
|
|
1818
|
-
for (const p of acceptedProposals) {
|
|
1819
|
-
acceptedCountByRef.set(p.ref, (acceptedCountByRef.get(p.ref) ?? 0) + 1);
|
|
1820
|
-
}
|
|
1821
|
-
}
|
|
1822
|
-
catch {
|
|
1823
|
-
// best-effort: if the query itself fails, accepted counts stay at 0
|
|
1824
|
-
}
|
|
1825
|
-
// Update each ref's outcome row and collect the resulting outcome scores.
|
|
1826
|
-
const rawOutcomeScores = new Map();
|
|
1827
|
-
const projectedByWriteKey = new Map();
|
|
1828
|
-
const nowForOutcome = Date.now();
|
|
1829
|
-
for (const r of mergedRefs) {
|
|
1830
|
-
const fb = feedbackSummary.get(r.ref) ?? { positive: 0, negative: 0 };
|
|
1831
|
-
const valenceResult = computeValenceScore(fb);
|
|
1832
|
-
try {
|
|
1833
|
-
const writeKey = outcomeWriteKey(r.ref, itemRefByRef);
|
|
1834
|
-
const inputs = {
|
|
1835
|
-
// Key by item_ref when resolved, else by the conceptId. Keep
|
|
1836
|
-
// rawOutcomeScores keyed by r.ref, its in-memory identity.
|
|
1837
|
-
ref: writeKey,
|
|
1838
|
-
currentRetrievalCount: retrievalCounts.get(r.ref) ?? 0,
|
|
1839
|
-
lastRetrievedAt: lastUseMsByRef.get(r.ref) ?? 0,
|
|
1840
|
-
acceptedChangeCount: acceptedCountByRef.get(r.ref) ?? 0,
|
|
1841
|
-
negativeFeedbackCount: fb.negative,
|
|
1842
|
-
valence: valenceResult.valence,
|
|
1843
|
-
utilityScore: utilityMap.get(r.ref),
|
|
1844
|
-
now: nowForOutcome,
|
|
1845
|
-
};
|
|
1846
|
-
const result = persist
|
|
1847
|
-
? updateAssetOutcome(outcomeDb, inputs)
|
|
1848
|
-
: projectAssetOutcome(getAssetOutcome(outcomeDb, writeKey), inputs);
|
|
1849
|
-
rawOutcomeScores.set(r.ref, result.outcomeScore);
|
|
1850
|
-
projectedByWriteKey.set(writeKey, result.outcomeScore);
|
|
1851
|
-
}
|
|
1852
|
-
catch {
|
|
1853
|
-
// best-effort per-ref: skip this ref's outcome update on failure
|
|
1854
|
-
}
|
|
1855
|
-
}
|
|
1856
|
-
// Compute stash-wide max outcome_score for normalisation (diversity floor).
|
|
1857
|
-
// Read ALL rows (not just this run's batch) so the normalisation is
|
|
1858
|
-
// stash-relative, not pool-relative.
|
|
1859
|
-
let maxOutcomeScore = 0;
|
|
1860
|
-
try {
|
|
1861
|
-
const allOutcomes = getAllAssetOutcomes(outcomeDb);
|
|
1862
|
-
const scoreByRef = new Map(allOutcomes.map((row) => [row.asset_ref, row.outcome_score]));
|
|
1863
|
-
for (const [ref, score] of projectedByWriteKey)
|
|
1864
|
-
scoreByRef.set(ref, score);
|
|
1865
|
-
for (const score of scoreByRef.values()) {
|
|
1866
|
-
if (score > maxOutcomeScore)
|
|
1867
|
-
maxOutcomeScore = score;
|
|
1868
|
-
}
|
|
1869
|
-
// Keep the normalization denominator within the writer's score bounds.
|
|
1870
|
-
maxOutcomeScore = Math.min(maxOutcomeScore, OUTCOME_SCORE_MAX);
|
|
1871
|
-
// Proxy-adequacy tripwire (two-tailed): inverted (corr < −0.3) and
|
|
1872
|
-
// dead (|corr| < 0.1 at n ≥ 500) both emit health events.
|
|
1873
|
-
const adequacy = persist ? computeProxyAdequacy(allOutcomes) : undefined;
|
|
1874
|
-
if (adequacy?.isInverted) {
|
|
1875
|
-
appendEvent({
|
|
1876
|
-
eventType: "outcome_proxy_inverted",
|
|
1877
|
-
ref: undefined,
|
|
1878
|
-
metadata: {
|
|
1879
|
-
correlation: adequacy.correlation,
|
|
1880
|
-
n: adequacy.n,
|
|
1881
|
-
note: "corr(outcome_score, accepted_change_rate) < −0.3: high-outcome_score assets have LOW accepted-change rates — the proxy's 'doing well' signal is inverted, so the coarse retrieval-delta signal is no longer trustworthy and the 0.10+ rich in-session signal is no longer deferrable. See plan §WS-2 proxy-adequacy tripwire.",
|
|
1882
|
-
},
|
|
1883
|
-
}, eventsCtx);
|
|
1884
|
-
}
|
|
1885
|
-
if (adequacy?.isDead) {
|
|
1886
|
-
appendEvent({
|
|
1887
|
-
eventType: "outcome_proxy_dead",
|
|
1888
|
-
ref: undefined,
|
|
1889
|
-
metadata: {
|
|
1890
|
-
correlation: adequacy.correlation,
|
|
1891
|
-
n: adequacy.n,
|
|
1892
|
-
note: "|corr(outcome_score, accepted_change_rate)| < 0.1 at n ≥ 500: outcome_score is statistically unrelated to improvement outcomes — the proxy is noise, not signal. Rank contributions derived from it are not currently informative.",
|
|
1893
|
-
},
|
|
1894
|
-
}, eventsCtx);
|
|
1895
|
-
}
|
|
1896
|
-
}
|
|
1897
|
-
catch {
|
|
1898
|
-
// best-effort: tripwire failure never blocks ranking
|
|
1899
|
-
}
|
|
1900
|
-
// Convert raw outcome scores → normalised outcomeSalience values in [0,1].
|
|
1901
|
-
for (const [ref, score] of rawOutcomeScores) {
|
|
1902
|
-
const normalised = outcomeScoreToSalience(score, maxOutcomeScore);
|
|
1903
|
-
outcomeSalienceByRef.set(ref, normalised);
|
|
1904
|
-
}
|
|
1905
|
-
// Also fetch outcome scores for refs NOT updated this run (stale or absent)
|
|
1906
|
-
// so the outcomeSalience read path works for all refs in the batch.
|
|
1907
|
-
// Chunk-5 flip F5e — query by each missing ref's WRITE key (item_ref,
|
|
1908
|
-
// else bare) and map the stored-key result back to the bare `r.ref`
|
|
1909
|
-
// identity that outcomeSalienceByRef is keyed on.
|
|
1910
|
-
const missingRefs = mergedRefs.map((r) => r.ref).filter((ref) => !rawOutcomeScores.has(ref));
|
|
1911
|
-
if (missingRefs.length > 0) {
|
|
1912
|
-
const refByWriteKey = new Map();
|
|
1913
|
-
for (const ref of missingRefs)
|
|
1914
|
-
refByWriteKey.set(outcomeWriteKey(ref, itemRefByRef), ref);
|
|
1915
|
-
const storedScores = getOutcomeScoresByRef(outcomeDb, [...refByWriteKey.keys()]);
|
|
1916
|
-
for (const [writeKey, score] of storedScores) {
|
|
1917
|
-
const bareRef = refByWriteKey.get(writeKey) ?? writeKey;
|
|
1918
|
-
outcomeSalienceByRef.set(bareRef, outcomeScoreToSalience(score, maxOutcomeScore));
|
|
1919
|
-
}
|
|
1920
|
-
}
|
|
1921
|
-
}, { path: eventsCtx?.dbPath, borrowed: eventsCtx?.db });
|
|
1922
|
-
}
|
|
1923
|
-
catch (err) {
|
|
1924
|
-
rethrowIfTestIsolationError(err);
|
|
1925
|
-
// best-effort: outcome failures never block salience computation
|
|
1926
|
-
}
|
|
1927
|
-
return outcomeSalienceByRef;
|
|
1928
|
-
}
|
|
1929
|
-
/** WS-1 — compute the salience vector per ref (#644 provenance preserved). */
|
|
1930
|
-
function computeSalienceVectors(args) {
|
|
1931
|
-
const { mergedRefs, itemRefByRef, options, eventsCtx, retrievalCounts, lastUseMsByRef, utilityMap, outcomeSalienceByRef, persist, } = args;
|
|
1932
|
-
// Compute the salience vector for every ref in the merged set.
|
|
1933
|
-
// retrievalCounts now covers the full candidate set (feedback-bearing + zero-feedback)
|
|
1934
|
-
// so feedback refs get their genuine retrieval frequency, not a 0-floor fallback.
|
|
1935
|
-
// outcomeSalienceByRef is populated by WS-2 above (or empty on first run).
|
|
1936
|
-
//
|
|
1937
|
-
// R1 loop closure: the outcome weight is ON by default (the G2 saturation
|
|
1938
|
-
// cap makes it safe). Operators opt out with
|
|
1939
|
-
// improve.salience.outcomeWeightEnabled: false in the config.
|
|
1940
|
-
const salienceConfig = (options.config ?? loadConfig()).improve?.salience;
|
|
1941
|
-
const outcomeWeightEnabled = salienceConfig?.outcomeWeightEnabled !== false;
|
|
905
|
+
});
|
|
906
|
+
const now = Date.now();
|
|
1942
907
|
const salienceMap = new Map();
|
|
1943
|
-
const nowForSalience = Date.now();
|
|
1944
|
-
// #644 — preserve content-derived encoding scores across runs.
|
|
1945
|
-
//
|
|
1946
|
-
// Before computing the salience vector, load each ref's stored encoding score
|
|
1947
|
-
// and its provenance. When the stored row carries a genuine content-derived
|
|
1948
|
-
// score (written by the distill path via `scoreEncodingSalience`), pass that
|
|
1949
|
-
// value back in as `inputs.encodingSalience` so `computeSalience` does NOT fall
|
|
1950
|
-
// back to the type-weight stub — keeping both the persisted `encoding_salience`
|
|
1951
|
-
// AND the derived `rank_score` keyed on real novelty/magnitude/prediction-error.
|
|
1952
|
-
// Refs that have never been content-scored keep the type-weight stub fallback.
|
|
1953
|
-
const storedEncodingByRef = new Map();
|
|
1954
|
-
try {
|
|
1955
|
-
if (persist || eventsCtx?.db) {
|
|
1956
|
-
withStateDb((dbForStoredEncoding) => {
|
|
1957
|
-
for (const r of mergedRefs) {
|
|
1958
|
-
const row = readAssetSalienceForImproveRef(dbForStoredEncoding, r.ref, itemRefByRef.get(r.ref));
|
|
1959
|
-
if (row && isContentEncodingRow(row)) {
|
|
1960
|
-
storedEncodingByRef.set(r.ref, row.encoding_salience);
|
|
1961
|
-
}
|
|
1962
|
-
}
|
|
1963
|
-
}, { path: eventsCtx?.dbPath, borrowed: eventsCtx?.db });
|
|
1964
|
-
}
|
|
1965
|
-
}
|
|
1966
|
-
catch (err) {
|
|
1967
|
-
rethrowIfTestIsolationError(err);
|
|
1968
|
-
// best-effort: if DB unavailable, fall back to type-weight stub (prior behaviour)
|
|
1969
|
-
}
|
|
1970
908
|
for (const r of mergedRefs) {
|
|
1971
|
-
const
|
|
1972
|
-
|
|
1973
|
-
const fp = r.filePath;
|
|
1974
|
-
if (!fp)
|
|
1975
|
-
return undefined;
|
|
1976
|
-
try {
|
|
1977
|
-
return fs.statSync(fp).size;
|
|
1978
|
-
}
|
|
1979
|
-
catch {
|
|
1980
|
-
return undefined;
|
|
1981
|
-
}
|
|
1982
|
-
})();
|
|
1983
|
-
const storedEncoding = storedEncodingByRef.get(r.ref);
|
|
1984
|
-
const vector = computeSalience({
|
|
909
|
+
const encoding = storedEncoding.get(r.ref);
|
|
910
|
+
salienceMap.set(r.ref, computeSalience({
|
|
1985
911
|
ref: r.ref,
|
|
1986
|
-
type,
|
|
1987
|
-
|
|
1988
|
-
// stub is NOT re-asserted over a real distill-written encoding score.
|
|
1989
|
-
...(storedEncoding !== undefined ? { encodingSalience: storedEncoding } : {}),
|
|
912
|
+
type: assetTypeOf(r.ref),
|
|
913
|
+
...(encoding !== undefined ? { encodingSalience: encoding } : {}),
|
|
1990
914
|
retrievalFreq: retrievalCounts.get(r.ref) ?? 0,
|
|
1991
915
|
lastUseMs: lastUseMsByRef.get(r.ref),
|
|
1992
916
|
utilityScore: utilityMap.get(r.ref),
|
|
1993
|
-
outcomeSalience:
|
|
1994
|
-
sizeBytes,
|
|
1995
|
-
now
|
|
917
|
+
outcomeSalience: outcomeSalience.get(r.ref),
|
|
918
|
+
sizeBytes: fileSize(r.filePath),
|
|
919
|
+
now,
|
|
1996
920
|
outcomeWeightEnabled,
|
|
1997
|
-
});
|
|
1998
|
-
salienceMap.set(r.ref, vector);
|
|
921
|
+
}));
|
|
1999
922
|
}
|
|
2000
|
-
|
|
923
|
+
const refByKey = new Map(mergedRefs.map((r) => [keyOf(r), r.ref]));
|
|
924
|
+
const pendingForgettingRefs = withRunState(eventsCtx, persist, (db) => {
|
|
925
|
+
// Positions are stash-wide: every stored row of this source, with this
|
|
926
|
+
// run's scores overlaid under the same keys.
|
|
927
|
+
const before = new Map();
|
|
928
|
+
for (const [ref, score] of getAllRankScores(db)) {
|
|
929
|
+
const boundary = ref.indexOf("//");
|
|
930
|
+
if (options.sourceName && (boundary >= 0 ? ref.slice(0, boundary) : undefined) !== options.sourceName)
|
|
931
|
+
continue;
|
|
932
|
+
before.set(ref, score);
|
|
933
|
+
}
|
|
934
|
+
let forgetting = [];
|
|
935
|
+
if (before.size > 0) {
|
|
936
|
+
const after = new Map(before);
|
|
937
|
+
for (const r of mergedRefs)
|
|
938
|
+
after.set(keyOf(r), salienceMap.get(r.ref)?.rankScore ?? 0);
|
|
939
|
+
const report = buildRankChangeReport(toRankPositions(before), toRankPositions(after));
|
|
940
|
+
if (report.forgettingCandidates.length > 0) {
|
|
941
|
+
const drops = report.forgettingCandidates
|
|
942
|
+
.slice(0, 5)
|
|
943
|
+
.map((e) => `${e.ref} (#${e.oldRank}→#${e.newRank})`)
|
|
944
|
+
.join(", ");
|
|
945
|
+
warn(`[improve/salience] WS-1 rank-change report: ${report.forgettingCandidates.length} asset(s) fell from top-200 to below position 500. Top drops: ${drops}`);
|
|
946
|
+
forgetting = report.forgettingCandidates.map((e) => refByKey.get(e.ref) ?? e.ref);
|
|
947
|
+
}
|
|
948
|
+
if (persist) {
|
|
949
|
+
appendEvent({
|
|
950
|
+
eventType: "improve_salience_rank_change",
|
|
951
|
+
ref: undefined,
|
|
952
|
+
metadata: {
|
|
953
|
+
stashSize: before.size,
|
|
954
|
+
totalChanged: report.allChanges.length,
|
|
955
|
+
forgettingCandidates: report.forgettingCandidates.length,
|
|
956
|
+
topDrops: report.forgettingCandidates
|
|
957
|
+
.slice(0, 10)
|
|
958
|
+
.map((e) => ({ ref: e.ref, oldRank: e.oldRank, newRank: e.newRank })),
|
|
959
|
+
},
|
|
960
|
+
}, eventsCtx);
|
|
961
|
+
}
|
|
962
|
+
}
|
|
963
|
+
if (persist) {
|
|
964
|
+
for (const r of mergedRefs)
|
|
965
|
+
upsertAssetSalience(db, keyOf(r), salienceMap.get(r.ref), now);
|
|
966
|
+
}
|
|
967
|
+
return forgetting;
|
|
968
|
+
}) ?? [];
|
|
969
|
+
return { salienceMap, pendingForgettingRefs };
|
|
2001
970
|
}
|
|
2002
|
-
/** 1-indexed
|
|
971
|
+
/** 1-indexed positions by score desc (ref asc on ties). */
|
|
2003
972
|
function toRankPositions(scores) {
|
|
2004
973
|
const sorted = [...scores.entries()].sort(([refA, a], [refB, b]) => b !== a ? b - a : refA < refB ? -1 : refA > refB ? 1 : 0);
|
|
2005
974
|
return new Map(sorted.map(([ref], i) => [ref, i + 1]));
|
|
2006
975
|
}
|
|
2007
976
|
/**
|
|
2008
|
-
*
|
|
2009
|
-
*
|
|
2010
|
-
*
|
|
2011
|
-
* key (no stashSize double-count); and
|
|
2012
|
-
* `refByWriteKey` reverses a write key back to its filesystem-facing bare ref.
|
|
977
|
+
* Update each ref's outcome row and return its outcome salience, normalized
|
|
978
|
+
* against the stash-wide maximum. Without state.db on a plan-only run, the
|
|
979
|
+
* values a live run would insert are projected instead.
|
|
2013
980
|
*/
|
|
2014
|
-
function
|
|
2015
|
-
const
|
|
2016
|
-
const
|
|
2017
|
-
const
|
|
2018
|
-
|
|
2019
|
-
|
|
2020
|
-
|
|
2021
|
-
|
|
2022
|
-
|
|
2023
|
-
|
|
2024
|
-
|
|
2025
|
-
|
|
2026
|
-
|
|
2027
|
-
|
|
2028
|
-
|
|
2029
|
-
|
|
2030
|
-
|
|
2031
|
-
|
|
2032
|
-
|
|
2033
|
-
|
|
2034
|
-
|
|
2035
|
-
|
|
2036
|
-
|
|
2037
|
-
// Forgetting-safety report (plan §WS-1 step 7) — stash-wide rank comparison:
|
|
2038
|
-
//
|
|
2039
|
-
// BEFORE persisting the new rankScores, read ALL existing rows from state.db
|
|
2040
|
-
// (not just the per-run candidate pool). This gives stash-wide rank positions so
|
|
2041
|
-
// the top-200/below-500 thresholds are meaningful.
|
|
2042
|
-
//
|
|
2043
|
-
// Two distinct scenarios:
|
|
2044
|
-
//
|
|
2045
|
-
// A. First WS-1 run (table empty): the old stash-wide combinedEligibilityScore
|
|
2046
|
-
// ordering was never persisted in state.db (asset_salience is a new WS-1 table).
|
|
2047
|
-
// However, the old formula's inputs are available in-scope for every candidate
|
|
2048
|
-
// in the current pool: utility comes from utilityMap and the attention term
|
|
2049
|
-
// from feedbackSummary (positive/negative counts). We reconstruct the old
|
|
2050
|
-
// combinedEligibilityScore = utility * UTILITY_WEIGHT + attention * FEEDBACK_WEIGHT
|
|
2051
|
-
// for every ref in salienceMap and rank them, giving a candidate-pool-scoped
|
|
2052
|
-
// old ordering. This is a partial reconstruction (only current-pool refs, not
|
|
2053
|
-
// stash-wide), but it is the most faithful comparison possible at cutover and
|
|
2054
|
-
// allows the top-200→below-500 forgetting guard to fire if the formula change
|
|
2055
|
-
// dramatically reorders the candidate pool.
|
|
2056
|
-
// WS-1 step 7 — the stash-wide
|
|
2057
|
-
// ordering was unreconstructable (no prior state.db snapshot), so this candidate-
|
|
2058
|
-
// pool partial reconstruction is the documented resolution for the first-run case.
|
|
2059
|
-
// Emit `improve_salience_first_run` to mark the cutover moment and include the
|
|
2060
|
-
// reconstructed comparison result in the metadata.
|
|
2061
|
-
//
|
|
2062
|
-
// B. Subsequent runs (table has rows): use ALL existing rows as old ranks, merge
|
|
2063
|
-
// them with the current run's salienceMap updates for new ranks, and call
|
|
2064
|
-
// buildRankChangeReport with stash-wide positions. This detects real rank drift
|
|
2065
|
-
// — e.g. a retrieval-pattern shift causing a previously top-200 asset to slip
|
|
2066
|
-
// below position 500.
|
|
2067
|
-
//
|
|
2068
|
-
// Measurement-protocol deferral (plan §269, Part-V):
|
|
2069
|
-
// The Part-V T0 baseline (scripts/akm-eval + health report) and the throughput/
|
|
2070
|
-
// quality gate are deferred pending owner sign-off. Full measurement requires a
|
|
2071
|
-
// before/after `akm health` report. Owner-acknowledged deferral: WS-2 landing
|
|
2072
|
-
// will re-introduce outcome salience and trigger the full re-tuning pass at that
|
|
2073
|
-
// time. salience.ts already accepts outcomeSalience directly as an input
|
|
2074
|
-
// (see SalienceInputs.outcomeSalience); no separate hook is needed.
|
|
2075
|
-
//
|
|
2076
|
-
// Forgetting-safety collection: populated inside scenario B below, consumed
|
|
2077
|
-
// after the try/catch to union candidates into mergedRefs before the sort.
|
|
2078
|
-
// Only refs from a real pre-existing ordering (scenario B) are collected;
|
|
2079
|
-
// empty on scenario A or when no candidates dropped below the threshold.
|
|
2080
|
-
let pendingForgettingRefs = [];
|
|
2081
|
-
try {
|
|
2082
|
-
if (!persist && !eventsCtx?.db)
|
|
2083
|
-
return pendingForgettingRefs;
|
|
2084
|
-
withStateDb((stateDb) => {
|
|
2085
|
-
// Step 7: stash-wide rank-change report BEFORE overwriting the table.
|
|
2086
|
-
//
|
|
2087
|
-
// Load ALL existing rows so rank positions are stash-relative, not pool-relative.
|
|
2088
|
-
// Source-scope by the `<bundle>//` prefix and fold each in-pool asset's
|
|
2089
|
-
// stored spelling onto its single write key so
|
|
2090
|
-
// the merge below never double-counts one asset across two spellings.
|
|
2091
|
-
const allStoredScores = getAllRankScores(stateDb);
|
|
2092
|
-
const existingAllScores = new Map();
|
|
2093
|
-
for (const [ref, score] of allStoredScores) {
|
|
2094
|
-
if (options.sourceName) {
|
|
2095
|
-
const boundary = ref.indexOf("//");
|
|
2096
|
-
const prefix = boundary >= 0 ? ref.slice(0, boundary) : undefined;
|
|
2097
|
-
const belongs = prefix === options.sourceName;
|
|
2098
|
-
if (!belongs)
|
|
2099
|
-
continue;
|
|
2100
|
-
}
|
|
2101
|
-
existingAllScores.set(normalizeStoredKey.get(ref) ?? ref, score);
|
|
2102
|
-
}
|
|
2103
|
-
if (existingAllScores.size === 0) {
|
|
2104
|
-
// Scenario A: first WS-1 run — table empty.
|
|
2105
|
-
//
|
|
2106
|
-
// Reconstruct the old combinedEligibilityScore ordering for the current
|
|
2107
|
-
// candidate pool using inputs that are already in-scope: utility from
|
|
2108
|
-
// utilityMap and the attention term from feedbackSummary (positive/negative
|
|
2109
|
-
// counts). Old formula: score = utility * UTILITY_WEIGHT + attention * FEEDBACK_WEIGHT.
|
|
2110
|
-
//
|
|
2111
|
-
// Limitation: this covers only the current-run candidate pool, not the full
|
|
2112
|
-
// stash. The stash-wide ordering was never persisted (asset_salience is a new
|
|
2113
|
-
// WS-1 table), so this is the most faithful comparison possible at cutover.
|
|
2114
|
-
// WS-1 step 7.
|
|
2115
|
-
const reconstructedOldScores = new Map();
|
|
2116
|
-
for (const ref of salienceMap.keys()) {
|
|
2117
|
-
const utility = utilityMap.get(ref) ?? 0;
|
|
2118
|
-
const fb = feedbackSummary.get(ref) ?? { positive: 0, negative: 0 };
|
|
2119
|
-
const attention = computeValenceScore(fb).attention;
|
|
2120
|
-
reconstructedOldScores.set(ref, utility * UTILITY_WEIGHT + attention * FEEDBACK_WEIGHT);
|
|
2121
|
-
}
|
|
2122
|
-
// Assign 1-indexed rank positions sorted by score desc (tie-break: ref asc).
|
|
2123
|
-
const oldRanks = toRankPositions(reconstructedOldScores);
|
|
2124
|
-
const newRanks = toRankPositions(new Map([...salienceMap.entries()].map(([ref, v]) => [ref, v.rankScore])));
|
|
2125
|
-
const firstRunReport = buildRankChangeReport(oldRanks, newRanks);
|
|
2126
|
-
if (firstRunReport.forgettingCandidates.length > 0) {
|
|
2127
|
-
warn(`[improve/salience] WS-1 first-run rank-change report: ${firstRunReport.forgettingCandidates.length} asset(s) fell from top-200 to below position 500 (cutover formula change). ` +
|
|
2128
|
-
`Top drops: ${firstRunReport.forgettingCandidates
|
|
2129
|
-
.slice(0, 5)
|
|
2130
|
-
.map((e) => `${e.ref} (#${e.oldRank}→#${e.newRank})`)
|
|
2131
|
-
.join(", ")}`);
|
|
2132
|
-
pendingForgettingRefs = firstRunReport.forgettingCandidates.map((e) => e.ref);
|
|
2133
|
-
}
|
|
2134
|
-
if (persist) {
|
|
2135
|
-
appendEvent({
|
|
2136
|
-
eventType: "improve_salience_first_run",
|
|
2137
|
-
ref: undefined,
|
|
2138
|
-
metadata: {
|
|
2139
|
-
candidateCount: salienceMap.size,
|
|
2140
|
-
note: "first WS-1 salience run — partial reconstruction of old combinedEligibilityScore ordering for candidate pool (stash-wide ordering not available); WS-1 step 7",
|
|
2141
|
-
forgettingCandidates: firstRunReport.forgettingCandidates.length,
|
|
2142
|
-
topDrops: firstRunReport.forgettingCandidates.slice(0, 10).map((e) => ({
|
|
2143
|
-
ref: e.ref,
|
|
2144
|
-
oldRank: e.oldRank,
|
|
2145
|
-
newRank: e.newRank,
|
|
2146
|
-
})),
|
|
2147
|
-
},
|
|
2148
|
-
}, eventsCtx);
|
|
2149
|
-
}
|
|
2150
|
-
}
|
|
2151
|
-
else {
|
|
2152
|
-
// Scenario B: subsequent run — compare stash-wide old vs. new ranks.
|
|
2153
|
-
//
|
|
2154
|
-
// Build new scores by merging the full table with this run's updates.
|
|
2155
|
-
// Refs in salienceMap override their stored value; refs not in this run
|
|
2156
|
-
// retain their stored value unchanged. This gives a complete stash-wide
|
|
2157
|
-
// picture of what the new ordering looks like after this run.
|
|
2158
|
-
const mergedNewScores = new Map(existingAllScores);
|
|
2159
|
-
for (const [ref, vector] of salienceMap) {
|
|
2160
|
-
// Chunk-5 flip F5e — key this run's fresh scores by the WRITE key so
|
|
2161
|
-
// they overwrite (never duplicate) the same asset's normalized stored row.
|
|
2162
|
-
mergedNewScores.set(wk(ref), vector.rankScore);
|
|
2163
|
-
}
|
|
2164
|
-
// Assign 1-indexed rank positions sorted by score desc (tie-break: ref asc).
|
|
2165
|
-
const oldRanks = toRankPositions(existingAllScores);
|
|
2166
|
-
const newRanks = toRankPositions(mergedNewScores);
|
|
2167
|
-
const report = buildRankChangeReport(oldRanks, newRanks);
|
|
2168
|
-
if (report.forgettingCandidates.length > 0) {
|
|
2169
|
-
warn(`[improve/salience] WS-1 rank-change report: ${report.forgettingCandidates.length} asset(s) fell from top-200 to below position 500. ` +
|
|
2170
|
-
`Top drops: ${report.forgettingCandidates
|
|
2171
|
-
.slice(0, 5)
|
|
2172
|
-
.map((e) => `${e.ref} (#${e.oldRank}→#${e.newRank})`)
|
|
2173
|
-
.join(", ")}`);
|
|
2174
|
-
// Collect refs for protective consolidation pass (plan §WS-1 step 7).
|
|
2175
|
-
// These are force-included in the candidate pool (mergedRefs) after
|
|
2176
|
-
// this try block, bypassing cooldown/signal-delta gating.
|
|
2177
|
-
// Chunk-5 flip F5e — map an in-pool candidate's write-key spelling
|
|
2178
|
-
// back to its bare `r.ref` so applyForgettingSafety re-stamps the
|
|
2179
|
-
// existing pool ref. Other stored spellings stay qualified here;
|
|
2180
|
-
// the downstream admission boundary resolves them only when they
|
|
2181
|
-
// match an exact current-plan item_ref.
|
|
2182
|
-
pendingForgettingRefs = report.forgettingCandidates.map((e) => refByWriteKey.get(e.ref) ?? e.ref);
|
|
2183
|
-
}
|
|
2184
|
-
if (persist) {
|
|
2185
|
-
appendEvent({
|
|
2186
|
-
eventType: "improve_salience_rank_change",
|
|
2187
|
-
ref: undefined,
|
|
2188
|
-
metadata: {
|
|
2189
|
-
stashSize: existingAllScores.size,
|
|
2190
|
-
totalChanged: report.allChanges.length,
|
|
2191
|
-
forgettingCandidates: report.forgettingCandidates.length,
|
|
2192
|
-
topDrops: report.forgettingCandidates.slice(0, 10).map((e) => ({
|
|
2193
|
-
ref: e.ref,
|
|
2194
|
-
oldRank: e.oldRank,
|
|
2195
|
-
newRank: e.newRank,
|
|
2196
|
-
})),
|
|
2197
|
-
},
|
|
2198
|
-
}, eventsCtx);
|
|
2199
|
-
}
|
|
2200
|
-
}
|
|
2201
|
-
if (persist) {
|
|
2202
|
-
for (const [ref, vector] of salienceMap) {
|
|
2203
|
-
// Persist salience under item_ref when resolved, else the conceptId.
|
|
2204
|
-
upsertAssetSalience(stateDb, wk(ref), vector, nowForSalience);
|
|
2205
|
-
}
|
|
2206
|
-
}
|
|
2207
|
-
}, { path: eventsCtx?.dbPath, borrowed: eventsCtx?.db });
|
|
2208
|
-
}
|
|
2209
|
-
catch (err) {
|
|
2210
|
-
rethrowIfTestIsolationError(err);
|
|
2211
|
-
// best-effort: salience persistence failure never blocks ranking
|
|
981
|
+
function updateOutcomeScores(args) {
|
|
982
|
+
const { mergedRefs, feedback, eventsCtx, persist } = args;
|
|
983
|
+
const now = Date.now();
|
|
984
|
+
const inputsFor = (r, accepted) => {
|
|
985
|
+
const fb = feedback.get(r.ref) ?? { positive: 0, negative: 0 };
|
|
986
|
+
return {
|
|
987
|
+
ref: keyOf(r),
|
|
988
|
+
currentRetrievalCount: args.retrievalCounts.get(r.ref) ?? 0,
|
|
989
|
+
lastRetrievedAt: args.lastUseMsByRef.get(r.ref) ?? 0,
|
|
990
|
+
acceptedChangeCount: accepted,
|
|
991
|
+
negativeFeedbackCount: fb.negative,
|
|
992
|
+
valence: computeValenceScore(fb).valence,
|
|
993
|
+
utilityScore: args.utilityMap.get(r.ref),
|
|
994
|
+
now,
|
|
995
|
+
};
|
|
996
|
+
};
|
|
997
|
+
const out = new Map();
|
|
998
|
+
if (!persist && !eventsCtx?.db) {
|
|
999
|
+
const projected = new Map(mergedRefs.map((r) => [r.ref, projectAssetOutcome(undefined, inputsFor(r, 0)).outcomeScore]));
|
|
1000
|
+
const max = Math.min(OUTCOME_SCORE_MAX, Math.max(0, ...projected.values()));
|
|
1001
|
+
for (const [ref, score] of projected)
|
|
1002
|
+
out.set(ref, outcomeScoreToSalience(score, max));
|
|
1003
|
+
return out;
|
|
2212
1004
|
}
|
|
2213
|
-
|
|
2214
|
-
|
|
2215
|
-
|
|
2216
|
-
|
|
2217
|
-
|
|
2218
|
-
|
|
2219
|
-
|
|
2220
|
-
|
|
2221
|
-
|
|
2222
|
-
let mergedRefs = args.mergedRefs;
|
|
2223
|
-
// ── Protective consolidation pass (plan §WS-1 step 7) ─────────────────────
|
|
2224
|
-
// Forgetting candidates detected in scenario B are force-injected into
|
|
2225
|
-
// mergedRefs here, BEFORE the effectiveScore sort, bypassing cooldown and
|
|
2226
|
-
// signal-delta gating. The lane may only reuse exact objects from this
|
|
2227
|
-
// invocation's post-cleanup/post-validation plan. Stale, out-of-scope, and
|
|
2228
|
-
// differently-qualified durable state must never synthesize executable work.
|
|
2229
|
-
if (pendingForgettingRefs.length > 0 && scope.mode !== "ref" && allowFallbacks) {
|
|
2230
|
-
const existingRefSet = new Set(mergedRefs.map((r) => r.ref));
|
|
2231
|
-
const eligibleByRef = new Map(eligibleRefs.map((candidate) => [candidate.ref, candidate]));
|
|
2232
|
-
const eligibleByItemRef = new Map();
|
|
2233
|
-
for (const candidate of eligibleRefs) {
|
|
2234
|
-
if (candidate.itemRef)
|
|
2235
|
-
eligibleByItemRef.set(candidate.itemRef, candidate);
|
|
2236
|
-
}
|
|
2237
|
-
const newForgettingRefs = [];
|
|
2238
|
-
const forgettingRefSet = new Set();
|
|
2239
|
-
for (const stateRef of pendingForgettingRefs) {
|
|
2240
|
-
const boundary = stateRef.indexOf("//");
|
|
2241
|
-
const candidate = eligibleByItemRef.get(stateRef) ?? (boundary < 0 ? eligibleByRef.get(bareImproveRef(stateRef)) : undefined);
|
|
2242
|
-
if (!candidate || forgettingRefSet.has(candidate.ref))
|
|
2243
|
-
continue;
|
|
2244
|
-
forgettingRefSet.add(candidate.ref);
|
|
2245
|
-
if (!existingRefSet.has(candidate.ref)) {
|
|
2246
|
-
newForgettingRefs.push(candidate);
|
|
2247
|
-
existingRefSet.add(candidate.ref);
|
|
2248
|
-
}
|
|
2249
|
-
// Always stamp the lane in the attribution map (overwrites weaker lanes;
|
|
2250
|
-
// stronger reactive signals — scope/signal-delta/proactive — are written
|
|
2251
|
-
// after this block so they take precedence).
|
|
2252
|
-
eligibilitySourceByRef.set(candidate.ref, "forgetting-safety");
|
|
2253
|
-
}
|
|
2254
|
-
if (newForgettingRefs.length > 0) {
|
|
2255
|
-
mergedRefs = dedupeRefs([...mergedRefs, ...newForgettingRefs]);
|
|
1005
|
+
withRunState(eventsCtx, persist, (db) => {
|
|
1006
|
+
const accepted = new Map();
|
|
1007
|
+
try {
|
|
1008
|
+
const rows = listStateProposals(db, {
|
|
1009
|
+
status: "accepted",
|
|
1010
|
+
...(args.primaryStashDir ? { stashDir: args.primaryStashDir } : {}),
|
|
1011
|
+
});
|
|
1012
|
+
for (const p of rows)
|
|
1013
|
+
accepted.set(p.ref, (accepted.get(p.ref) ?? 0) + 1);
|
|
2256
1014
|
}
|
|
2257
|
-
|
|
2258
|
-
|
|
2259
|
-
// proactive < forgetting-safety < signal-delta
|
|
2260
|
-
// Scope mode is already excluded by the outer guard (`scope.mode !== "ref"`).
|
|
2261
|
-
// forgetting-safety sits above proactive so that a ref flagged as a
|
|
2262
|
-
// forgetting candidate is always visible to S5/WS-5 as such, even when it
|
|
2263
|
-
// was also due for a proactive maintenance run. signal-delta overrides
|
|
2264
|
-
// forgetting-safety because a ref with fresh feedback is reactive and
|
|
2265
|
-
// doesn't need the protective pass label for measurement purposes.
|
|
2266
|
-
for (const r of highSalienceRefs)
|
|
2267
|
-
eligibilitySourceByRef.set(r.ref, "high-salience");
|
|
2268
|
-
for (const r of proactiveRefs)
|
|
2269
|
-
eligibilitySourceByRef.set(r.ref, "proactive");
|
|
2270
|
-
// Apply forgetting-safety OVER proactive and high-salience (already
|
|
2271
|
-
// stamped in the loop above via
|
|
2272
|
-
// `eligibilitySourceByRef.set(ref, "forgetting-safety")`). No-op here: the
|
|
2273
|
-
// set() calls above for proactive/high-salience overwrite the earlier
|
|
2274
|
-
// forgetting-safety stamp — so we re-apply forgetting-safety now for those
|
|
2275
|
-
// refs that are both forgetting candidates AND in another fallback lane.
|
|
2276
|
-
for (const ref of forgettingRefSet) {
|
|
2277
|
-
eligibilitySourceByRef.set(ref, "forgetting-safety");
|
|
1015
|
+
catch {
|
|
1016
|
+
// Accepted counts stay 0.
|
|
2278
1017
|
}
|
|
2279
|
-
|
|
2280
|
-
|
|
2281
|
-
eligibilitySourceByRef.set(r.ref, "signal-delta");
|
|
2282
|
-
// Update eligibilitySource on the ref objects themselves for any refs whose
|
|
2283
|
-
// lane changed (covers both new stubs and pre-existing refs).
|
|
1018
|
+
const raw = new Map();
|
|
1019
|
+
const byKey = new Map();
|
|
2284
1020
|
for (const r of mergedRefs) {
|
|
2285
|
-
r.eligibilitySource = eligibilitySourceByRef.get(r.ref) ?? "unknown";
|
|
2286
|
-
}
|
|
2287
|
-
}
|
|
2288
|
-
return mergedRefs;
|
|
2289
|
-
}
|
|
2290
|
-
/**
|
|
2291
|
-
* Pass: eligibility-filter — replay selection (#610), the no-op dampener sort,
|
|
2292
|
-
* coverage gaps, the disk-existence guard, the --limit slice, and the summary
|
|
2293
|
-
* info emits.
|
|
2294
|
-
*/
|
|
2295
|
-
async function filterEligibility(args) {
|
|
2296
|
-
const { scope, options, replayEligibleRefs, eventsCtx, salienceMap, eligibilitySourceByRef, distillOnlyRefs, persist, } = args;
|
|
2297
|
-
const { signalAndRetrievalRefs, signalFiltered } = args.summary;
|
|
2298
|
-
const validationFailureRefs = args.validationFailureRefs;
|
|
2299
|
-
const replay = applyReplaySelection({
|
|
2300
|
-
scope,
|
|
2301
|
-
options,
|
|
2302
|
-
plannedRefs: replayEligibleRefs,
|
|
2303
|
-
eventsCtx,
|
|
2304
|
-
mergedRefs: args.mergedRefs,
|
|
2305
|
-
salienceMap,
|
|
2306
|
-
eligibilitySourceByRef,
|
|
2307
|
-
persist,
|
|
2308
|
-
});
|
|
2309
|
-
const mergedRefs = replay.mergedRefs;
|
|
2310
|
-
const { replayRefSet, replayBudget } = replay;
|
|
2311
|
-
// Build no-op map for consolidation-selection dampener (plan §WS-1 step 8).
|
|
2312
|
-
// Reads consecutive_no_ops from the SAME pinned db handle used elsewhere in
|
|
2313
|
-
// this function. The effective score is used ONLY for processing/selection
|
|
2314
|
-
// order — the persisted rank_score in asset_salience is never mutated here.
|
|
2315
|
-
const noOpMap = new Map();
|
|
2316
|
-
try {
|
|
2317
|
-
const noOpDb = eventsCtx?.db ?? (persist && eventsCtx?.dbPath ? openStateDatabase(eventsCtx.dbPath) : null);
|
|
2318
|
-
if (noOpDb) {
|
|
2319
|
-
const ownsNoOpDb = !eventsCtx?.db;
|
|
2320
1021
|
try {
|
|
2321
|
-
|
|
2322
|
-
|
|
2323
|
-
|
|
1022
|
+
const inputs = inputsFor(r, accepted.get(r.ref) ?? 0);
|
|
1023
|
+
const result = persist
|
|
1024
|
+
? updateAssetOutcome(db, inputs)
|
|
1025
|
+
: projectAssetOutcome(getAssetOutcome(db, inputs.ref), inputs);
|
|
1026
|
+
raw.set(r.ref, result.outcomeScore);
|
|
1027
|
+
byKey.set(inputs.ref, result.outcomeScore);
|
|
2324
1028
|
}
|
|
2325
|
-
|
|
2326
|
-
|
|
2327
|
-
noOpDb.close();
|
|
1029
|
+
catch {
|
|
1030
|
+
// This ref keeps its stored score.
|
|
2328
1031
|
}
|
|
2329
1032
|
}
|
|
2330
|
-
|
|
2331
|
-
|
|
2332
|
-
|
|
2333
|
-
|
|
2334
|
-
|
|
2335
|
-
|
|
2336
|
-
|
|
2337
|
-
|
|
2338
|
-
|
|
2339
|
-
|
|
2340
|
-
|
|
2341
|
-
|
|
2342
|
-
|
|
2343
|
-
|
|
2344
|
-
|
|
2345
|
-
|
|
2346
|
-
|
|
2347
|
-
|
|
2348
|
-
|
|
2349
|
-
|
|
2350
|
-
|
|
2351
|
-
// proactive / high-salience) survive as labels set above.
|
|
2352
|
-
const effectiveScore = (ref) => {
|
|
2353
|
-
const rankScore = salienceMap.get(ref)?.rankScore ?? 0;
|
|
2354
|
-
const noOps = noOpMap.get(ref) ?? 0;
|
|
2355
|
-
return noOps >= SALIENCE_NO_OP_DAMPEN_THRESHOLD ? rankScore * SALIENCE_NO_OP_DAMPEN_FACTOR : rankScore;
|
|
2356
|
-
};
|
|
2357
|
-
const sorted = [...mergedRefs].sort((a, b) => {
|
|
2358
|
-
const scoreA = effectiveScore(a.ref);
|
|
2359
|
-
const scoreB = effectiveScore(b.ref);
|
|
2360
|
-
if (scoreB !== scoreA)
|
|
2361
|
-
return scoreB - scoreA;
|
|
2362
|
-
// Stable tie-break: deterministic regardless of input ordering.
|
|
2363
|
-
return a.ref < b.ref ? -1 : a.ref > b.ref ? 1 : 0;
|
|
2364
|
-
});
|
|
2365
|
-
// Phase 0: surface coverage gaps from zero-result search queries
|
|
2366
|
-
let coverageGaps = [];
|
|
2367
|
-
try {
|
|
2368
|
-
const dbForGaps = persist
|
|
2369
|
-
? openExistingDatabase()
|
|
2370
|
-
: openReadonlyExistingDatabase(undefined, { isolatedSnapshot: true });
|
|
2371
|
-
if (dbForGaps) {
|
|
2372
|
-
try {
|
|
2373
|
-
coverageGaps = getZeroResultSearches(dbForGaps);
|
|
2374
|
-
}
|
|
2375
|
-
finally {
|
|
2376
|
-
closeDatabase(dbForGaps);
|
|
1033
|
+
// Normalize stash-wide (every row, this run's overlaid), within the writer's bound.
|
|
1034
|
+
let max = 0;
|
|
1035
|
+
try {
|
|
1036
|
+
const scores = new Map(getAllAssetOutcomes(db).map((row) => [row.asset_ref, row.outcome_score]));
|
|
1037
|
+
for (const [key, score] of byKey)
|
|
1038
|
+
scores.set(key, score);
|
|
1039
|
+
for (const score of scores.values())
|
|
1040
|
+
if (score > max)
|
|
1041
|
+
max = score;
|
|
1042
|
+
max = Math.min(max, OUTCOME_SCORE_MAX);
|
|
1043
|
+
}
|
|
1044
|
+
catch {
|
|
1045
|
+
max = 0;
|
|
1046
|
+
}
|
|
1047
|
+
for (const [ref, score] of raw)
|
|
1048
|
+
out.set(ref, outcomeScoreToSalience(score, max));
|
|
1049
|
+
const missing = mergedRefs.filter((r) => !raw.has(r.ref));
|
|
1050
|
+
if (missing.length > 0) {
|
|
1051
|
+
const refByKey = new Map(missing.map((r) => [keyOf(r), r.ref]));
|
|
1052
|
+
for (const [key, score] of getOutcomeScoresByRef(db, [...refByKey.keys()])) {
|
|
1053
|
+
out.set(refByKey.get(key) ?? key, outcomeScoreToSalience(score, max));
|
|
2377
1054
|
}
|
|
2378
1055
|
}
|
|
2379
|
-
}
|
|
2380
|
-
catch (err) {
|
|
2381
|
-
rethrowIfTestIsolationError(err);
|
|
2382
|
-
// best-effort
|
|
2383
|
-
}
|
|
2384
|
-
const diskCheck = await dropRefsMissingOnDisk({ sorted, options, eventsCtx, persist });
|
|
2385
|
-
const assetMissingOnDisk = diskCheck.assetMissingOnDisk;
|
|
2386
|
-
const actionableRefs = diskCheck.actionableRefs;
|
|
2387
|
-
// Re-split actionableRefs (sorted) into reflect-path vs distill-only-path while
|
|
2388
|
-
// preserving sort order. distillOnlyRefs participate in the sort so --limit
|
|
2389
|
-
// picks them by score, not by arbitrary position.
|
|
2390
|
-
// ── Phase 5: --limit applies to the post-cooldown actionable set ──────────
|
|
2391
|
-
//
|
|
2392
|
-
// #610 ADDITIVITY: replay-lane refs are budgeted SEPARATELY from the --limit
|
|
2393
|
-
// fresh slice. Without this split, a high-rankScore replay ref could sort above
|
|
2394
|
-
// a fresh ref in the single combined slice and STEAL its slot (violating AC2).
|
|
2395
|
-
// We partition into the replay lane vs the rest, apply --limit to the
|
|
2396
|
-
// non-replay (fresh) refs only, then APPEND up to `replayBudget` replay refs
|
|
2397
|
-
// after the fresh slice. Sort order within each partition is preserved.
|
|
2398
|
-
//
|
|
2399
|
-
// Default replayBudget=0 reduces this to the exact pre-#610 expression: with no
|
|
2400
|
-
// replay refs, `nonReplayLoop === allLoopRefs`, so `baseLoop === old slice` and
|
|
2401
|
-
// `replayLoop.slice(0, 0) === []` — byte-identical.
|
|
2402
|
-
const selection = selectEffectiveImproveRefs({
|
|
2403
|
-
rankedRefs: actionableRefs,
|
|
2404
|
-
distillOnlyRefs,
|
|
2405
|
-
limit: options.limit,
|
|
2406
|
-
replayBudget,
|
|
2407
1056
|
});
|
|
2408
|
-
|
|
2409
|
-
const distillOnlyRefsResult = selection.distillOnlyRefs;
|
|
2410
|
-
if (signalAndRetrievalRefs.length > 0) {
|
|
2411
|
-
info(`[improve] ${signalAndRetrievalRefs.length} refs with usage signals (${signalFiltered.length} feedback${replayRefSet.size > 0 ? `, ${replayRefSet.size} replay` : ""})`);
|
|
2412
|
-
}
|
|
2413
|
-
if (validationFailureRefs.size > 0) {
|
|
2414
|
-
info(`[improve] ${validationFailureRefs.size} with validation failures excluded`);
|
|
2415
|
-
}
|
|
2416
|
-
if (persist && assetMissingOnDisk.length > 0) {
|
|
2417
|
-
info(`[improve] ${assetMissingOnDisk.length} candidates dropped — file not on disk`);
|
|
2418
|
-
}
|
|
2419
|
-
const deferredCount = actionableRefs.length - loopRefs.length;
|
|
2420
|
-
info(`[improve] ${actionableRefs.length} actionable; ${loopRefs.length} will be processed` +
|
|
2421
|
-
(options.limit && deferredCount > 0 ? ` (--limit ${options.limit} applied; ${deferredCount} deferred)` : ""));
|
|
2422
|
-
return {
|
|
2423
|
-
loopRefs,
|
|
2424
|
-
actionableRefs,
|
|
2425
|
-
distillOnlyRefs: distillOnlyRefsResult,
|
|
2426
|
-
coverageGaps,
|
|
2427
|
-
limitRemoved: selection.limitRemoved,
|
|
2428
|
-
missingDiskCount: assetMissingOnDisk.length,
|
|
2429
|
-
replayBudget,
|
|
2430
|
-
preDiskRefs: sorted,
|
|
2431
|
-
};
|
|
1057
|
+
return out;
|
|
2432
1058
|
}
|
|
2433
|
-
/**
|
|
2434
|
-
|
|
2435
|
-
|
|
1059
|
+
/**
|
|
1060
|
+
* Forgetting safety: inject this plan's own candidates that fell out of the
|
|
1061
|
+
* top ranks, past the signal gate (never for a ref scope or with
|
|
1062
|
+
* `--require-feedback-signal`). Attribution afterwards: high-salience <
|
|
1063
|
+
* proactive < forgetting-safety < signal-delta.
|
|
1064
|
+
*/
|
|
1065
|
+
export function applyForgettingSafety(args) {
|
|
1066
|
+
const { eligibilitySourceByRef } = args;
|
|
2436
1067
|
let mergedRefs = args.mergedRefs;
|
|
2437
|
-
|
|
2438
|
-
|
|
2439
|
-
|
|
2440
|
-
|
|
2441
|
-
|
|
2442
|
-
|
|
2443
|
-
|
|
2444
|
-
|
|
2445
|
-
|
|
2446
|
-
|
|
2447
|
-
|
|
2448
|
-
|
|
2449
|
-
|
|
2450
|
-
|
|
2451
|
-
|
|
2452
|
-
|
|
2453
|
-
if (replayBudget > 0 && scope.mode !== "ref" && !options.requireFeedbackSignal) {
|
|
2454
|
-
try {
|
|
2455
|
-
if (!persist && !eventsCtx?.db)
|
|
2456
|
-
return { mergedRefs, replayRefSet, replayBudget };
|
|
2457
|
-
withStateDb((replayDb) => {
|
|
2458
|
-
const alreadyInPool = new Set(mergedRefs.map((r) => r.ref));
|
|
2459
|
-
const storedRankScores = getAllRankScores(replayDb);
|
|
2460
|
-
const plannedByRef = new Map(plannedRefs.map((planned) => [planned.ref, planned]));
|
|
2461
|
-
const plannedByItemRef = new Map();
|
|
2462
|
-
for (const planned of plannedRefs) {
|
|
2463
|
-
if (planned.itemRef)
|
|
2464
|
-
plannedByItemRef.set(planned.itemRef, planned);
|
|
2465
|
-
}
|
|
2466
|
-
// Replay can only revisit an entry selected into THIS invocation's
|
|
2467
|
-
// source/type plan. Match durable item_ref rows by exact provenance;
|
|
2468
|
-
// legacy bare rows may match the current plan's concept ref. Folding
|
|
2469
|
-
// both spellings onto the planned ref also prevents duplicate budget
|
|
2470
|
-
// spend when old and current state rows coexist.
|
|
2471
|
-
const allRankScores = new Map();
|
|
2472
|
-
for (const [stateRef, score] of storedRankScores) {
|
|
2473
|
-
const boundary = stateRef.indexOf("//");
|
|
2474
|
-
if (options.sourceName && (boundary < 0 || stateRef.slice(0, boundary) !== options.sourceName))
|
|
2475
|
-
continue;
|
|
2476
|
-
const planned = plannedByItemRef.get(stateRef) ?? (boundary < 0 ? plannedByRef.get(bareImproveRef(stateRef)) : undefined);
|
|
2477
|
-
if (!planned)
|
|
2478
|
-
continue;
|
|
2479
|
-
const previous = allRankScores.get(planned.ref);
|
|
2480
|
-
if (previous === undefined || score > previous)
|
|
2481
|
-
allRankScores.set(planned.ref, score);
|
|
2482
|
-
}
|
|
2483
|
-
// Candidate universe = every current-plan salience match NOT already in the
|
|
2484
|
-
// pool, ordered by rank_score desc with a deterministic ref-string tie-break
|
|
2485
|
-
// (mirrors the main sort). Converged refs (consecutive_no_ops >= dampener
|
|
2486
|
-
// threshold) are fully EXCLUDED — a stronger skip than the dampener (which
|
|
2487
|
-
// only halves order).
|
|
2488
|
-
let convergedSkipped = 0;
|
|
2489
|
-
const candidates = [];
|
|
2490
|
-
for (const [ref, rankScore] of allRankScores) {
|
|
2491
|
-
if (alreadyInPool.has(ref))
|
|
2492
|
-
continue;
|
|
2493
|
-
const planned = plannedByRef.get(ref);
|
|
2494
|
-
if (!planned)
|
|
2495
|
-
continue;
|
|
2496
|
-
const noOps = readConsecutiveNoOpsForImproveRef(replayDb, ref, planned.itemRef);
|
|
2497
|
-
if (noOps >= SALIENCE_NO_OP_DAMPEN_THRESHOLD) {
|
|
2498
|
-
convergedSkipped++;
|
|
2499
|
-
continue;
|
|
2500
|
-
}
|
|
2501
|
-
candidates.push({ planned, rankScore });
|
|
2502
|
-
}
|
|
2503
|
-
candidates.sort((a, b) => b.rankScore !== a.rankScore
|
|
2504
|
-
? b.rankScore - a.rankScore
|
|
2505
|
-
: a.planned.ref < b.planned.ref
|
|
2506
|
-
? -1
|
|
2507
|
-
: a.planned.ref > b.planned.ref
|
|
2508
|
-
? 1
|
|
2509
|
-
: 0);
|
|
2510
|
-
const candidatePool = candidates.length;
|
|
2511
|
-
const selected = candidates.slice(0, replayBudget);
|
|
2512
|
-
const newReplayRefs = [];
|
|
2513
|
-
for (const { planned } of selected) {
|
|
2514
|
-
const ref = planned.ref;
|
|
2515
|
-
replayRefSet.add(ref);
|
|
2516
|
-
newReplayRefs.push({
|
|
2517
|
-
...planned,
|
|
2518
|
-
eligibilitySource: "replay",
|
|
2519
|
-
});
|
|
2520
|
-
// Seed the salienceMap so the sort/effectiveScore can rank the replay ref.
|
|
2521
|
-
if (!salienceMap.has(ref)) {
|
|
2522
|
-
salienceMap.set(ref, {
|
|
2523
|
-
encoding: 0,
|
|
2524
|
-
outcome: 0,
|
|
2525
|
-
retrieval: 0,
|
|
2526
|
-
rankScore: allRankScores.get(ref) ?? 0,
|
|
2527
|
-
});
|
|
2528
|
-
}
|
|
2529
|
-
}
|
|
2530
|
-
if (newReplayRefs.length > 0) {
|
|
2531
|
-
mergedRefs = dedupeRefs([...mergedRefs, ...newReplayRefs]);
|
|
2532
|
-
// Replay is the WEAKEST lane: stamp 'replay' ONLY for refs not already
|
|
2533
|
-
// keyed by a stronger lane.
|
|
2534
|
-
for (const ref of replayRefSet) {
|
|
2535
|
-
if (!eligibilitySourceByRef.has(ref))
|
|
2536
|
-
eligibilitySourceByRef.set(ref, "replay");
|
|
2537
|
-
}
|
|
2538
|
-
for (const r of mergedRefs) {
|
|
2539
|
-
r.eligibilitySource = eligibilitySourceByRef.get(r.ref) ?? "unknown";
|
|
2540
|
-
}
|
|
2541
|
-
}
|
|
2542
|
-
// Aggregated observability event (never per-ref).
|
|
2543
|
-
if (persist) {
|
|
2544
|
-
appendEvent({
|
|
2545
|
-
eventType: "improve_replay_selected",
|
|
2546
|
-
ref: undefined,
|
|
2547
|
-
metadata: {
|
|
2548
|
-
count: newReplayRefs.length,
|
|
2549
|
-
budget: replayBudget,
|
|
2550
|
-
convergedSkipped,
|
|
2551
|
-
candidatePool,
|
|
2552
|
-
},
|
|
2553
|
-
}, eventsCtx);
|
|
2554
|
-
}
|
|
2555
|
-
}, { path: eventsCtx?.dbPath, borrowed: eventsCtx?.db });
|
|
2556
|
-
}
|
|
2557
|
-
catch (err) {
|
|
2558
|
-
rethrowIfTestIsolationError(err);
|
|
2559
|
-
// best-effort: if DB unavailable, replayRefSet stays empty
|
|
1068
|
+
if (args.pendingForgettingRefs.length === 0 || args.scope.mode === "ref" || !args.allowFallbacks)
|
|
1069
|
+
return mergedRefs;
|
|
1070
|
+
const present = new Set(mergedRefs.map((r) => r.ref));
|
|
1071
|
+
const byRef = new Map(args.eligibleRefs.map((c) => [c.ref, c]));
|
|
1072
|
+
const byItemRef = new Map(args.eligibleRefs.flatMap((c) => (c.itemRef ? [[c.itemRef, c]] : [])));
|
|
1073
|
+
const added = [];
|
|
1074
|
+
const forgetting = new Set();
|
|
1075
|
+
for (const stored of args.pendingForgettingRefs) {
|
|
1076
|
+
// A qualified spelling must match this plan's exact item_ref.
|
|
1077
|
+
const candidate = byItemRef.get(stored) ?? (stored.includes("//") ? undefined : byRef.get(stripBundle(stored)));
|
|
1078
|
+
if (!candidate || forgetting.has(candidate.ref))
|
|
1079
|
+
continue;
|
|
1080
|
+
forgetting.add(candidate.ref);
|
|
1081
|
+
if (!present.has(candidate.ref)) {
|
|
1082
|
+
added.push(candidate);
|
|
1083
|
+
present.add(candidate.ref);
|
|
2560
1084
|
}
|
|
2561
1085
|
}
|
|
2562
|
-
|
|
1086
|
+
if (added.length > 0)
|
|
1087
|
+
mergedRefs = dedupeRefs([...mergedRefs, ...added]);
|
|
1088
|
+
if (forgetting.size === 0)
|
|
1089
|
+
return mergedRefs;
|
|
1090
|
+
for (const r of args.highSalienceRefs)
|
|
1091
|
+
eligibilitySourceByRef.set(r.ref, "high-salience");
|
|
1092
|
+
for (const r of args.proactiveRefs)
|
|
1093
|
+
eligibilitySourceByRef.set(r.ref, "proactive");
|
|
1094
|
+
for (const ref of forgetting)
|
|
1095
|
+
eligibilitySourceByRef.set(ref, "forgetting-safety");
|
|
1096
|
+
for (const r of args.signalFiltered)
|
|
1097
|
+
eligibilitySourceByRef.set(r.ref, "signal-delta");
|
|
1098
|
+
for (const r of mergedRefs)
|
|
1099
|
+
r.eligibilitySource = eligibilitySourceByRef.get(r.ref) ?? "unknown";
|
|
1100
|
+
return mergedRefs;
|
|
2563
1101
|
}
|
|
2564
|
-
/**
|
|
2565
|
-
async function dropRefsMissingOnDisk(
|
|
2566
|
-
const
|
|
2567
|
-
|
|
2568
|
-
// set — i.e. the genuinely processable refs in priority order. Note: this is
|
|
2569
|
-
// a semantic shift from earlier code where actionableRefs was the pre-cooldown
|
|
2570
|
-
// sorted set; the new meaning matches reality and is documented on
|
|
2571
|
-
// ImprovePreparationResult.actionableRefs.
|
|
2572
|
-
//
|
|
2573
|
-
// Final guard: drop any candidate whose backing file is no longer on disk.
|
|
2574
|
-
// Phase 1 validation captures missing files at the start of preparation, but
|
|
2575
|
-
// the gap between that check and dispatch can be minutes on large stashes —
|
|
2576
|
-
// long enough for a checkpoint / git checkout / external cleanup to delete
|
|
2577
|
-
// the asset. Empirically (improve-critical-review 2026-05-20) the single
|
|
2578
|
-
// biggest reject category was "Asset no longer exists on disk" (604/1407 =
|
|
2579
|
-
// 43%), meaning reflect/distill was producing proposals against deleted refs.
|
|
2580
|
-
// A cheap existsSync per surviving candidate eliminates that wasted work.
|
|
2581
|
-
const assetMissingOnDisk = [];
|
|
2582
|
-
const existsCheckedActionable = [];
|
|
1102
|
+
/** Drop candidates whose file vanished since planning, with one aggregate event. */
|
|
1103
|
+
async function dropRefsMissingOnDisk(sorted, options, eventsCtx, persist) {
|
|
1104
|
+
const actionableRefs = [];
|
|
1105
|
+
const missing = [];
|
|
2583
1106
|
for (const candidate of sorted) {
|
|
2584
|
-
// #591: prefer the path pre-resolved at planning time (synchronous
|
|
2585
|
-
// existsSync) over a serial async DB lookup per ref.
|
|
2586
1107
|
const filePath = candidate.filePath && fs.existsSync(candidate.filePath)
|
|
2587
1108
|
? candidate.filePath
|
|
2588
1109
|
: await findAssetFilePath(candidate.ref, options.stashDir);
|
|
2589
|
-
if (filePath && fs.existsSync(filePath))
|
|
2590
|
-
|
|
2591
|
-
|
|
2592
|
-
|
|
2593
|
-
assetMissingOnDisk.push(candidate.ref);
|
|
2594
|
-
}
|
|
1110
|
+
if (filePath && fs.existsSync(filePath))
|
|
1111
|
+
actionableRefs.push(candidate);
|
|
1112
|
+
else
|
|
1113
|
+
missing.push(candidate.ref);
|
|
2595
1114
|
}
|
|
2596
|
-
|
|
2597
|
-
|
|
2598
|
-
|
|
2599
|
-
|
|
2600
|
-
|
|
2601
|
-
|
|
2602
|
-
ref: undefined,
|
|
2603
|
-
metadata: {
|
|
2604
|
-
reason: "asset_missing_on_disk",
|
|
2605
|
-
count: assetMissingOnDisk.length,
|
|
2606
|
-
refs: assetMissingOnDisk.slice(0, 50),
|
|
2607
|
-
},
|
|
2608
|
-
}, eventsCtx);
|
|
1115
|
+
if (persist && missing.length > 0) {
|
|
1116
|
+
recordImproveSkip(eventsCtx, undefined, {
|
|
1117
|
+
reason: "asset_missing_on_disk",
|
|
1118
|
+
count: missing.length,
|
|
1119
|
+
refs: missing.slice(0, 50),
|
|
1120
|
+
});
|
|
2609
1121
|
}
|
|
2610
|
-
return { actionableRefs
|
|
1122
|
+
return { actionableRefs, missing };
|
|
2611
1123
|
}
|