akm-cli 0.9.17-alpha.2 → 0.9.17-alpha.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +756 -0
- package/dist/akm +94 -196
- package/dist/cli/shared.js +6 -2
- package/dist/cli.js +22 -9
- package/dist/commands/agent/agent-dispatch.js +1 -1
- package/dist/commands/command/command-execution.js +24 -62
- package/dist/commands/feedback-cli.js +0 -1
- package/dist/commands/health/accept-rate.js +2 -2
- package/dist/commands/health/checks.js +30 -75
- package/dist/commands/health/config-skew.js +38 -0
- package/dist/commands/health/egress.js +54 -0
- package/dist/commands/health/html-report.js +0 -38
- package/dist/commands/health/improve-metrics.js +123 -562
- package/dist/commands/health/plugin-staleness.js +53 -3
- package/dist/commands/health/renderers.js +12 -4
- package/dist/commands/health/report-view-model.js +11 -106
- package/dist/commands/health/types-improve.js +4 -19
- package/dist/commands/health/windows.js +64 -73
- package/dist/commands/health.js +122 -143
- package/dist/commands/improve/consolidate/chunking.js +25 -100
- package/dist/commands/improve/consolidate/sanitize.js +54 -149
- package/dist/commands/improve/consolidate.js +538 -1075
- package/dist/commands/improve/content-hash.js +16 -24
- package/dist/commands/improve/distill/content-repair.js +18 -100
- package/dist/commands/improve/distill-guards.js +20 -81
- package/dist/commands/improve/distill-promotion-policy.js +23 -243
- package/dist/commands/improve/distill.js +608 -1075
- package/dist/commands/improve/eligibility.js +126 -400
- package/dist/commands/improve/execution.js +3 -5
- package/dist/commands/improve/extract.js +487 -1046
- package/dist/commands/improve/feedback-valence.js +0 -25
- package/dist/commands/improve/improve-cli.js +29 -166
- package/dist/commands/improve/improve-result-file.js +10 -66
- package/dist/commands/improve/improve-strategies.js +12 -7
- package/dist/commands/improve/improve-usage-report.js +18 -64
- package/dist/commands/improve/improve.js +443 -1063
- package/dist/commands/improve/ledger.js +114 -0
- package/dist/commands/improve/locks.js +2 -8
- package/dist/commands/improve/loop-stages.js +459 -1172
- package/dist/commands/improve/memory/derived-ref.js +12 -77
- package/dist/commands/improve/memory/memory-belief.js +14 -118
- package/dist/commands/improve/memory/memory-improve.js +4 -3
- package/dist/commands/improve/outcome-loop.js +28 -156
- package/dist/commands/improve/planner.js +5 -10
- package/dist/commands/improve/preparation.js +851 -2339
- package/dist/commands/improve/proactive-maintenance.js +34 -101
- package/dist/commands/improve/reflect-noise.js +104 -280
- package/dist/commands/improve/reflect.js +621 -1367
- package/dist/commands/improve/salience.js +46 -232
- package/dist/commands/improve/session-asset.js +19 -100
- package/dist/commands/improve/stage.js +323 -0
- package/dist/commands/proposal/drain.js +251 -644
- package/dist/commands/proposal/proposal-cli.js +3 -18
- package/dist/commands/proposal/proposal-types.js +20 -41
- package/dist/commands/proposal/proposal.js +1 -2
- package/dist/commands/proposal/propose.js +134 -160
- package/dist/commands/proposal/repository.js +502 -1487
- package/dist/commands/proposal/validators/proposal-quality-validators.js +71 -174
- package/dist/commands/proposal/validators/proposal-validators.js +1 -1
- package/dist/commands/proposal/validators/proposals.js +13 -89
- package/dist/commands/read/curate.js +63 -413
- package/dist/commands/read/search-cli.js +16 -33
- package/dist/commands/read/search.js +17 -23
- package/dist/commands/read/show.js +2 -13
- package/dist/commands/sources/bundle-cli.js +25 -2
- package/dist/commands/sources/bundle-config-ops.js +7 -0
- package/dist/commands/sources/dangerous-env-audit.js +1 -2
- package/dist/commands/sources/info.js +2 -11
- package/dist/commands/sources/installed-stashes.js +197 -746
- package/dist/commands/sources/schema-repair.js +98 -129
- package/dist/commands/sources/source-add.js +62 -12
- package/dist/commands/sources/stash-cli.js +1 -1
- package/dist/commands/tasks/explain.js +10 -13
- package/dist/commands/tasks/tasks-cli.js +9 -8
- package/dist/commands/tasks/tasks.js +326 -930
- package/dist/commands/tasks/validate.js +42 -21
- package/dist/commands/workflow/plan.js +22 -29
- package/dist/commands/workflow-cli.js +4 -4
- package/dist/core/adapter/adapters/akm-adapter.js +0 -1
- package/dist/core/adapter/adapters/akm-lint.js +2 -3
- package/dist/core/adapter/adapters/akm-metadata.js +11 -12
- package/dist/core/adapter/adapters/akm-workflow-adapter.js +1 -1
- package/dist/core/adapter/execution-source.js +17 -29
- package/dist/core/asset/resolve-ref.js +1 -1
- package/dist/core/bundle-id.js +42 -5
- package/dist/core/bundle-rename.js +291 -0
- package/dist/core/config/config-io.js +1 -2
- package/dist/core/config/config-schema.js +1 -33
- package/dist/core/config/config-walker.js +1 -1
- package/dist/core/config/config.js +163 -68
- package/dist/core/config/legacy-source-shape-shim.js +38 -9
- package/dist/core/config/schema/embedding.js +20 -5
- package/dist/core/config/schema/engines.js +5 -0
- package/dist/core/config/schema/execution.js +1 -1
- package/dist/core/config/schema/experimental.js +1 -1
- package/dist/core/config/schema/improve-processes.js +21 -95
- package/dist/core/config/schema/improve.js +4 -42
- package/dist/core/config/schema/scheduler.js +12 -12
- package/dist/core/config/schema/search.js +6 -22
- package/dist/core/env-secret-ref.js +0 -1
- package/dist/core/errors.js +8 -9
- package/dist/core/file-lock.js +76 -173
- package/dist/core/logs-db.js +2 -2
- package/dist/core/paths.js +0 -27
- package/dist/core/redaction.js +109 -2
- package/dist/core/run-lock.js +2 -5
- package/dist/core/spawn-env.js +1 -1
- package/dist/core/state/migrations.js +108 -61
- package/dist/core/state-db-scope.js +2 -4
- package/dist/core/state-db.js +126 -692
- package/dist/core/type-presentation.js +1 -9
- package/dist/core/write-source.js +293 -1012
- package/dist/execution/input-contract.js +1 -1
- package/dist/execution/resolved-request.js +135 -689
- package/dist/execution/source.js +63 -257
- package/dist/execution/target-ref.js +1 -1
- package/dist/indexer/bundle-identity-guard.js +2 -2
- package/dist/indexer/db/graph-db.js +106 -46
- package/dist/indexer/ensure-index.js +44 -85
- package/dist/indexer/graph/graph-extraction.js +340 -562
- package/dist/indexer/graph/graph-related.js +130 -0
- package/dist/indexer/index-rebuild-lock.js +3 -11
- package/dist/indexer/index-writer-lock.js +8 -17
- package/dist/indexer/index-written-assets.js +139 -151
- package/dist/indexer/indexer.js +524 -846
- package/dist/indexer/materialize-embeddings.js +60 -397
- package/dist/indexer/passes/memory-inference.js +81 -90
- package/dist/indexer/passes/metadata.js +132 -200
- package/dist/indexer/read-preflight.js +0 -7
- package/dist/indexer/scan/doc-to-entry.js +1 -3
- package/dist/indexer/scan/drain-dir.js +1 -1
- package/dist/indexer/search/db-search.js +181 -590
- package/dist/indexer/search/fts-query.js +30 -41
- package/dist/indexer/search/ranking.js +28 -154
- package/dist/indexer/search/search-attribution.js +12 -32
- package/dist/indexer/search/search-fields.js +11 -15
- package/dist/indexer/search/search-hit-enrichers.js +54 -85
- package/dist/indexer/search/search-source.js +1 -4
- package/dist/indexer/usage/usage-events.js +2 -7
- package/dist/integrations/agent/engine-fallback.js +23 -40
- package/dist/integrations/agent/engine-resolution.js +93 -183
- package/dist/integrations/agent/execution.js +507 -0
- package/dist/integrations/agent/model-map.js +28 -156
- package/dist/integrations/agent/request-lowering.js +66 -141
- package/dist/integrations/agent/runner-dispatch.js +143 -321
- package/dist/integrations/agent/runner.js +54 -14
- package/dist/integrations/lockfile.js +53 -101
- package/dist/llm/embedders/deterministic.js +2 -3
- package/dist/llm/embedders/profile.js +71 -0
- package/dist/llm/embedders/remote.js +10 -15
- package/dist/llm/graph-extract.js +3 -12
- package/dist/llm/index-passes.js +3 -5
- package/dist/llm/memory-infer.js +1 -2
- package/dist/llm/metadata-enhance.js +1 -2
- package/dist/llm/structured-call.js +5 -24
- package/dist/output/generic-render.js +23 -11
- package/dist/output/html-render.js +13 -10
- package/dist/output/render-registry.js +3 -32
- package/dist/output/shapes/helpers.js +2 -34
- package/dist/output/shapes/passthrough.js +1 -9
- package/dist/{indexer/search/ranking-types.js → output/text/bundle-rename.js} +4 -1
- package/dist/output/text/command-format.js +60 -23
- package/dist/output/text/helpers.js +1 -1
- package/dist/output/text/migrate.js +5 -14
- package/dist/output/text/proposal-format.js +1 -2
- package/dist/output/text/workflow-format.js +0 -32
- package/dist/output/text.js +2 -0
- package/dist/registry/factory.js +4 -19
- package/dist/registry/network.js +66 -220
- package/dist/registry/providers/index.js +0 -2
- package/dist/registry/providers/skills-sh.js +3 -14
- package/dist/registry/providers/static-index.js +24 -26
- package/dist/registry/resolve.js +55 -131
- package/dist/scripts/akm-migrate-node.js +43937 -93313
- package/dist/scripts/akm-migrate.js +43697 -93071
- package/dist/setup/registry-stash-loader.js +4 -13
- package/dist/setup/semantic-assets.js +3 -44
- package/dist/setup/setup.js +1 -1
- package/dist/setup/steps/tasks.js +25 -15
- package/dist/sources/provider-factory.js +17 -18
- package/dist/sources/providers/filesystem.js +2 -3
- package/dist/sources/providers/git-install.js +7 -1
- package/dist/sources/providers/git-provider.js +0 -3
- package/dist/sources/providers/git-stash.js +0 -17
- package/dist/sources/providers/npm.js +2 -4
- package/dist/sources/providers/provider-utils.js +5 -10
- package/dist/sources/providers/website.js +0 -2
- package/dist/sources/snapshot-fetchers/website-ingest.js +1 -1
- package/dist/sources/website-url.js +2 -2
- package/dist/storage/database.js +9 -35
- package/dist/storage/repositories/improve-ledger-repository.js +168 -0
- package/dist/storage/repositories/index-connection.js +34 -70
- package/dist/storage/repositories/index-entries-repository.js +69 -111
- package/dist/storage/repositories/index-entry-mapper.js +1 -2
- package/dist/storage/repositories/index-entry-schema.js +83 -269
- package/dist/storage/repositories/index-fts-repository.js +86 -256
- package/dist/storage/repositories/index-llm-cache-repository.js +17 -0
- package/dist/storage/repositories/index-meta-repository.js +6 -4
- package/dist/storage/repositories/index-schema.js +192 -220
- package/dist/storage/repositories/index-utility-repository.js +8 -29
- package/dist/storage/repositories/index-vec-repository.js +133 -414
- package/dist/storage/repositories/outcome-repository.js +2 -1
- package/dist/storage/repositories/proposals-repository.js +35 -0
- package/dist/storage/repositories/registry-index-cache-repository.js +100 -0
- package/dist/storage/repositories/task-history-repository.js +26 -4
- package/dist/storage/repositories/workflow-runs-repository.js +53 -244
- package/dist/storage/sqlite-migrations.js +136 -0
- package/dist/storage/sqlite-pragmas.js +11 -9
- package/dist/storage/sqlite-transaction.js +170 -0
- package/dist/storage/state-db-integrity.js +34 -27
- package/dist/tasks/activation-config.js +134 -62
- package/dist/tasks/backends/cron.js +129 -277
- package/dist/tasks/backends/exec-utils.js +2 -5
- package/dist/tasks/backends/launchd.js +125 -745
- package/dist/tasks/backends/schtasks.js +101 -620
- package/dist/tasks/prepare/prepare-support.js +5 -15
- package/dist/tasks/prepare/prepare.js +0 -2
- package/dist/tasks/resolve-akm-bin.js +20 -79
- package/dist/tasks/run/attempt-lifecycle.js +0 -1
- package/dist/tasks/scheduler-binding.js +18 -238
- package/dist/tasks/scheduler-invocation.js +52 -52
- package/dist/tasks/scheduler-lock.js +53 -0
- package/dist/tasks/scheduler-sync.js +363 -679
- package/dist/tasks/source/parse-task-source.js +160 -10
- package/dist/tasks/source/task-source-v3-frozen.js +3 -4
- package/dist/tasks/source/task-to-v4.js +2 -2
- package/dist/workflows/authoring/authoring.js +3 -12
- package/dist/workflows/compile.js +211 -0
- package/dist/workflows/concurrency-policy.js +13 -74
- package/dist/workflows/exec/child-invocation.js +3 -17
- package/dist/workflows/exec/child-workflow.js +32 -141
- package/dist/workflows/exec/dispatch-redaction.js +13 -53
- package/dist/workflows/exec/environment.js +98 -0
- package/dist/workflows/exec/exec-unit.js +33 -140
- package/dist/workflows/exec/frozen-judge.js +7 -59
- package/dist/workflows/exec/native-executor.js +82 -341
- package/dist/workflows/exec/param-secrets.js +29 -47
- package/dist/workflows/exec/run-workflow.js +154 -387
- package/dist/workflows/exec/scheduler.js +9 -36
- package/dist/workflows/exec/step-work.js +127 -430
- package/dist/workflows/exec/unit-dispatch.js +11 -63
- package/dist/workflows/exec/unit-writer.js +8 -52
- package/dist/workflows/exec/worktree.js +39 -273
- package/dist/workflows/freeze/child-output-references.js +4 -15
- package/dist/workflows/freeze/environment.js +99 -92
- package/dist/workflows/freeze/freeze.js +172 -0
- package/dist/workflows/freeze/step-values.js +19 -21
- package/dist/workflows/freeze/targets/child-workflow.js +23 -92
- package/dist/workflows/freeze/targets/command.js +10 -33
- package/dist/workflows/freeze/targets/script.js +5 -12
- package/dist/workflows/freeze/targets/shell.js +3 -6
- package/dist/workflows/freeze/targets/task.js +25 -80
- package/dist/workflows/freeze/task-bindings.js +20 -67
- package/dist/workflows/{source-ir/github-yaml.js → github-yaml.js} +88 -206
- package/dist/workflows/ir/params.js +6 -51
- package/dist/workflows/ir/plan-hash.js +2 -34
- package/dist/workflows/parser.js +140 -43
- package/dist/{commands/improve/consolidate/types.js → workflows/plan.js} +2 -1
- package/dist/workflows/renderer.js +36 -69
- package/dist/workflows/resource-limits.js +12 -120
- package/dist/workflows/runtime/agent-identity.js +8 -40
- package/dist/workflows/runtime/run-outputs.js +3 -6
- package/dist/workflows/runtime/run-plan.js +316 -0
- package/dist/workflows/runtime/runs.js +48 -200
- package/dist/workflows/runtime/workflow-asset-loader.js +24 -57
- package/dist/workflows/{source-ir/semantics.js → source-semantics.js} +16 -20
- package/dist/workflows/validate-summary.js +2 -7
- package/docs/integration/bundling-akm.md +49 -42
- package/docs/migration/README.md +1 -0
- package/docs/migration/release-notes/0.9.17.md +41 -0
- package/docs/migration/v0.9.1-to-v0.9.2.md +19 -7
- package/docs/reference/cli.md +182 -125
- package/docs/reference/configuration.md +49 -56
- package/docs/reference/data-and-telemetry.md +19 -20
- package/docs/reference/tasks.md +86 -38
- package/docs/reference/workflow-schema.md +14 -18
- package/docs/reference/workflows.md +6 -9
- package/package.json +1 -1
- package/schemas/akm-config.json +87 -406
- package/dist/commands/health/advisories.js +0 -150
- package/dist/commands/health/metrics.js +0 -329
- package/dist/commands/health/surfaces.js +0 -102
- package/dist/commands/improve/anti-collapse.js +0 -83
- package/dist/commands/improve/collapse-detector.js +0 -432
- package/dist/commands/improve/consolidate/eligibility.js +0 -48
- package/dist/commands/improve/consolidate/merge.js +0 -146
- package/dist/commands/improve/distill/promote-memory.js +0 -329
- package/dist/commands/improve/distill/quality-gate.js +0 -500
- package/dist/commands/improve/memory/memory-contradiction-detect.js +0 -291
- package/dist/commands/improve/proposal-envelope.js +0 -31
- package/dist/commands/improve/run-context.js +0 -123
- package/dist/commands/improve/shared.js +0 -21
- package/dist/commands/improve/source-identity.js +0 -28
- package/dist/commands/improve/triage.js +0 -96
- package/dist/commands/proposal/drain-policies.js +0 -151
- package/dist/commands/sources/update-transaction.js +0 -220
- package/dist/core/action-contributors.js +0 -28
- package/dist/core/config/config-version-shim.js +0 -101
- package/dist/core/config/retired-experimental-keys-shim.js +0 -62
- package/dist/core/fs-txn.js +0 -405
- package/dist/core/lexical-score.js +0 -25
- package/dist/core/maintenance-barrier.js +0 -167
- package/dist/execution/executable-identity.js +0 -105
- package/dist/execution/guarded-source.js +0 -427
- package/dist/indexer/graph/graph-boost.js +0 -427
- package/dist/indexer/graph/graph-dedup.js +0 -95
- package/dist/indexer/search/name-match.js +0 -35
- package/dist/indexer/search/ranking-contributors.js +0 -515
- package/dist/indexer/walk/project-context.js +0 -192
- package/dist/integrations/agent/execution-cascade.js +0 -566
- package/dist/integrations/agent/execution-definitions.js +0 -202
- package/dist/integrations/agent/execution-lowering.js +0 -841
- package/dist/integrations/agent/execution-preparation.js +0 -98
- package/dist/integrations/agent/inline-execution.js +0 -74
- package/dist/registry/create-provider-registry.js +0 -29
- package/dist/registry/pinned-request-helper.js +0 -247
- package/dist/registry/pinned-transport.js +0 -717
- package/dist/sources/providers/index.js +0 -14
- package/dist/storage/engines/sqlite-migrations.js +0 -271
- package/dist/storage/repositories/canaries-repository.js +0 -107
- package/dist/storage/repositories/embedding-salvage-repository.js +0 -184
- package/dist/storage/repositories/registry-cache.js +0 -113
- package/dist/tasks/scheduler-sync-preview.js +0 -52
- package/dist/workflows/freeze/resolve-steps.js +0 -86
- package/dist/workflows/freeze/source-freeze.js +0 -64
- package/dist/workflows/ir/compile.js +0 -321
- package/dist/workflows/ir/environment-v4.js +0 -330
- package/dist/workflows/ir/freeze-v4.js +0 -153
- package/dist/workflows/ir/schema-v4.js +0 -745
- package/dist/workflows/ir/schema.js +0 -354
- package/dist/workflows/program/schema.js +0 -78
- package/dist/workflows/runtime/checkin.js +0 -57
- package/dist/workflows/runtime/plan-classifier.js +0 -196
- package/dist/workflows/runtime/unit-checkin.js +0 -45
- package/dist/workflows/runtime/unit-phases.js +0 -20
- package/dist/workflows/schema.js +0 -4
- package/dist/workflows/source-ir/compile.js +0 -200
- package/dist/workflows/source-ir/program.js +0 -50
- package/dist/workflows/source-ir/result.js +0 -26
- package/dist/workflows/source-ir/schema.js +0 -786
- package/dist/workflows/source-ir/triggers.js +0 -79
- package/dist/workflows/source-ir/uses.js +0 -40
- package/dist/workflows/validator.js +0 -60
|
@@ -1,13 +1,14 @@
|
|
|
1
1
|
// This Source Code Form is subject to the terms of the Mozilla Public
|
|
2
2
|
// License, v. 2.0. If a copy of the MPL was not distributed with this
|
|
3
3
|
// file, You can obtain one at https://mozilla.org/MPL/2.0/.
|
|
4
|
+
/** The improve loop (reflect + distill per ref), the post-loop checks and the maintenance passes. */
|
|
4
5
|
import fs from "node:fs";
|
|
5
6
|
import path from "node:path";
|
|
6
7
|
import { parseRefInput } from "../../core/asset/resolve-ref.js";
|
|
7
8
|
import { daysToMs } from "../../core/common.js";
|
|
8
|
-
import { DEFAULT_GRAPH_EXTRACTION_BATCH_SIZE, loadConfig } from "../../core/config/config.js";
|
|
9
|
+
import { DEFAULT_GRAPH_EXTRACTION_BATCH_SIZE, loadConfig, } from "../../core/config/config.js";
|
|
9
10
|
import { UsageError } from "../../core/errors.js";
|
|
10
|
-
import { appendEvent
|
|
11
|
+
import { appendEvent } from "../../core/events.js";
|
|
11
12
|
import { openLogsDatabase, purgeOldTaskLogs } from "../../core/logs-db.js";
|
|
12
13
|
import { getDbPath, getTaskLogDir } from "../../core/paths.js";
|
|
13
14
|
import { withStateDb } from "../../core/state-db.js";
|
|
@@ -18,100 +19,81 @@ import { deriveWritableBundleIds } from "../../indexer/installations.js";
|
|
|
18
19
|
import { collectPendingMemories, runMemoryInferencePass, } from "../../indexer/passes/memory-inference.js";
|
|
19
20
|
import { resolveSourceEntries } from "../../indexer/search/search-source.js";
|
|
20
21
|
import { isProcessEnabled } from "../../llm/feature-gate.js";
|
|
21
|
-
import { withLlmStage } from "../../llm/usage-telemetry.js";
|
|
22
|
-
import { purgeOldCycleMetrics } from "../../storage/repositories/canaries-repository.js";
|
|
23
22
|
import { purgeOldEvents } from "../../storage/repositories/events-repository.js";
|
|
24
23
|
import { purgeOldImproveRuns } from "../../storage/repositories/improve-runs-repository.js";
|
|
25
24
|
import { closeDatabase, openIndexDatabase } from "../../storage/repositories/index-connection.js";
|
|
26
25
|
import { getLiveRefSnapshot, isRefLiveInSnapshot, } from "../../storage/repositories/index-entries-repository.js";
|
|
27
26
|
import { clearAssetOutcomeMissing, countAssetOutcomeMissing, deleteAssetOutcomeMissingBefore, listAssetOutcomeMissingState, stampAssetOutcomeMissing, } from "../../storage/repositories/outcome-repository.js";
|
|
28
27
|
import { clearAssetSalienceMissing, countAssetSalienceMissing, deleteAssetSalienceMissingBefore, listAssetSalienceMissingState, stampAssetSalienceMissing, } from "../../storage/repositories/salience-repository.js";
|
|
29
|
-
import { readFreelistInfo,
|
|
28
|
+
import { readFreelistInfo, STATE_DB_VACUUMED_EVENT, vacuumIfReclaimable } from "../../storage/state-db-integrity.js";
|
|
30
29
|
import { purgeOldTaskLogFiles } from "../../tasks/run/task-log.js";
|
|
31
|
-
import {
|
|
30
|
+
import { expireStaleProposals, purgeOrphanProposals } from "../proposal/repository.js";
|
|
32
31
|
import { checkDeadUrls } from "../url-checker.js";
|
|
33
|
-
import { DEFAULT_RETENTION_DAYS as CYCLE_METRICS_RETENTION_DAYS, runCollapseDetector } from "./collapse-detector.js";
|
|
34
|
-
import { defaultLookup, deriveLessonRef } from "./distill.js";
|
|
35
|
-
import { wouldPromoteMemoryToKnowledge } from "./distill/promote-memory.js";
|
|
36
|
-
import { deriveKnowledgeRef } from "./distill-promotion-policy.js";
|
|
37
|
-
// Eligibility / candidate-selection predicates live in ./eligibility.
|
|
38
32
|
import { findAssetFilePath, isDistillCandidateRef } from "./eligibility.js";
|
|
39
33
|
import { shouldSkipRef } from "./improve-strategies.js";
|
|
40
|
-
import {
|
|
34
|
+
import { recordLedgerAttempt, stateKey, stripBundle } from "./ledger.js";
|
|
35
|
+
import { pushRecentError } from "./preparation.js";
|
|
41
36
|
import { recordNoOp, resetConsecutiveNoOps } from "./salience.js";
|
|
42
|
-
import { errMessage } from "./
|
|
43
|
-
import { bareImproveRef, durableImproveRef } from "./source-identity.js";
|
|
44
|
-
// ── improve loop / post-loop / maintenance stages ───────────────────
|
|
45
|
-
// The cycle stages run by akmImprove, extracted from improve.ts.
|
|
46
|
-
/** O-5 / #378: rolling per-originator error-window cap. */
|
|
47
|
-
const RECENT_ERRORS_CAP = 3;
|
|
48
|
-
/** O-5 / #378: push a per-originator error into the rolling window. */
|
|
49
|
-
function pushRecentError(recentErrors, originator, msg) {
|
|
50
|
-
if (!recentErrors[originator])
|
|
51
|
-
recentErrors[originator] = [];
|
|
52
|
-
recentErrors[originator].push(msg);
|
|
53
|
-
if (recentErrors[originator].length > RECENT_ERRORS_CAP)
|
|
54
|
-
recentErrors[originator].shift();
|
|
55
|
-
}
|
|
56
|
-
/**
|
|
57
|
-
* Build the per-run loop environment from the run context: the derived guards
|
|
58
|
-
* and the pending-proposal preload.
|
|
59
|
-
*/
|
|
37
|
+
import { attributeStage, errMessage } from "./stage.js";
|
|
60
38
|
export function prepareImproveLoopEnv(args) {
|
|
61
|
-
const
|
|
62
|
-
|
|
63
|
-
// now RunContext's required eventsCtx / optional signal — renamed local
|
|
64
|
-
// aliases so the rest of this function (and the ImproveLoopEnv object
|
|
65
|
-
// literal below) is unchanged. `primaryStashDir` deliberately comes from
|
|
66
|
-
// the state's honest optional field, NOT `ctx.stashDir`: the rare
|
|
67
|
-
// unresolvable-primary path must keep skipping the `if (primaryStashDir)`
|
|
68
|
-
// guards below (see the field's doc in ./improve-run-types).
|
|
69
|
-
const primaryStashDir = args.primaryStashDir;
|
|
70
|
-
const eventsCtx = ctx.eventsCtx;
|
|
71
|
-
const budgetSignal = ctx.signal;
|
|
72
|
-
// O-1 (#364): compute remaining budget at call time so each sub-call
|
|
73
|
-
// receives only its fair share of the wall-clock budget.
|
|
74
|
-
const remainingBudgetMs = () => Math.max(0, budgetMs - (Date.now() - startMs));
|
|
75
|
-
// Build a Set for O(1) membership test — these refs skip the reflect call (Bug D2).
|
|
76
|
-
const distillOnlyRefSet = new Set(distillOnlyRefs.map((r) => r.ref));
|
|
77
|
-
// requirePlannedRefs guard: when the distill profile sets this flag, skip
|
|
78
|
-
// distill for distill-only refs if the reflect phase produced no planned refs.
|
|
79
|
-
// Prevents the distill loop from generating hundreds of distill-skipped events
|
|
80
|
-
// on quiet passes (all refs on reflect cooldown, no new signal to distill).
|
|
81
|
-
const requirePlannedRefs = improveProfile?.processes?.distill?.requirePlannedRefs === true;
|
|
82
|
-
const hasReflectEligibleRefs = loopRefs.some((r) => !distillOnlyRefSet.has(r.ref));
|
|
83
|
-
const skipDistillDueToRequirePlannedRefs = requirePlannedRefs && !hasReflectEligibleRefs;
|
|
84
|
-
// Pre-load all pending proposals once instead of querying per asset in the loop.
|
|
85
|
-
const dedupeStashDirForProposals = primaryStashDir ?? options.stashDir;
|
|
86
|
-
const pendingProposalRefSet = new Set(dedupeStashDirForProposals
|
|
87
|
-
? listProposals(dedupeStashDirForProposals, { status: "pending" }).map((p) => p.ref)
|
|
88
|
-
: []);
|
|
39
|
+
const distillOnlyRefSet = new Set(args.distillOnlyRefs.map((r) => r.ref));
|
|
40
|
+
const requirePlannedRefs = args.improveProfile?.processes?.distill?.requirePlannedRefs === true;
|
|
89
41
|
return {
|
|
90
|
-
scope,
|
|
91
|
-
options,
|
|
92
|
-
primaryStashDir,
|
|
93
|
-
reflectFn,
|
|
94
|
-
distillFn,
|
|
95
|
-
signalBearingSet,
|
|
96
|
-
distillCooledRefs,
|
|
42
|
+
scope: args.scope,
|
|
43
|
+
options: args.options,
|
|
44
|
+
primaryStashDir: args.primaryStashDir,
|
|
45
|
+
reflectFn: args.reflectFn,
|
|
46
|
+
distillFn: args.distillFn,
|
|
47
|
+
signalBearingSet: args.signalBearingSet,
|
|
48
|
+
distillCooledRefs: args.distillCooledRefs,
|
|
97
49
|
distillOnlyRefSet,
|
|
98
|
-
recentErrors,
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
pendingProposalRefSet,
|
|
106
|
-
remainingBudgetMs,
|
|
50
|
+
recentErrors: args.recentErrors,
|
|
51
|
+
eventsCtx: args.eventsCtx,
|
|
52
|
+
improveProfile: args.improveProfile,
|
|
53
|
+
resolvedPlan: args.resolvedPlan,
|
|
54
|
+
budgetSignal: args.budgetSignal,
|
|
55
|
+
skipDistillDueToRequirePlannedRefs: requirePlannedRefs && args.loopRefs.every((r) => distillOnlyRefSet.has(r.ref)),
|
|
56
|
+
remainingBudgetMs: () => Math.max(0, args.budgetMs - (Date.now() - args.startMs)),
|
|
107
57
|
};
|
|
108
58
|
}
|
|
109
59
|
/**
|
|
110
|
-
*
|
|
111
|
-
*
|
|
112
|
-
* `continue` in the old inline loop body is an early `return` inside the
|
|
113
|
-
* passes; the orchestrator folds the returned tally and owns the run counters.
|
|
60
|
+
* Record a loop attempt in the improve ledger. Proposals and quality
|
|
61
|
+
* rejections record themselves; the loop records what they cannot see.
|
|
114
62
|
*/
|
|
63
|
+
function recordLoopAttempt(planned, env, source, outcome, detail) {
|
|
64
|
+
const stashDir = env.primaryStashDir ?? env.options.stashDir;
|
|
65
|
+
if (!stashDir || env.options.dryRun)
|
|
66
|
+
return;
|
|
67
|
+
recordLedgerAttempt({ eventsCtx: env.eventsCtx }, {
|
|
68
|
+
stashDir,
|
|
69
|
+
ref: stateKey(planned.ref, planned.itemRef),
|
|
70
|
+
source,
|
|
71
|
+
outcome,
|
|
72
|
+
...(detail !== undefined ? { detail } : {}),
|
|
73
|
+
});
|
|
74
|
+
}
|
|
75
|
+
/** Plasticity counter: repeated no-ops dampen an asset's selection score; a change lifts it. */
|
|
76
|
+
function recordPlasticity(env, planned, outcome) {
|
|
77
|
+
const db = env.eventsCtx?.db;
|
|
78
|
+
if (!db || !outcome)
|
|
79
|
+
return;
|
|
80
|
+
const key = stateKey(planned.ref, planned.itemRef);
|
|
81
|
+
try {
|
|
82
|
+
if (outcome === "noop")
|
|
83
|
+
recordNoOp(db, key);
|
|
84
|
+
else
|
|
85
|
+
resetConsecutiveNoOps(db, key);
|
|
86
|
+
}
|
|
87
|
+
catch {
|
|
88
|
+
// best-effort
|
|
89
|
+
}
|
|
90
|
+
}
|
|
91
|
+
function recordSkip(tally, ref, reason, event) {
|
|
92
|
+
tally.actions.push({ ref, mode: "distill-skipped", result: { ok: true, reason } });
|
|
93
|
+
if (event)
|
|
94
|
+
appendEvent({ eventType: "improve_skipped", ref, metadata: { reason: event.reason } }, event.env.eventsCtx);
|
|
95
|
+
}
|
|
96
|
+
/** One loop iteration: reflect, then distill. A distill UsageError is a validation failure. */
|
|
115
97
|
export async function processImproveLoopRef(planned, env) {
|
|
116
98
|
const tally = {
|
|
117
99
|
actions: [],
|
|
@@ -120,20 +102,15 @@ export async function processImproveLoopRef(planned, env) {
|
|
|
120
102
|
memoryRefsForInference: [],
|
|
121
103
|
};
|
|
122
104
|
try {
|
|
123
|
-
// Bug D2: distillOnlyRefs skip the reflect call but still run the distill path.
|
|
124
|
-
// Bug D1: in-loop distill-cooldown check removed — distill-cooled candidates
|
|
125
|
-
// have their synthetic actions emitted in runImprovePreparationStage.
|
|
126
105
|
const isDistillOnly = env.distillOnlyRefSet.has(planned.ref);
|
|
127
|
-
const
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
await runLoopDistillPass(planned,
|
|
106
|
+
const parsed = parseRefInput(planned.ref);
|
|
107
|
+
if (!isDistillOnly)
|
|
108
|
+
await runLoopReflectPass(planned, env, tally);
|
|
109
|
+
await runLoopDistillPass(planned, parsed.type, isDistillOnly, env, tally);
|
|
131
110
|
}
|
|
132
111
|
catch (err) {
|
|
133
|
-
// B7: UsageError thrown by akmDistill on validation_failed should be recorded
|
|
134
|
-
// as mode:"distill" with outcome:"validation_failed", NOT as a generic error.
|
|
135
|
-
// The distill_invoked event was already emitted inside akmDistill before the throw.
|
|
136
112
|
if (err instanceof UsageError) {
|
|
113
|
+
recordLoopAttempt(planned, env, "distill", "failed", err.message);
|
|
137
114
|
tally.actions.push({
|
|
138
115
|
ref: planned.ref,
|
|
139
116
|
mode: "distill",
|
|
@@ -141,583 +118,177 @@ export async function processImproveLoopRef(planned, env) {
|
|
|
141
118
|
});
|
|
142
119
|
}
|
|
143
120
|
else {
|
|
144
|
-
tally.actions.push({
|
|
145
|
-
ref: planned.ref,
|
|
146
|
-
mode: "error",
|
|
147
|
-
result: { ok: false, error: errMessage(err) },
|
|
148
|
-
});
|
|
121
|
+
tally.actions.push({ ref: planned.ref, mode: "error", result: { ok: false, error: errMessage(err) } });
|
|
149
122
|
}
|
|
150
123
|
}
|
|
151
124
|
return tally;
|
|
152
125
|
}
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
*/
|
|
159
|
-
async function runLoopReflectPass(planned, isDistillOnly, env, tally) {
|
|
160
|
-
const { options, primaryStashDir, reflectFn, eventsCtx, improveProfile, resolvedPlan, budgetSignal } = env;
|
|
161
|
-
// B6: derived memories are machine-generated; skip reflect to avoid noisy proposals.
|
|
162
|
-
// shouldDistillMemoryRef already returns false for .derived refs, so the distill
|
|
163
|
-
// path is also a no-op for them — we just avoid unnecessary agent spawns.
|
|
164
|
-
// D2: distillOnlyRefs also skip the reflect call (reflect-cooled, distill path only).
|
|
165
|
-
if (!isDistillOnly && !planned.ref.endsWith(".derived")) {
|
|
166
|
-
// Type guard: skip reflect for unsupported types (script, env, task, etc.)
|
|
167
|
-
// and raw wiki directories, driven by the active improve profile.
|
|
168
|
-
const reflectSkip = shouldSkipRef(planned.ref, "reflect", improveProfile);
|
|
169
|
-
if (reflectSkip.skip) {
|
|
170
|
-
tally.actions.push({
|
|
171
|
-
ref: planned.ref,
|
|
172
|
-
mode: "reflect-skipped",
|
|
173
|
-
result: { ok: true, reason: reflectSkip.reason },
|
|
174
|
-
});
|
|
175
|
-
}
|
|
176
|
-
else {
|
|
177
|
-
// O-5 / #378: only inject reflect-originator errors into the reflect call.
|
|
178
|
-
// Cross-task errors (e.g. schema-repair) must NOT contaminate reflect prompts.
|
|
179
|
-
const reflectErrors = env.recentErrors.reflect ?? [];
|
|
180
|
-
if (reflectErrors.length > 0)
|
|
181
|
-
tally.reflectsWithErrorContext++;
|
|
182
|
-
// O-1 (#364): pass remaining budget as timeoutMs so the agent spawn is
|
|
183
|
-
// bounded by the wall-clock deadline rather than the default per-profile timeout.
|
|
184
|
-
const reflectBudgetMs = env.remainingBudgetMs();
|
|
185
|
-
const reflectEngine = resolvedPlan.processes.reflect.runner?.engine;
|
|
186
|
-
const reflectTarget = options.sourceName && primaryStashDir ? { source: options.sourceName, root: primaryStashDir } : undefined;
|
|
187
|
-
// Re-enter canonical named-engine lowering with the config snapshot and
|
|
188
|
-
// process profile frozen into the invocation plan. The loop never injects
|
|
189
|
-
// a RunnerSpec seam or observes later caller-config mutations.
|
|
190
|
-
const reflectCallArgs = {
|
|
191
|
-
ref: planned.ref,
|
|
192
|
-
// Carry the resolved item_ref so reflect uses the same durable key for
|
|
193
|
-
// events and state reads.
|
|
194
|
-
...(planned.itemRef ? { itemRef: planned.itemRef } : {}),
|
|
195
|
-
task: options.task,
|
|
196
|
-
// Active strategy supplies non-engine process tuning.
|
|
197
|
-
...(improveProfile ? { improveProfile } : {}),
|
|
198
|
-
config: resolvedPlan.config,
|
|
199
|
-
...(primaryStashDir ? { stashDir: primaryStashDir } : {}),
|
|
200
|
-
...(reflectTarget ? { target: reflectTarget } : {}),
|
|
201
|
-
...(reflectErrors.length > 0 ? { avoidPatterns: [...reflectErrors] } : {}),
|
|
202
|
-
eventSource: "improve",
|
|
203
|
-
// #639 — resolve the low-value filter from the ACTIVE improve profile
|
|
204
|
-
// (default off when unset), so the running strategy decides.
|
|
205
|
-
lowValueFilter: improveProfile.processes?.reflect?.lowValueFilter?.enabled === true,
|
|
206
|
-
...(reflectBudgetMs > 0 ? { timeoutMs: reflectBudgetMs } : {}),
|
|
207
|
-
signal: budgetSignal,
|
|
208
|
-
// R25: reflect's event emits reuse the run's long-lived state.db handle.
|
|
209
|
-
eventsCtx: env.eventsCtx,
|
|
210
|
-
// Attribution: carry the eligibility lane so reflect stamps it on
|
|
211
|
-
// the reflect_invoked event and the persisted proposal.
|
|
212
|
-
...(planned.eligibilitySource ? { eligibilitySource: planned.eligibilitySource } : {}),
|
|
213
|
-
};
|
|
214
|
-
// R9 (tier2-0917): the fingerprint/rejection-backoff guard `createProposal`
|
|
215
|
-
// runs AFTER reflect's ~39s generation + judge is computable from inputs
|
|
216
|
-
// available before dispatch. Check it here first — on a hit, skip the LLM
|
|
217
|
-
// call entirely and synthesize the same "cooldown" envelope reflect.ts's
|
|
218
|
-
// createProposal-skip branch returns, so everything below (mode
|
|
219
|
-
// classification, improve_reflect_outcome, plasticity) is unchanged. This
|
|
220
|
-
// pre-check is an optimisation only — createProposal's post-generation
|
|
221
|
-
// check stays authoritative (see checkProposalGuard's doc comment).
|
|
222
|
-
const guardStash = primaryStashDir ?? options.stashDir;
|
|
223
|
-
const guardSkip = guardStash
|
|
224
|
-
? checkProposalGuard({
|
|
225
|
-
stash: guardStash,
|
|
226
|
-
ref: planned.ref,
|
|
227
|
-
source: "reflect",
|
|
228
|
-
...(reflectTarget ? { target: reflectTarget } : {}),
|
|
229
|
-
...(reflectEngine ? { modelId: reflectEngine } : {}),
|
|
230
|
-
})
|
|
231
|
-
: undefined;
|
|
232
|
-
let reflectResult;
|
|
233
|
-
if (guardSkip) {
|
|
234
|
-
// Mirror reflect.ts's buildReflectEventEmitters().emitInvoked(): the
|
|
235
|
-
// signal-delta cursor (buildLatestProposalTsMap) reads `reflect_invoked`
|
|
236
|
-
// events regardless of outcome, so it must still advance for this ref
|
|
237
|
-
// even though reflectFn was never called.
|
|
238
|
-
appendEvent({
|
|
239
|
-
eventType: "reflect_invoked",
|
|
240
|
-
ref: planned.itemRef ?? durableImproveRef(planned.ref),
|
|
241
|
-
metadata: {
|
|
242
|
-
...(options.task ? { task: options.task } : {}),
|
|
243
|
-
...(reflectEngine ? { engine: reflectEngine } : {}),
|
|
244
|
-
...(planned.eligibilitySource ? { eligibilitySource: planned.eligibilitySource } : {}),
|
|
245
|
-
},
|
|
246
|
-
}, eventsCtx);
|
|
247
|
-
// Mirror reflect.ts's buildReflectEventEmitters().emitFailed(): every
|
|
248
|
-
// reflect_invoked must be paired with a reflect_completed so observers
|
|
249
|
-
// building closed-loop telemetry see balanced invoke/complete pairs.
|
|
250
|
-
// reflectFn is never called on this path, so reflect.ts's own
|
|
251
|
-
// emitFailed (fired from its post-generation cooldown branch) never
|
|
252
|
-
// runs either — this is the pre-generation guard's own pairing.
|
|
253
|
-
appendEvent({
|
|
254
|
-
eventType: "reflect_completed",
|
|
255
|
-
ref: planned.itemRef ?? durableImproveRef(planned.ref),
|
|
256
|
-
metadata: {
|
|
257
|
-
source: "reflect",
|
|
258
|
-
ok: false,
|
|
259
|
-
reason: "cooldown",
|
|
260
|
-
subreason: "pre_generation_guard",
|
|
261
|
-
proposalSkipReason: guardSkip.reason,
|
|
262
|
-
...(guardSkip.existingProposalId ? { existingProposalId: guardSkip.existingProposalId } : {}),
|
|
263
|
-
},
|
|
264
|
-
}, eventsCtx);
|
|
265
|
-
reflectResult = {
|
|
266
|
-
schemaVersion: 2,
|
|
267
|
-
ok: false,
|
|
268
|
-
reason: "cooldown",
|
|
269
|
-
error: `Proposal skipped (${guardSkip.reason}): ${guardSkip.message}`,
|
|
270
|
-
ref: planned.ref,
|
|
271
|
-
...(reflectEngine ? { engine: reflectEngine } : {}),
|
|
272
|
-
exitCode: null,
|
|
273
|
-
};
|
|
274
|
-
}
|
|
275
|
-
else {
|
|
276
|
-
reflectResult = await withLlmStage("reflect", () => reflectFn(reflectCallArgs), {
|
|
277
|
-
engine: reflectEngine,
|
|
278
|
-
process: "reflect",
|
|
279
|
-
});
|
|
280
|
-
}
|
|
281
|
-
const isCooldown = !reflectResult.ok && reflectResult.reason === "cooldown";
|
|
282
|
-
// Content-policy guard hits (reflect size-rail rejections) are NOT
|
|
283
|
-
// LLM faults — the agent responded fine, the downstream guard
|
|
284
|
-
// blocked the output. Route them to a distinct `reflect-guard-rejected`
|
|
285
|
-
// mode so health metrics can split deterministic guard hits out of
|
|
286
|
-
// true LLM failures. See
|
|
287
|
-
// `/tmp/akm-health-investigations/metrics-taxonomy-review.md` §1a.
|
|
288
|
-
const isGuardReject = !reflectResult.ok && reflectResult.reason === "content_policy_reject";
|
|
289
|
-
// Type-guard rejection (reflect refused a script/env/task ref) is
|
|
290
|
-
// also NOT an LLM failure — the LLM is never invoked. Route to the
|
|
291
|
-
// existing `reflect-skipped` bucket so it does not inflate the
|
|
292
|
-
// failure-rate numerator. ~9% of `reflect-failed` events in the
|
|
293
|
-
// user's stack were this case; see review §1a row "Reflect refused
|
|
294
|
-
// asset type".
|
|
295
|
-
const isTypeRefused = !reflectResult.ok && reflectResult.reason === "unsupported_type";
|
|
296
|
-
// Noise-gate suppression (#580): the candidate edit was an empty
|
|
297
|
-
// diff or a cosmetic-only reformat of the current asset. Like
|
|
298
|
-
// `unsupported_type`, this is a deterministic skip — not an LLM
|
|
299
|
-
// fault — so it routes to the `reflect-skipped` bucket and stays
|
|
300
|
-
// out of recentErrors/avoidPatterns.
|
|
301
|
-
const isNoChange = !reflectResult.ok && reflectResult.reason === "no_change";
|
|
302
|
-
// Quality-gate rejection (R3): the judge rejected an otherwise
|
|
303
|
-
// well-parsed proposal. Stays in the `reflect-failed` bucket for
|
|
304
|
-
// metrics continuity, but — like the deterministic skips above —
|
|
305
|
-
// must not be injected into recentErrors/avoidPatterns: the judge's
|
|
306
|
-
// rejection text is not a reusable "avoid this pattern" lesson, and
|
|
307
|
-
// feeding it back in poisoned later prompts in the same run.
|
|
308
|
-
const isQualityRejected = !reflectResult.ok && reflectResult.reason === "quality_rejected";
|
|
309
|
-
tally.actions.push({
|
|
310
|
-
ref: planned.ref,
|
|
311
|
-
mode: reflectResult.ok
|
|
312
|
-
? "reflect"
|
|
313
|
-
: isCooldown
|
|
314
|
-
? "reflect-cooldown"
|
|
315
|
-
: isGuardReject
|
|
316
|
-
? "reflect-guard-rejected"
|
|
317
|
-
: isTypeRefused || isNoChange
|
|
318
|
-
? "reflect-skipped"
|
|
319
|
-
: "reflect-failed",
|
|
320
|
-
result: reflectResult,
|
|
321
|
-
});
|
|
322
|
-
// Cooldown skips, guard rejects, type-refused skips, noise-gate
|
|
323
|
-
// skips, and quality-gate rejections are not failures — do not
|
|
324
|
-
// pollute recentErrors with them (those get injected as
|
|
325
|
-
// `avoidPatterns` into the next reflect prompt). Guard rejects ARE
|
|
326
|
-
// worth showing the LLM as a learn-signal so the next iteration sees
|
|
327
|
-
// "your last expansion was too large"; type-refused, no-change, and
|
|
328
|
-
// quality-rejected are deterministic/judge-side and add no learning
|
|
329
|
-
// signal.
|
|
330
|
-
if (!reflectResult.ok && !isCooldown && !isTypeRefused && !isNoChange && !isQualityRejected) {
|
|
331
|
-
const errMsg = reflectResult.error ?? reflectResult.reason ?? "unknown reflect error";
|
|
332
|
-
tally.recentErrorPushes.push({ originator: "reflect", message: errMsg });
|
|
333
|
-
}
|
|
334
|
-
// improve_reflect_outcome — per-asset metric for tuning the reflect path.
|
|
335
|
-
appendEvent({
|
|
336
|
-
eventType: "improve_reflect_outcome",
|
|
337
|
-
ref: planned.ref,
|
|
338
|
-
metadata: {
|
|
339
|
-
ok: reflectResult.ok,
|
|
340
|
-
durationMs: reflectResult.ok ? reflectResult.durationMs : undefined,
|
|
341
|
-
engine: reflectResult.engine,
|
|
342
|
-
reason: reflectResult.ok ? undefined : reflectResult.reason,
|
|
343
|
-
},
|
|
344
|
-
}, eventsCtx);
|
|
345
|
-
// Plasticity counter (plan §WS-1 step 8): record no-ops so the
|
|
346
|
-
// WS-1 selection comparator (effectiveScore, ~line 3073) can dampen
|
|
347
|
-
// repeatedly-silent assets during consolidation-selection.
|
|
348
|
-
// A no_change reflect means the LLM was invoked but found nothing to
|
|
349
|
-
// improve — the asset is stable. Track it. A successful reflect means
|
|
350
|
-
// the asset changed; reset the counter so the dampener lifts.
|
|
351
|
-
// Use the same item_ref-or-conceptId salience key as preparation/distill.
|
|
352
|
-
const plasticityKey = planned.itemRef ?? durableImproveRef(planned.ref);
|
|
353
|
-
if (isNoChange && eventsCtx?.db) {
|
|
354
|
-
try {
|
|
355
|
-
recordNoOp(eventsCtx.db, plasticityKey);
|
|
356
|
-
}
|
|
357
|
-
catch {
|
|
358
|
-
// best-effort: plasticity counter failure never blocks the run
|
|
359
|
-
}
|
|
360
|
-
}
|
|
361
|
-
else if (reflectResult.ok && eventsCtx?.db) {
|
|
362
|
-
try {
|
|
363
|
-
resetConsecutiveNoOps(eventsCtx.db, plasticityKey);
|
|
364
|
-
}
|
|
365
|
-
catch {
|
|
366
|
-
// best-effort
|
|
367
|
-
}
|
|
368
|
-
}
|
|
369
|
-
} // end else (reflect type/profile check)
|
|
370
|
-
}
|
|
371
|
-
else if (!isDistillOnly && planned.ref.endsWith(".derived")) {
|
|
372
|
-
// B6: .derived refs skip reflect; record synthetic skip action.
|
|
373
|
-
tally.actions.push({
|
|
374
|
-
ref: planned.ref,
|
|
375
|
-
mode: "distill-skipped",
|
|
376
|
-
result: { ok: true, reason: "derived-memory-reflect-skipped" },
|
|
377
|
-
});
|
|
378
|
-
appendEvent({
|
|
379
|
-
eventType: "improve_skipped",
|
|
380
|
-
ref: planned.ref,
|
|
381
|
-
metadata: { reason: "derived_memory_reflect_skipped" },
|
|
382
|
-
}, eventsCtx);
|
|
383
|
-
}
|
|
384
|
-
}
|
|
385
|
-
/**
|
|
386
|
-
* Distill half of one loop iteration: the profile / requirePlannedRefs /
|
|
387
|
-
* candidate-type / weak-signal / cooldown gates, then the pending-proposal and
|
|
388
|
-
* reject-grace dedup checks, then {@link invokeDistillAndRecord}. Each gate
|
|
389
|
-
* that was a `continue` in the old inline loop body is an early `return` here.
|
|
390
|
-
*/
|
|
391
|
-
async function runLoopDistillPass(planned, parsedPlannedRef, isDistillOnly, env, tally) {
|
|
392
|
-
const { options, primaryStashDir, eventsCtx, improveProfile, resolvedPlan } = env;
|
|
393
|
-
const hasRecentFeedbackSignal = env.signalBearingSet.has(planned.ref);
|
|
394
|
-
const explicitRefScope = env.scope.mode === "ref";
|
|
395
|
-
// Profile gate: apply the full type-filter / raw-wiki / disabled rules to
|
|
396
|
-
// distill so callers who configure `profile.processes.distill.allowedTypes`
|
|
397
|
-
// or land on raw-wiki refs get a recorded skip action instead of silently
|
|
398
|
-
// proceeding.
|
|
399
|
-
const distillSkip = shouldSkipRef(planned.ref, "distill", improveProfile);
|
|
400
|
-
if (distillSkip.skip) {
|
|
401
|
-
tally.actions.push({
|
|
402
|
-
ref: planned.ref,
|
|
403
|
-
mode: "distill-skipped",
|
|
404
|
-
result: { ok: true, reason: distillSkip.reason },
|
|
405
|
-
});
|
|
126
|
+
async function runLoopReflectPass(planned, env, tally) {
|
|
127
|
+
const { options, primaryStashDir, improveProfile, resolvedPlan } = env;
|
|
128
|
+
// Derived memories are machine-generated: never reflected.
|
|
129
|
+
if (planned.ref.endsWith(".derived")) {
|
|
130
|
+
recordSkip(tally, planned.ref, "derived-memory-reflect-skipped", { env, reason: "derived_memory_reflect_skipped" });
|
|
406
131
|
return;
|
|
407
132
|
}
|
|
408
|
-
|
|
409
|
-
|
|
410
|
-
|
|
411
|
-
tally.actions.push({
|
|
412
|
-
ref: planned.ref,
|
|
413
|
-
mode: "distill-skipped",
|
|
414
|
-
result: { ok: true, reason: "require_planned_refs" },
|
|
415
|
-
});
|
|
133
|
+
const reflectSkip = shouldSkipRef(planned.ref, "reflect", improveProfile);
|
|
134
|
+
if (reflectSkip.skip) {
|
|
135
|
+
tally.actions.push({ ref: planned.ref, mode: "reflect-skipped", result: { ok: true, reason: reflectSkip.reason } });
|
|
416
136
|
return;
|
|
417
137
|
}
|
|
418
|
-
//
|
|
419
|
-
|
|
420
|
-
|
|
421
|
-
|
|
422
|
-
const
|
|
423
|
-
|
|
424
|
-
|
|
425
|
-
|
|
426
|
-
|
|
427
|
-
|
|
428
|
-
|
|
429
|
-
(
|
|
430
|
-
|
|
431
|
-
|
|
432
|
-
|
|
433
|
-
|
|
434
|
-
|
|
435
|
-
|
|
436
|
-
|
|
437
|
-
|
|
438
|
-
|
|
439
|
-
|
|
440
|
-
|
|
441
|
-
|
|
442
|
-
|
|
443
|
-
|
|
444
|
-
|
|
445
|
-
|
|
446
|
-
|
|
447
|
-
|
|
448
|
-
|
|
449
|
-
|
|
450
|
-
|
|
451
|
-
|
|
452
|
-
|
|
453
|
-
|
|
454
|
-
|
|
455
|
-
|
|
456
|
-
|
|
457
|
-
|
|
458
|
-
|
|
459
|
-
|
|
460
|
-
|
|
461
|
-
|
|
462
|
-
|
|
463
|
-
if (recentlyRejectedLesson) {
|
|
464
|
-
const rejectedEntry = env.rejectedProposalsByRef.get(lessonRef) ?? env.rejectedProposalsByRef.get(knowledgeRef);
|
|
465
|
-
const rejectedAgeMs = rejectedEntry ? Date.now() - new Date(rejectedEntry.ts).getTime() : 0;
|
|
466
|
-
if (rejectedAgeMs < DISTILL_REJECT_COOLDOWN_MS) {
|
|
467
|
-
tally.actions.push({
|
|
468
|
-
ref: planned.ref,
|
|
469
|
-
mode: "distill-skipped",
|
|
470
|
-
result: { ok: true, reason: "distill reject grace window" },
|
|
471
|
-
});
|
|
472
|
-
appendEvent({
|
|
473
|
-
eventType: "improve_skipped",
|
|
474
|
-
ref: planned.ref,
|
|
475
|
-
metadata: {
|
|
476
|
-
reason: "distill_reject_grace_window",
|
|
477
|
-
},
|
|
478
|
-
}, eventsCtx);
|
|
479
|
-
return;
|
|
480
|
-
}
|
|
481
|
-
}
|
|
482
|
-
// R9 extension (r2-6, tier2-0917; PRECHECK, tier3-0917): the
|
|
483
|
-
// fingerprint/rejection-backoff guard `createProposal` runs AFTER
|
|
484
|
-
// distill's ~generation + judge is computable from inputs available
|
|
485
|
-
// before dispatch — mirror the reflect pre-check above so a guard hit
|
|
486
|
-
// skips the LLM call entirely. Distill's real `createProposal` call
|
|
487
|
-
// always targets the derived lesson/knowledge ref (`effectiveLessonRef`
|
|
488
|
-
// in distill.ts), never the input ref.
|
|
489
|
-
// Which ref that is: for every non-memory distill-candidate type,
|
|
490
|
-
// `targetKind` defaults to "lesson" (distill.ts ~L882, `invokeDistill
|
|
491
|
-
// AndRecord` above only ever sets `proposalKind: "auto"` for memory
|
|
492
|
-
// refs) and is never overridden to "knowledge", so lessonRef is the
|
|
493
|
-
// ONLY real target. For memory refs (`proposalKind: "auto"`), the
|
|
494
|
-
// target is decided at dispatch by `planMemoryKnowledgePromotion`
|
|
495
|
-
// (knowledgeRef via promotion, lessonRef as fallback) — that decision
|
|
496
|
-
// IS cheap and LLM-free (a deterministic score over the asset content
|
|
497
|
-
// + its feedback history, plus one lookup for an existing knowledge
|
|
498
|
-
// file), so it is pre-checked exactly via `wouldPromoteMemoryToKnowledge`,
|
|
499
|
-
// a thin wrapper that delegates to `planMemoryKnowledgePromotion`
|
|
500
|
-
// itself so this can never drift from distill's real decision. A
|
|
501
|
-
// guard hit on the ref distill would NOT have targeted must never
|
|
502
|
-
// suppress a legitimate dispatch.
|
|
503
|
-
// §23.6 fingerprint model-id term: distill resolves models, not
|
|
504
|
-
// engines (unlike reflect), so this must match `distillRunner?.
|
|
505
|
-
// connection.model` in distill.ts, not the engine name.
|
|
506
|
-
const distillModelId = resolvedPlan.processes.distill.runner?.connection.model;
|
|
507
|
-
let realTargetRef = lessonRef;
|
|
508
|
-
if (parsedPlannedRef.type === "memory") {
|
|
509
|
-
// distill.ts's real dispatch (akmDistill) always derives
|
|
510
|
-
// durableInputRef from options.ref alone (durableImproveRef(inputRef),
|
|
511
|
-
// never itemRef) and reads/scores content via that ref
|
|
512
|
-
// (loadAndScoreInputSalience's `lookup(durableInputRef)`); mirror
|
|
513
|
-
// that here so the pre-check can never read/score a different file
|
|
514
|
-
// than the real dispatch would. itemRef is preferred only for the
|
|
515
|
-
// feedback-events query, matching readDistillFeedback's
|
|
516
|
-
// `ref: options.itemRef ?? durableInputRef`.
|
|
517
|
-
const durableInputRef = durableImproveRef(planned.ref);
|
|
518
|
-
const feedbackRef = planned.itemRef ?? durableInputRef;
|
|
519
|
-
const lookup = (ref) => defaultLookup(ref, dedupeStashDir);
|
|
520
|
-
const filePath = await lookup(durableInputRef);
|
|
521
|
-
const assetContent = filePath && fs.existsSync(filePath) ? fs.readFileSync(filePath, "utf8") : null;
|
|
522
|
-
// PRECHECK (tier3-0917-r3, r3-4): reuse the loop's long-lived
|
|
523
|
-
// eventsCtx.db handle when one is open, instead of opening a fresh
|
|
524
|
-
// read-only state.db connection per memory ref (R25). Degrades to
|
|
525
|
-
// the previous readOnly-open when no live handle is present (e.g.
|
|
526
|
-
// this function invoked without a run-scoped eventsCtx), via the
|
|
527
|
-
// same readOnlyEventsContext helper reflect.ts's read call sites use.
|
|
528
|
-
const { events: feedbackEvents } = readEvents({ ref: feedbackRef, type: "feedback" }, readOnlyEventsContext(eventsCtx));
|
|
529
|
-
const promotesToKnowledge = await wouldPromoteMemoryToKnowledge({
|
|
530
|
-
inputRef: planned.ref,
|
|
531
|
-
durableInputRef,
|
|
532
|
-
assetContent,
|
|
533
|
-
feedbackEvents,
|
|
534
|
-
config: options.config ?? loadConfig(),
|
|
535
|
-
stash: dedupeStashDir,
|
|
536
|
-
lookup,
|
|
537
|
-
});
|
|
538
|
-
if (promotesToKnowledge)
|
|
539
|
-
realTargetRef = knowledgeRef;
|
|
540
|
-
}
|
|
541
|
-
const guardSkip = checkProposalGuard({
|
|
542
|
-
stash: dedupeStashDir,
|
|
543
|
-
ref: realTargetRef,
|
|
544
|
-
source: "distill",
|
|
545
|
-
...(distillModelId ? { modelId: distillModelId } : {}),
|
|
138
|
+
// Only reflect's own recent errors reach its prompt.
|
|
139
|
+
const reflectErrors = env.recentErrors.reflect ?? [];
|
|
140
|
+
if (reflectErrors.length > 0)
|
|
141
|
+
tally.reflectsWithErrorContext++;
|
|
142
|
+
const budgetMs = env.remainingBudgetMs();
|
|
143
|
+
const reflectArgs = {
|
|
144
|
+
ref: planned.ref,
|
|
145
|
+
...(planned.itemRef ? { itemRef: planned.itemRef } : {}),
|
|
146
|
+
task: options.task,
|
|
147
|
+
...(improveProfile ? { improveProfile } : {}),
|
|
148
|
+
config: resolvedPlan.config,
|
|
149
|
+
...(primaryStashDir ? { stashDir: primaryStashDir } : {}),
|
|
150
|
+
...(options.sourceName && primaryStashDir ? { target: { source: options.sourceName, root: primaryStashDir } } : {}),
|
|
151
|
+
...(reflectErrors.length > 0 ? { avoidPatterns: [...reflectErrors] } : {}),
|
|
152
|
+
eventSource: "improve",
|
|
153
|
+
lowValueFilter: improveProfile.processes?.reflect?.lowValueFilter?.enabled === true,
|
|
154
|
+
...(budgetMs > 0 ? { timeoutMs: budgetMs } : {}),
|
|
155
|
+
signal: env.budgetSignal,
|
|
156
|
+
eventsCtx: env.eventsCtx,
|
|
157
|
+
...(planned.eligibilitySource ? { eligibilitySource: planned.eligibilitySource } : {}),
|
|
158
|
+
};
|
|
159
|
+
const result = await attributeStage(resolvedPlan, "reflect", () => env.reflectFn(reflectArgs));
|
|
160
|
+
const reason = result.ok ? undefined : result.reason;
|
|
161
|
+
// A refused type or an unchanged asset is a deterministic skip, not an LLM
|
|
162
|
+
// fault; a guard rejection (size rail) gets its own bucket for health.
|
|
163
|
+
const skipped = reason === "unsupported_type" || reason === "no_change";
|
|
164
|
+
tally.actions.push({
|
|
165
|
+
ref: planned.ref,
|
|
166
|
+
mode: result.ok
|
|
167
|
+
? "reflect"
|
|
168
|
+
: reason === "content_policy_reject"
|
|
169
|
+
? "reflect-guard-rejected"
|
|
170
|
+
: skipped
|
|
171
|
+
? "reflect-skipped"
|
|
172
|
+
: "reflect-failed",
|
|
173
|
+
result,
|
|
174
|
+
});
|
|
175
|
+
// A quality rejection recorded itself, and the judge's text is no lesson for
|
|
176
|
+
// the next prompt; skips revisit on the `unchanged` cadence.
|
|
177
|
+
if (!result.ok && reason !== "quality_rejected") {
|
|
178
|
+
recordLoopAttempt(planned, env, "reflect", skipped ? "unchanged" : "failed", reason);
|
|
179
|
+
if (!skipped) {
|
|
180
|
+
tally.recentErrorPushes.push({
|
|
181
|
+
originator: "reflect",
|
|
182
|
+
message: result.error ?? reason ?? "unknown reflect error",
|
|
546
183
|
});
|
|
547
|
-
if (guardSkip) {
|
|
548
|
-
tally.actions.push({
|
|
549
|
-
ref: planned.ref,
|
|
550
|
-
mode: "distill-skipped",
|
|
551
|
-
result: { ok: true, reason: guardSkip.reason },
|
|
552
|
-
});
|
|
553
|
-
// Mirror distill.ts's own proposal-skip branch (the post-generation
|
|
554
|
-
// guard `createProposal` hits): emit `distill_invoked` with a
|
|
555
|
-
// `skipped` outcome so the signal-delta cursor
|
|
556
|
-
// (buildLatestProposalTsMap, eligibility.ts) advances for this ref
|
|
557
|
-
// even though distillFn was never called.
|
|
558
|
-
appendEvent({
|
|
559
|
-
eventType: "distill_invoked",
|
|
560
|
-
// Use item_ref when resolved, otherwise the input conceptId —
|
|
561
|
-
// matches distill.ts's own distill_invoked key.
|
|
562
|
-
ref: planned.itemRef ?? durableImproveRef(planned.ref),
|
|
563
|
-
metadata: {
|
|
564
|
-
outcome: "skipped",
|
|
565
|
-
proposalRef: realTargetRef,
|
|
566
|
-
message: guardSkip.message,
|
|
567
|
-
skipReason: guardSkip.reason,
|
|
568
|
-
...(planned.eligibilitySource ? { eligibilitySource: planned.eligibilitySource } : {}),
|
|
569
|
-
},
|
|
570
|
-
}, eventsCtx);
|
|
571
|
-
return;
|
|
572
|
-
}
|
|
573
184
|
}
|
|
574
|
-
await invokeDistillAndRecord(planned, parsedPlannedRef, env, tally);
|
|
575
185
|
}
|
|
576
|
-
|
|
577
|
-
|
|
578
|
-
|
|
579
|
-
|
|
580
|
-
|
|
186
|
+
appendEvent({
|
|
187
|
+
eventType: "improve_reflect_outcome",
|
|
188
|
+
ref: planned.ref,
|
|
189
|
+
metadata: {
|
|
190
|
+
ok: result.ok,
|
|
191
|
+
durationMs: result.ok ? result.durationMs : undefined,
|
|
192
|
+
engine: result.engine,
|
|
193
|
+
reason,
|
|
194
|
+
},
|
|
195
|
+
}, env.eventsCtx);
|
|
196
|
+
recordPlasticity(env, planned, reason === "no_change" ? "noop" : result.ok ? "changed" : undefined);
|
|
197
|
+
}
|
|
198
|
+
async function runLoopDistillPass(planned, refType, isDistillOnly, env, tally) {
|
|
199
|
+
const { options, primaryStashDir, improveProfile, resolvedPlan } = env;
|
|
200
|
+
const distillSkip = shouldSkipRef(planned.ref, "distill", improveProfile);
|
|
201
|
+
if (distillSkip.skip)
|
|
202
|
+
return recordSkip(tally, planned.ref, distillSkip.reason);
|
|
203
|
+
if (env.skipDistillDueToRequirePlannedRefs && isDistillOnly) {
|
|
204
|
+
return recordSkip(tally, planned.ref, "require_planned_refs");
|
|
205
|
+
}
|
|
206
|
+
const explicitRefScope = env.scope.mode === "ref";
|
|
207
|
+
const weakMemorySignal = !isDistillOnly && refType === "memory" && !env.signalBearingSet.has(planned.ref) && !explicitRefScope;
|
|
208
|
+
if (weakMemorySignal) {
|
|
209
|
+
return recordSkip(tally, planned.ref, "memory requires recent feedback signal", {
|
|
210
|
+
env,
|
|
211
|
+
reason: "memory_distill_requires_feedback",
|
|
581
212
|
});
|
|
582
|
-
appendEvent({
|
|
583
|
-
eventType: "improve_skipped",
|
|
584
|
-
ref: planned.ref,
|
|
585
|
-
metadata: { reason: "memory_distill_requires_feedback" },
|
|
586
|
-
}, eventsCtx);
|
|
587
213
|
}
|
|
588
|
-
|
|
589
|
-
|
|
590
|
-
|
|
591
|
-
|
|
592
|
-
|
|
593
|
-
|
|
594
|
-
const { options, primaryStashDir, distillFn, eventsCtx, improveProfile, resolvedPlan, budgetSignal } = env;
|
|
595
|
-
const distillResult = await withLlmStage("distill", () => distillFn({
|
|
214
|
+
// The ledger holds cooled refs; an explicit `--scope` ref overrides it.
|
|
215
|
+
if (!isDistillCandidateRef(planned.ref, options.stashDir))
|
|
216
|
+
return;
|
|
217
|
+
if (env.distillCooledRefs.has(planned.ref) && !explicitRefScope)
|
|
218
|
+
return;
|
|
219
|
+
const result = await attributeStage(resolvedPlan, "distill", () => env.distillFn({
|
|
596
220
|
ref: planned.ref,
|
|
597
|
-
// Carry the resolved item_ref so distill matches preparation's state key.
|
|
598
221
|
...(planned.itemRef ? { itemRef: planned.itemRef } : {}),
|
|
599
|
-
...(
|
|
222
|
+
...(refType === "memory" ? { proposalKind: "auto" } : {}),
|
|
600
223
|
...(primaryStashDir ? { stashDir: primaryStashDir } : {}),
|
|
601
|
-
// Active profile so distill's per-process reads honor `--profile`.
|
|
602
224
|
...(improveProfile ? { improveProfile } : {}),
|
|
603
225
|
config: options.config,
|
|
604
226
|
llmRunner: resolvedPlan.processes.distill.runner,
|
|
605
|
-
signal: budgetSignal,
|
|
606
|
-
|
|
607
|
-
eventsCtx,
|
|
608
|
-
// Attribution: carry the eligibility lane so distill stamps it on the
|
|
609
|
-
// distill_invoked event and the persisted proposal.
|
|
227
|
+
signal: env.budgetSignal,
|
|
228
|
+
eventsCtx: env.eventsCtx,
|
|
610
229
|
...(planned.eligibilitySource ? { eligibilitySource: planned.eligibilitySource } : {}),
|
|
611
|
-
})
|
|
612
|
-
tally.actions.push({ ref: planned.ref, mode: "distill", result
|
|
613
|
-
|
|
614
|
-
|
|
615
|
-
|
|
616
|
-
|
|
230
|
+
}));
|
|
231
|
+
tally.actions.push({ ref: planned.ref, mode: "distill", result });
|
|
232
|
+
// `queued` and the quality outcomes recorded themselves; a transport failure
|
|
233
|
+
// or a disabled process is not an attempt, so the ref stays eligible.
|
|
234
|
+
if (result.outcome === "skipped") {
|
|
235
|
+
recordLoopAttempt(planned, env, "distill", "unchanged", result.skipReason ?? result.message);
|
|
617
236
|
}
|
|
618
|
-
|
|
619
|
-
|
|
620
|
-
// quality gate — the asset is not yielding useful distill output.
|
|
621
|
-
// queued: a proposal was produced; reset the no-op counter.
|
|
622
|
-
if (eventsCtx?.db) {
|
|
623
|
-
// Use the same item_ref-or-conceptId key as the distill/preparation writers.
|
|
624
|
-
const plasticityKey = planned.itemRef ?? durableImproveRef(planned.ref);
|
|
625
|
-
try {
|
|
626
|
-
if (distillResult.outcome === "quality_rejected" || distillResult.outcome === "skipped") {
|
|
627
|
-
recordNoOp(eventsCtx.db, plasticityKey);
|
|
628
|
-
}
|
|
629
|
-
else if (distillResult.outcome === "queued") {
|
|
630
|
-
resetConsecutiveNoOps(eventsCtx.db, plasticityKey);
|
|
631
|
-
}
|
|
632
|
-
}
|
|
633
|
-
catch {
|
|
634
|
-
// best-effort: plasticity counter failure never blocks the run
|
|
635
|
-
}
|
|
636
|
-
}
|
|
637
|
-
}
|
|
638
|
-
/**
|
|
639
|
-
* Wall-clock budget exhausted mid-loop (O-1 / #364): emit the improve_skipped
|
|
640
|
-
* events for the current and remaining refs (B11) and return the terminal
|
|
641
|
-
* error action for the orchestrator to record before breaking out of the loop.
|
|
642
|
-
*/
|
|
643
|
-
function recordBudgetExhausted(args) {
|
|
644
|
-
const { planned, loopRefs, completedCount, startMs, eventsCtx } = args;
|
|
645
|
-
const remaining = loopRefs.length - completedCount;
|
|
646
|
-
info(`[improve] budget exhausted after ${Math.round((Date.now() - startMs) / 60000)}min — ${remaining} assets skipped`);
|
|
647
|
-
appendEvent({
|
|
648
|
-
eventType: "improve_skipped",
|
|
649
|
-
ref: planned.ref,
|
|
650
|
-
metadata: {
|
|
651
|
-
reason: "budget_exhausted",
|
|
652
|
-
remaining,
|
|
653
|
-
},
|
|
654
|
-
}, eventsCtx);
|
|
655
|
-
// B11: Emit improve_skipped for all remaining assets that will not be processed.
|
|
656
|
-
for (const remainingRef of loopRefs.slice(completedCount + 1)) {
|
|
657
|
-
appendEvent({
|
|
658
|
-
eventType: "improve_skipped",
|
|
659
|
-
ref: remainingRef.ref,
|
|
660
|
-
metadata: { reason: "budget_exhausted_batch", remaining: loopRefs.length - completedCount - 1 },
|
|
661
|
-
}, eventsCtx);
|
|
237
|
+
if (refType === "memory" && !(result.outcome === "queued" && result.proposalKind === "knowledge")) {
|
|
238
|
+
tally.memoryRefsForInference.push(planned.ref);
|
|
662
239
|
}
|
|
663
|
-
|
|
664
|
-
|
|
665
|
-
|
|
666
|
-
|
|
667
|
-
|
|
240
|
+
recordPlasticity(env, planned, result.outcome === "quality_rejected" || result.outcome === "skipped"
|
|
241
|
+
? "noop"
|
|
242
|
+
: result.outcome === "queued"
|
|
243
|
+
? "changed"
|
|
244
|
+
: undefined);
|
|
668
245
|
}
|
|
669
246
|
export async function runImproveLoopStage(args) {
|
|
670
|
-
const {
|
|
671
|
-
const eventsCtx = ctx.eventsCtx;
|
|
247
|
+
const { loopRefs, actions, recentErrors, startMs, budgetMs, eventsCtx } = args;
|
|
672
248
|
const env = prepareImproveLoopEnv(args);
|
|
673
|
-
let completedCount = 0;
|
|
674
249
|
let reflectsWithErrorContext = 0;
|
|
675
250
|
const memoryRefsForInference = new Set();
|
|
676
|
-
for (const planned of loopRefs) {
|
|
251
|
+
for (const [index, planned] of loopRefs.entries()) {
|
|
677
252
|
if (Date.now() - startMs >= budgetMs) {
|
|
678
|
-
|
|
253
|
+
const remaining = loopRefs.length - index;
|
|
254
|
+
info(`[improve] budget exhausted after ${Math.round((Date.now() - startMs) / 60000)}min — ${remaining} assets skipped`);
|
|
255
|
+
appendEvent({ eventType: "improve_skipped", ref: planned.ref, metadata: { reason: "budget_exhausted", remaining } }, eventsCtx);
|
|
256
|
+
for (const rest of loopRefs.slice(index + 1)) {
|
|
257
|
+
appendEvent({
|
|
258
|
+
eventType: "improve_skipped",
|
|
259
|
+
ref: rest.ref,
|
|
260
|
+
metadata: { reason: "budget_exhausted_batch", remaining: remaining - 1 },
|
|
261
|
+
}, eventsCtx);
|
|
262
|
+
}
|
|
263
|
+
actions.push({
|
|
264
|
+
ref: planned.ref,
|
|
265
|
+
mode: "error",
|
|
266
|
+
result: { ok: false, error: "timeout: improve wall-clock budget exhausted" },
|
|
267
|
+
});
|
|
679
268
|
break;
|
|
680
269
|
}
|
|
681
270
|
const tally = await processImproveLoopRef(planned, env);
|
|
682
|
-
// Fold the per-ref tally into run-level state — the passes never touch it.
|
|
683
271
|
actions.push(...tally.actions);
|
|
684
272
|
for (const push of tally.recentErrorPushes)
|
|
685
273
|
pushRecentError(recentErrors, push.originator, push.message);
|
|
686
274
|
reflectsWithErrorContext += tally.reflectsWithErrorContext;
|
|
687
275
|
for (const ref of tally.memoryRefsForInference)
|
|
688
276
|
memoryRefsForInference.add(ref);
|
|
689
|
-
|
|
690
|
-
info(`[improve] ${completedCount}/${loopRefs.length} ${planned.ref}`);
|
|
277
|
+
info(`[improve] ${index + 1}/${loopRefs.length} ${planned.ref}`);
|
|
691
278
|
}
|
|
692
279
|
return { reflectsWithErrorContext, memoryRefsForInference };
|
|
693
280
|
}
|
|
694
281
|
export async function runImprovePostLoopStage(args) {
|
|
695
|
-
const { scope,
|
|
696
|
-
const allWarnings = [...cleanupWarnings, ...(appliedCleanup?.warnings ?? [])];
|
|
282
|
+
const { scope, primaryStashDir, actionableRefs } = args;
|
|
283
|
+
const allWarnings = [...args.cleanupWarnings, ...(args.appliedCleanup?.warnings ?? [])];
|
|
697
284
|
info("[improve] post-loop maintenance starting");
|
|
698
|
-
const
|
|
699
|
-
options,
|
|
700
|
-
primaryStashDir,
|
|
701
|
-
actionableRefs,
|
|
702
|
-
memoryRefsForInference,
|
|
703
|
-
allWarnings,
|
|
704
|
-
// O-1 (#364): forward the budget signal to memory inference + graph extraction.
|
|
705
|
-
budgetSignal,
|
|
706
|
-
eventsCtx,
|
|
707
|
-
improveProfile,
|
|
708
|
-
resolvedPlan,
|
|
709
|
-
});
|
|
285
|
+
const maintenance = await runImproveMaintenancePasses({ ...args, allWarnings });
|
|
710
286
|
let deadUrls;
|
|
711
287
|
let deadUrlCoverage;
|
|
712
288
|
if (scope.mode === "all" && primaryStashDir && actionableRefs.length > 0) {
|
|
713
289
|
try {
|
|
714
|
-
// Every actionable knowledge ref is scanned
|
|
715
|
-
//
|
|
716
|
-
// `deadUrlCoverage.total` counted only those, so a real bundle reported
|
|
717
|
-
// checked === total while most refs were never looked at (#892). URL
|
|
718
|
-
// extraction (a regex over already-loaded text) is cheap; it is the
|
|
719
|
-
// network requests that are expensive, and those are bounded by
|
|
720
|
-
// `checkDeadUrls`'s concurrency limit, not by trimming what gets scanned.
|
|
290
|
+
// Every actionable knowledge ref is scanned; checkDeadUrls bounds the
|
|
291
|
+
// network concurrency (#892).
|
|
721
292
|
const knowledgeEntries = actionableRefs
|
|
722
293
|
.filter((r) => {
|
|
723
294
|
try {
|
|
@@ -728,9 +299,6 @@ export async function runImprovePostLoopStage(args) {
|
|
|
728
299
|
}
|
|
729
300
|
})
|
|
730
301
|
.map((r) => {
|
|
731
|
-
// The URL scan needs the document body; filePath is pre-resolved on
|
|
732
|
-
// eligible refs at planning time (#591). Best-effort — an unreadable
|
|
733
|
-
// or unresolved file contributes no URLs, same as before.
|
|
734
302
|
let body = "";
|
|
735
303
|
if (r.filePath) {
|
|
736
304
|
try {
|
|
@@ -754,114 +322,63 @@ export async function runImprovePostLoopStage(args) {
|
|
|
754
322
|
// best-effort
|
|
755
323
|
}
|
|
756
324
|
}
|
|
757
|
-
// ── R5: collapse/churn detector ────────────────────────────────────────────
|
|
758
|
-
// One snapshot per QUALIFYING cycle: consolidate processed work. Deterministic,
|
|
759
|
-
// observe-only, fail-open (the orchestrator catches everything) — and inert
|
|
760
|
-
// on the ~9-in-10 default-profile runs that touch no merges.
|
|
761
|
-
let cycleMetrics;
|
|
762
|
-
if (!options.dryRun && consolidationRan) {
|
|
763
|
-
cycleMetrics = runCollapseDetector({
|
|
764
|
-
runId: options.runId ?? "improve-adhoc",
|
|
765
|
-
...(improveProfile ? { improveProfile } : {}),
|
|
766
|
-
pass: "consolidate",
|
|
767
|
-
mergeFloorViolations: args.consolidationMergeFloorViolations ?? 0,
|
|
768
|
-
config: options.config ?? loadConfig(),
|
|
769
|
-
...(eventsCtx ? { eventsCtx } : {}),
|
|
770
|
-
});
|
|
771
|
-
}
|
|
772
325
|
return {
|
|
773
326
|
allWarnings,
|
|
774
327
|
deadUrls,
|
|
775
328
|
...(deadUrlCoverage ? { deadUrlCoverage } : {}),
|
|
776
|
-
...(
|
|
777
|
-
...(
|
|
778
|
-
...(
|
|
779
|
-
|
|
780
|
-
|
|
781
|
-
|
|
782
|
-
|
|
783
|
-
graphExtractionDurationMs: maintenanceResult.graphExtractionDurationMs,
|
|
784
|
-
orphansPurged: maintenanceResult.orphansPurged,
|
|
785
|
-
proposalsExpired: maintenanceResult.proposalsExpired,
|
|
329
|
+
...(maintenance.memoryInference ? { memoryInference: maintenance.memoryInference } : {}),
|
|
330
|
+
...(maintenance.graphExtraction ? { graphExtraction: maintenance.graphExtraction } : {}),
|
|
331
|
+
...(maintenance.actions && maintenance.actions.length > 0 ? { maintenanceActions: maintenance.actions } : {}),
|
|
332
|
+
memoryInferenceDurationMs: maintenance.memoryInferenceDurationMs,
|
|
333
|
+
graphExtractionDurationMs: maintenance.graphExtractionDurationMs,
|
|
334
|
+
orphansPurged: maintenance.orphansPurged,
|
|
335
|
+
proposalsExpired: maintenance.proposalsExpired,
|
|
786
336
|
};
|
|
787
337
|
}
|
|
788
|
-
|
|
789
|
-
|
|
338
|
+
/**
|
|
339
|
+
* Memory inference → index what it wrote → graph extraction → proposal hygiene
|
|
340
|
+
* (orphan purge, expiration) → orphan-state GC → retention purges. Warnings go
|
|
341
|
+
* to `allWarnings`.
|
|
342
|
+
*/
|
|
790
343
|
export async function runImproveMaintenancePasses(args) {
|
|
791
|
-
const { options, primaryStashDir,
|
|
792
|
-
if (!primaryStashDir)
|
|
793
|
-
return { memoryInferenceDurationMs: 0, graphExtractionDurationMs: 0 };
|
|
794
|
-
if (budgetSignal?.aborted)
|
|
344
|
+
const { options, primaryStashDir, allWarnings, budgetSignal, eventsCtx } = args;
|
|
345
|
+
if (!primaryStashDir || budgetSignal?.aborted)
|
|
795
346
|
return { memoryInferenceDurationMs: 0, graphExtractionDurationMs: 0 };
|
|
796
347
|
const config = options.config ?? loadConfig();
|
|
797
|
-
const sources = resolveSourceEntries(options.stashDir, config);
|
|
798
|
-
const memoryInferenceFn = options.memoryInferenceFn ?? runMemoryInferencePass;
|
|
799
|
-
const graphExtractionFn = options.graphExtractionFn ?? runGraphExtractionPass;
|
|
800
|
-
const openIndexDb = () => openIndexDatabase(getDbPath(), config.embedding?.dimension ? { embeddingDim: config.embedding.dimension } : undefined);
|
|
801
|
-
const dbCell = {};
|
|
802
348
|
const ctx = {
|
|
803
349
|
config,
|
|
804
|
-
sources,
|
|
350
|
+
sources: resolveSourceEntries(options.stashDir, config),
|
|
805
351
|
primaryStashDir,
|
|
806
352
|
eventsCtx,
|
|
807
353
|
budgetSignal,
|
|
808
354
|
improveProfile: args.improveProfile,
|
|
809
355
|
resolvedPlan: args.resolvedPlan,
|
|
810
|
-
memoryInferenceFn,
|
|
811
|
-
graphExtractionFn,
|
|
356
|
+
memoryInferenceFn: options.memoryInferenceFn ?? runMemoryInferencePass,
|
|
357
|
+
graphExtractionFn: options.graphExtractionFn ?? runGraphExtractionPass,
|
|
812
358
|
};
|
|
813
|
-
const
|
|
814
|
-
|
|
815
|
-
memoryRefsForInference,
|
|
816
|
-
allWarnings,
|
|
817
|
-
openIndexDb,
|
|
818
|
-
});
|
|
819
|
-
return {
|
|
820
|
-
...(collected.memoryInference ? { memoryInference: collected.memoryInference } : {}),
|
|
821
|
-
...(collected.graphExtraction ? { graphExtraction: collected.graphExtraction } : {}),
|
|
822
|
-
...(collected.actions.length > 0 ? { actions: collected.actions } : {}),
|
|
823
|
-
memoryInferenceDurationMs: collected.memoryInferenceDurationMs,
|
|
824
|
-
graphExtractionDurationMs: collected.graphExtractionDurationMs,
|
|
825
|
-
orphansPurged: collected.orphansPurged,
|
|
826
|
-
proposalsExpired: collected.proposalsExpired,
|
|
827
|
-
};
|
|
828
|
-
}
|
|
829
|
-
/**
|
|
830
|
-
* The maintenance sequence (formerly the ~389-line anonymous
|
|
831
|
-
* `withIndexWriterLease` callback, before #872 removed the index-rebuild
|
|
832
|
-
* lease): memory inference → reindex-after-inference → graph extraction →
|
|
833
|
-
* proposal hygiene (orphan purge, expiration) → retention purges. Each pass
|
|
834
|
-
* returns its results and warnings; this orchestrator folds warnings into the
|
|
835
|
-
* caller's `allWarnings` sink at the same points the inline code pushed them.
|
|
836
|
-
*/
|
|
837
|
-
async function runMaintenancePassesUnderLease(ctx, dbCell, args) {
|
|
838
|
-
const { allWarnings } = args;
|
|
359
|
+
const openIndexDb = () => openIndexDatabase(getDbPath());
|
|
360
|
+
const dbCell = {};
|
|
839
361
|
const actions = [];
|
|
840
362
|
try {
|
|
841
|
-
dbCell.current =
|
|
363
|
+
dbCell.current = openIndexDb();
|
|
842
364
|
const inference = await runMemoryInferenceMaintenancePass(ctx, dbCell, args.memoryRefsForInference);
|
|
843
365
|
if (inference.action)
|
|
844
366
|
actions.push(inference.action);
|
|
845
367
|
allWarnings.push(...inference.warnings);
|
|
846
|
-
const
|
|
847
|
-
|
|
848
|
-
|
|
849
|
-
|
|
850
|
-
|
|
851
|
-
info(`[improve] indexing ${memoryInference.writtenPaths.length} file(s) written by memory inference`);
|
|
368
|
+
const written = inference.memoryInference?.writtenPaths ?? [];
|
|
369
|
+
if (written.length > 0) {
|
|
370
|
+
// Index exactly the files inference wrote. indexWrittenAssets opens its
|
|
371
|
+
// own write handle, so ours closes first and reopens after.
|
|
372
|
+
info(`[improve] indexing ${written.length} file(s) written by memory inference`);
|
|
852
373
|
try {
|
|
853
|
-
|
|
854
|
-
// index.db WAL file, so the maintenance handle must be closed first
|
|
855
|
-
// and a fresh one reopened after, even on failure.
|
|
856
|
-
if (dbCell.current) {
|
|
374
|
+
if (dbCell.current)
|
|
857
375
|
closeDatabase(dbCell.current);
|
|
858
|
-
|
|
859
|
-
}
|
|
376
|
+
dbCell.current = undefined;
|
|
860
377
|
try {
|
|
861
|
-
await indexWrittenAssets(
|
|
378
|
+
await indexWrittenAssets(primaryStashDir, written);
|
|
862
379
|
}
|
|
863
380
|
finally {
|
|
864
|
-
dbCell.current =
|
|
381
|
+
dbCell.current = openIndexDb();
|
|
865
382
|
}
|
|
866
383
|
info("[improve] indexing after memory inference complete");
|
|
867
384
|
}
|
|
@@ -869,32 +386,22 @@ async function runMaintenancePassesUnderLease(ctx, dbCell, args) {
|
|
|
869
386
|
allWarnings.push(`indexing after memory inference failed: ${errMessage(err)}`);
|
|
870
387
|
}
|
|
871
388
|
}
|
|
872
|
-
const graph = await runGraphExtractionMaintenancePass(ctx, dbCell,
|
|
873
|
-
actionableRefs: args.actionableRefs,
|
|
874
|
-
memoryRefsForInference: args.memoryRefsForInference,
|
|
875
|
-
});
|
|
389
|
+
const graph = await runGraphExtractionMaintenancePass(ctx, dbCell, args);
|
|
876
390
|
if (graph.action)
|
|
877
391
|
actions.push(graph.action);
|
|
878
392
|
allWarnings.push(...graph.warnings);
|
|
879
|
-
const
|
|
880
|
-
allWarnings.push(...
|
|
881
|
-
|
|
882
|
-
// asset_salience/asset_outcome rows whose ref no longer resolves in
|
|
883
|
-
// index.db. Needs the SAME already-open index.db handle (dbCell.current)
|
|
884
|
-
// the passes above share.
|
|
885
|
-
const stateGc = runOrphanStateGcPass(ctx, dbCell);
|
|
886
|
-
allWarnings.push(...stateGc.warnings);
|
|
887
|
-
const expiration = runProposalExpirationPass(ctx);
|
|
888
|
-
allWarnings.push(...expiration.warnings);
|
|
393
|
+
const hygiene = runProposalHygienePass(ctx);
|
|
394
|
+
allWarnings.push(...hygiene.warnings);
|
|
395
|
+
allWarnings.push(...runOrphanStateGcPass(ctx, dbCell).warnings);
|
|
889
396
|
allWarnings.push(...runRetentionPurgePass(ctx).warnings);
|
|
890
397
|
return {
|
|
891
|
-
memoryInference,
|
|
892
|
-
graphExtraction: graph.graphExtraction,
|
|
893
|
-
actions,
|
|
398
|
+
...(inference.memoryInference ? { memoryInference: inference.memoryInference } : {}),
|
|
399
|
+
...(graph.graphExtraction ? { graphExtraction: graph.graphExtraction } : {}),
|
|
400
|
+
...(actions.length > 0 ? { actions } : {}),
|
|
894
401
|
memoryInferenceDurationMs: inference.durationMs,
|
|
895
402
|
graphExtractionDurationMs: graph.durationMs,
|
|
896
|
-
orphansPurged:
|
|
897
|
-
proposalsExpired:
|
|
403
|
+
orphansPurged: hygiene.orphansPurged,
|
|
404
|
+
proposalsExpired: hygiene.proposalsExpired,
|
|
898
405
|
};
|
|
899
406
|
}
|
|
900
407
|
finally {
|
|
@@ -902,542 +409,322 @@ async function runMaintenancePassesUnderLease(ctx, dbCell, args) {
|
|
|
902
409
|
closeDatabase(dbCell.current);
|
|
903
410
|
}
|
|
904
411
|
}
|
|
412
|
+
/** Time one LLM maintenance pass; a throw becomes a `<label> failed: …` warning. */
|
|
413
|
+
async function timedLlmPass(label, run) {
|
|
414
|
+
const start = Date.now();
|
|
415
|
+
try {
|
|
416
|
+
const result = await run();
|
|
417
|
+
return { result, durationMs: Date.now() - start, warnings: [] };
|
|
418
|
+
}
|
|
419
|
+
catch (err) {
|
|
420
|
+
return { durationMs: Date.now() - start, warnings: [`${label} failed: ${errMessage(err)}`] };
|
|
421
|
+
}
|
|
422
|
+
}
|
|
905
423
|
/**
|
|
906
|
-
* Memory inference
|
|
907
|
-
*
|
|
908
|
-
* was gated on memoryRefsForInference.size > 0 AND passed those refs as a
|
|
909
|
-
* candidateRefs filter. But memoryRefsForInference is populated from refs
|
|
910
|
-
* distilled THIS RUN — by the time that happens, those parents are
|
|
911
|
-
* already split (`inferenceProcessed: true`) and `isPendingMemory` excludes
|
|
912
|
-
* them. The genuinely-pending parents in the stash never entered the
|
|
913
|
-
* filter. Result: 0/0/0 for 25 consecutive runs.
|
|
914
|
-
*
|
|
915
|
-
* Fix: always run the pass when the feature is enabled; let the pass's
|
|
916
|
-
* own `collectPendingMemories` + `isPendingMemory` predicate find
|
|
917
|
-
* candidates from the filesystem-of-truth. The this-run set is still
|
|
918
|
-
* logged as a hint but no longer used as a filter.
|
|
424
|
+
* Memory inference over every pending parent in the stash. The pass discovers
|
|
425
|
+
* its own candidates; the refs distilled this run are only logged as a hint.
|
|
919
426
|
*/
|
|
920
427
|
export async function runMemoryInferenceMaintenancePass(ctx, dbCell, memoryRefsForInference) {
|
|
921
|
-
const { config, sources, primaryStashDir,
|
|
922
|
-
const
|
|
923
|
-
|
|
924
|
-
|
|
925
|
-
|
|
926
|
-
|
|
927
|
-
const minPendingCount =
|
|
928
|
-
|
|
929
|
-
if (!primaryStashDir || minPendingCount === undefined || minPendingCount <= 0)
|
|
930
|
-
return false;
|
|
428
|
+
const { config, sources, primaryStashDir, resolvedPlan } = ctx;
|
|
429
|
+
const settings = ctx.improveProfile?.processes?.memoryInference;
|
|
430
|
+
if (settings?.enabled === false) {
|
|
431
|
+
info("[improve] memory inference skipped (disabled by improve profile)");
|
|
432
|
+
return { durationMs: 0, warnings: [] };
|
|
433
|
+
}
|
|
434
|
+
const minPendingCount = settings?.minPendingCount;
|
|
435
|
+
if (primaryStashDir && minPendingCount !== undefined && minPendingCount > 0) {
|
|
931
436
|
const pending = collectPendingMemories(primaryStashDir).length;
|
|
932
437
|
if (pending < minPendingCount) {
|
|
933
438
|
info(`[improve] memory inference skipped (${pending} pending < minPendingCount ${minPendingCount})`);
|
|
934
|
-
return
|
|
935
|
-
}
|
|
936
|
-
return false;
|
|
937
|
-
})();
|
|
938
|
-
if (memoryInferenceDisabledByProfile) {
|
|
939
|
-
info("[improve] memory inference skipped (disabled by improve profile)");
|
|
940
|
-
}
|
|
941
|
-
else if (pendingBelowMinCount) {
|
|
942
|
-
// skipped — message already emitted above
|
|
943
|
-
}
|
|
944
|
-
else {
|
|
945
|
-
const hintRefs = memoryRefsForInference.size;
|
|
946
|
-
info(hintRefs > 0
|
|
947
|
-
? `[improve] memory inference starting (${hintRefs} hint refs touched this run; pass discovers all pending)`
|
|
948
|
-
: "[improve] memory inference starting (discovering pending parents)");
|
|
949
|
-
const inferenceStart = Date.now();
|
|
950
|
-
try {
|
|
951
|
-
// O-1 (#364): pass budget signal so a hung inference call is cancelled.
|
|
952
|
-
memoryInference = await withLlmStage("memory-inference", () => memoryInferenceFn({
|
|
953
|
-
config,
|
|
954
|
-
...(resolvedPlan
|
|
955
|
-
? {
|
|
956
|
-
llmRunner: resolvedPlan.processes.memoryInference.runner,
|
|
957
|
-
}
|
|
958
|
-
: {}),
|
|
959
|
-
sources,
|
|
960
|
-
signal: budgetSignal,
|
|
961
|
-
db: dbCell.current,
|
|
962
|
-
reEnrich: false,
|
|
963
|
-
onProgress: (event) => {
|
|
964
|
-
const current = event.currentRef ? ` ${event.currentRef}` : "";
|
|
965
|
-
info(`[improve] memory inference ${event.processed}/${event.total}${current} (written ${event.writtenFacts}, skipped ${event.skippedNoFacts})`);
|
|
966
|
-
},
|
|
967
|
-
}), { engine: resolvedPlan?.processes.memoryInference.runner?.engine, process: "memoryInference" });
|
|
968
|
-
durationMs = Date.now() - inferenceStart;
|
|
969
|
-
// Synthetic sentinel ref (ref-grammar decision D-R3): a colon-free
|
|
970
|
-
// `<domain>/_<marker>` label on the event row, never parsed as an asset
|
|
971
|
-
// ref. The domain is the asset stash-subdir for asset-scoped sentinels
|
|
972
|
-
// (`memories/…`) and the subsystem name for maintenance/artifact sentinels
|
|
973
|
-
// (`graph/…`, `events/…`, `proposals/…`, `health/…`, …). Readers match the
|
|
974
|
-
// event by `eventType`, never by this string.
|
|
975
|
-
action = { ref: "memories/_inference", mode: "memory-inference", result: memoryInference };
|
|
976
|
-
info(`[improve] memory inference complete (${memoryInference.writtenFacts} facts written from ${memoryInference.splitParents} parents)`);
|
|
977
|
-
}
|
|
978
|
-
catch (err) {
|
|
979
|
-
durationMs = Date.now() - inferenceStart;
|
|
980
|
-
warnings.push(`memory inference failed: ${errMessage(err)}`);
|
|
439
|
+
return { durationMs: 0, warnings: [] };
|
|
981
440
|
}
|
|
982
441
|
}
|
|
983
|
-
|
|
442
|
+
const hintRefs = memoryRefsForInference.size;
|
|
443
|
+
info(hintRefs > 0
|
|
444
|
+
? `[improve] memory inference starting (${hintRefs} hint refs touched this run; pass discovers all pending)`
|
|
445
|
+
: "[improve] memory inference starting (discovering pending parents)");
|
|
446
|
+
const pass = await timedLlmPass("memory inference", () => attributeStage(resolvedPlan, "memoryInference", () => ctx.memoryInferenceFn({
|
|
447
|
+
config,
|
|
448
|
+
...(resolvedPlan ? { llmRunner: resolvedPlan.processes.memoryInference.runner } : {}),
|
|
449
|
+
sources,
|
|
450
|
+
signal: ctx.budgetSignal,
|
|
451
|
+
db: dbCell.current,
|
|
452
|
+
reEnrich: false,
|
|
453
|
+
onProgress: (event) => {
|
|
454
|
+
const current = event.currentRef ? ` ${event.currentRef}` : "";
|
|
455
|
+
info(`[improve] memory inference ${event.processed}/${event.total}${current} (written ${event.writtenFacts}, skipped ${event.skippedNoFacts})`);
|
|
456
|
+
},
|
|
457
|
+
})));
|
|
458
|
+
const memoryInference = pass.result;
|
|
459
|
+
if (!memoryInference)
|
|
460
|
+
return { durationMs: pass.durationMs, warnings: pass.warnings };
|
|
461
|
+
info(`[improve] memory inference complete (${memoryInference.writtenFacts} facts written from ${memoryInference.splitParents} parents)`);
|
|
462
|
+
return {
|
|
463
|
+
memoryInference,
|
|
464
|
+
durationMs: pass.durationMs,
|
|
465
|
+
// Sentinel refs (`<domain>/_<marker>`) label maintenance events; never parsed as assets.
|
|
466
|
+
action: { ref: "memories/_inference", mode: "memory-inference", result: memoryInference },
|
|
467
|
+
warnings: pass.warnings,
|
|
468
|
+
};
|
|
984
469
|
}
|
|
985
470
|
/**
|
|
986
|
-
* Graph
|
|
987
|
-
*
|
|
988
|
-
*
|
|
989
|
-
* actionable refs (candidatePaths). Full-corpus scans are opt-in via
|
|
990
|
-
* profile.processes.graphExtraction.fullScan = true (used by the
|
|
991
|
-
* `graph-refresh` built-in profile and its weekly scheduled task).
|
|
992
|
-
* The empty-Set fallback is intentional when no refs were touched —
|
|
993
|
-
* the extractor's filter rejects every file and returns empty, keeping
|
|
994
|
-
* the pass invoked so the action is recorded and tests stay exercised.
|
|
471
|
+
* Graph extraction over the files this run touched, or the whole corpus when
|
|
472
|
+
* the profile sets `graphExtraction.fullScan` (the `graph-refresh` strategy).
|
|
473
|
+
* With nothing touched the pass still runs and extracts nothing.
|
|
995
474
|
*/
|
|
996
475
|
export async function runGraphExtractionMaintenancePass(ctx, dbCell, args) {
|
|
997
|
-
const { config, sources, primaryStashDir,
|
|
998
|
-
const
|
|
999
|
-
let graphExtraction;
|
|
1000
|
-
let durationMs = 0;
|
|
1001
|
-
let action;
|
|
476
|
+
const { config, sources, primaryStashDir, resolvedPlan } = ctx;
|
|
477
|
+
const settings = ctx.improveProfile?.processes?.graphExtraction;
|
|
1002
478
|
const graphEnabled = resolvedPlan ? true : isProcessEnabled("index", "graph_extraction", config);
|
|
1003
|
-
|
|
1004
|
-
const graphExtractionFullScan = improveProfile?.processes?.graphExtraction?.fullScan === true;
|
|
1005
|
-
// #624 P2: optional incremental high-signal-first cap. Unset = process all
|
|
1006
|
-
// eligible (byte-identical to today; no ranking/slice).
|
|
1007
|
-
const graphExtractionTopN = improveProfile?.processes?.graphExtraction?.topN;
|
|
1008
|
-
const graphExtractionIncludeTypes = improveProfile?.processes?.graphExtraction?.includeTypes ?? [
|
|
1009
|
-
...DEFAULT_GRAPH_EXTRACTION_INCLUDE_TYPES,
|
|
1010
|
-
];
|
|
1011
|
-
const graphExtractionBatchSize = improveProfile?.processes?.graphExtraction?.batchSize ?? DEFAULT_GRAPH_EXTRACTION_BATCH_SIZE;
|
|
1012
|
-
const graphExtractionMaxChunksPerAsset = improveProfile?.processes?.graphExtraction?.maxChunksPerAsset;
|
|
1013
|
-
// Build the set of refs actually touched this run.
|
|
1014
|
-
const touchedRefs = new Set();
|
|
1015
|
-
for (const r of args.actionableRefs)
|
|
1016
|
-
touchedRefs.add(r.ref);
|
|
1017
|
-
for (const r of args.memoryRefsForInference)
|
|
1018
|
-
touchedRefs.add(r);
|
|
1019
|
-
if (graphExtractionDisabledByProfile) {
|
|
479
|
+
if (settings?.enabled === false) {
|
|
1020
480
|
info("[improve] graph extraction skipped (disabled by improve profile)");
|
|
481
|
+
return { durationMs: 0, warnings: [] };
|
|
1021
482
|
}
|
|
1022
|
-
|
|
1023
|
-
|
|
1024
|
-
|
|
1025
|
-
|
|
1026
|
-
|
|
1027
|
-
|
|
1028
|
-
|
|
1029
|
-
|
|
1030
|
-
|
|
1031
|
-
|
|
1032
|
-
|
|
1033
|
-
|
|
1034
|
-
|
|
1035
|
-
|
|
1036
|
-
|
|
1037
|
-
|
|
1038
|
-
|
|
483
|
+
if (sources.length === 0)
|
|
484
|
+
return { durationMs: 0, warnings: [] };
|
|
485
|
+
if (!graphEnabled) {
|
|
486
|
+
info("[improve] graph extraction skipped (features.index.graph_extraction is disabled)");
|
|
487
|
+
return { durationMs: 0, warnings: [] };
|
|
488
|
+
}
|
|
489
|
+
const fullScan = settings?.fullScan === true;
|
|
490
|
+
info(`[improve] graph extraction starting${fullScan ? " (full-corpus scan)" : ""}`);
|
|
491
|
+
const pass = await timedLlmPass("graph extraction", async () => {
|
|
492
|
+
let candidatePaths;
|
|
493
|
+
if (!fullScan) {
|
|
494
|
+
candidatePaths = new Set();
|
|
495
|
+
const touched = new Set([...args.actionableRefs.map((r) => r.ref), ...args.memoryRefsForInference]);
|
|
496
|
+
if (primaryStashDir && touched.size > 0) {
|
|
497
|
+
const writableBundleIds = deriveWritableBundleIds(resolveSourceEntries(primaryStashDir));
|
|
498
|
+
const resolved = await Promise.all([...touched].map((ref) => findAssetFilePath(ref, primaryStashDir, writableBundleIds).catch(() => null)));
|
|
499
|
+
for (const p of resolved)
|
|
500
|
+
if (typeof p === "string" && p.length > 0)
|
|
501
|
+
candidatePaths.add(p);
|
|
1039
502
|
}
|
|
1040
|
-
|
|
503
|
+
}
|
|
504
|
+
return attributeStage(resolvedPlan, "graphExtraction", () => ctx.graphExtractionFn({
|
|
505
|
+
config,
|
|
506
|
+
...(resolvedPlan ? { llmRunner: resolvedPlan.processes.graphExtraction.runner } : {}),
|
|
507
|
+
sources,
|
|
508
|
+
signal: ctx.budgetSignal,
|
|
509
|
+
db: dbCell.current,
|
|
510
|
+
reEnrich: false,
|
|
511
|
+
onProgress: (event) => {
|
|
1041
512
|
const current = event.currentPath ? ` ${path.basename(event.currentPath)}` : "";
|
|
1042
513
|
info(`[improve] graph extraction ${event.processed}/${event.total}${current} (extracted ${event.extracted}, entities ${event.totalEntities}, relations ${event.totalRelations})`);
|
|
1043
|
-
}
|
|
1044
|
-
|
|
1045
|
-
|
|
1046
|
-
|
|
1047
|
-
|
|
1048
|
-
|
|
1049
|
-
|
|
1050
|
-
|
|
1051
|
-
|
|
1052
|
-
|
|
1053
|
-
|
|
1054
|
-
|
|
1055
|
-
|
|
1056
|
-
|
|
1057
|
-
|
|
1058
|
-
|
|
1059
|
-
|
|
1060
|
-
|
|
1061
|
-
|
|
1062
|
-
|
|
1063
|
-
? { maxChunksPerAsset: graphExtractionMaxChunksPerAsset }
|
|
1064
|
-
: {}),
|
|
1065
|
-
},
|
|
1066
|
-
}), { engine: resolvedPlan?.processes.graphExtraction.runner?.engine, process: "graphExtraction" });
|
|
1067
|
-
durationMs = Date.now() - extractionStart;
|
|
1068
|
-
// Synthetic sentinel ref (D-R3): `graph` has no asset stash-subdir, so the
|
|
1069
|
-
// colon-free `graph/_artifact` names the subsystem, per the sentinel
|
|
1070
|
-
// convention documented at the memory-inference writer above.
|
|
1071
|
-
action = { ref: "graph/_artifact", mode: "graph-extraction", result: graphExtraction };
|
|
1072
|
-
info(`[improve] graph extraction complete (${graphExtraction.quality.extractedFiles} files, ${graphExtraction.quality.entityCount} entities, ${graphExtraction.quality.relationCount} relations)`);
|
|
1073
|
-
}
|
|
1074
|
-
catch (err) {
|
|
1075
|
-
durationMs = Date.now() - extractionStart;
|
|
1076
|
-
warnings.push(`graph extraction failed: ${errMessage(err)}`);
|
|
1077
|
-
}
|
|
1078
|
-
}
|
|
1079
|
-
else if (sources.length > 0 && !graphEnabled) {
|
|
1080
|
-
info("[improve] graph extraction skipped (features.index.graph_extraction is disabled)");
|
|
1081
|
-
}
|
|
1082
|
-
return { graphExtraction, durationMs, action, warnings };
|
|
514
|
+
},
|
|
515
|
+
options: {
|
|
516
|
+
candidatePaths,
|
|
517
|
+
includeTypes: settings?.includeTypes ?? [...DEFAULT_GRAPH_EXTRACTION_INCLUDE_TYPES],
|
|
518
|
+
batchSize: settings?.batchSize ?? DEFAULT_GRAPH_EXTRACTION_BATCH_SIZE,
|
|
519
|
+
...(settings?.topN != null ? { topN: settings.topN } : {}),
|
|
520
|
+
...(settings?.maxChunksPerAsset != null ? { maxChunksPerAsset: settings.maxChunksPerAsset } : {}),
|
|
521
|
+
},
|
|
522
|
+
}));
|
|
523
|
+
});
|
|
524
|
+
const graphExtraction = pass.result;
|
|
525
|
+
if (!graphExtraction)
|
|
526
|
+
return { durationMs: pass.durationMs, warnings: pass.warnings };
|
|
527
|
+
info(`[improve] graph extraction complete (${graphExtraction.quality.extractedFiles} files, ${graphExtraction.quality.entityCount} entities, ${graphExtraction.quality.relationCount} relations)`);
|
|
528
|
+
return {
|
|
529
|
+
graphExtraction,
|
|
530
|
+
durationMs: pass.durationMs,
|
|
531
|
+
action: { ref: "graph/_artifact", mode: "graph-extraction", result: graphExtraction },
|
|
532
|
+
warnings: pass.warnings,
|
|
533
|
+
};
|
|
1083
534
|
}
|
|
1084
535
|
/**
|
|
1085
|
-
*
|
|
1086
|
-
*
|
|
1087
|
-
* promoted assets from accept flows during this run are already present.
|
|
536
|
+
* Reject pending proposals whose target no longer exists, then expire pending
|
|
537
|
+
* proposals past the retention window; each emits a roll-up event.
|
|
1088
538
|
*/
|
|
1089
|
-
function
|
|
1090
|
-
const { primaryStashDir,
|
|
539
|
+
function runProposalHygienePass(ctx) {
|
|
540
|
+
const { primaryStashDir, eventsCtx } = ctx;
|
|
1091
541
|
const warnings = [];
|
|
1092
542
|
let orphansPurged = 0;
|
|
543
|
+
let proposalsExpired = 0;
|
|
1093
544
|
try {
|
|
1094
|
-
const
|
|
1095
|
-
orphansPurged =
|
|
1096
|
-
if (
|
|
1097
|
-
info(`[improve] orphan purge: ${
|
|
545
|
+
const purge = purgeOrphanProposals(primaryStashDir, ctx.sources.map((s) => s.path));
|
|
546
|
+
orphansPurged = purge.rejected;
|
|
547
|
+
if (purge.rejected > 0) {
|
|
548
|
+
info(`[improve] orphan purge: ${purge.rejected}/${purge.checked} orphaned proposals rejected (${purge.durationMs}ms)`);
|
|
1098
549
|
}
|
|
1099
550
|
appendEvent({
|
|
1100
551
|
eventType: "proposal_orphan_purge",
|
|
1101
552
|
ref: "proposals/_orphan-purge",
|
|
1102
553
|
metadata: {
|
|
1103
|
-
checked:
|
|
1104
|
-
rejected:
|
|
1105
|
-
durationMs:
|
|
1106
|
-
byType:
|
|
1107
|
-
orphans:
|
|
554
|
+
checked: purge.checked,
|
|
555
|
+
rejected: purge.rejected,
|
|
556
|
+
durationMs: purge.durationMs,
|
|
557
|
+
byType: purge.byType,
|
|
558
|
+
orphans: purge.orphans.map((o) => o.ref),
|
|
1108
559
|
},
|
|
1109
560
|
}, eventsCtx);
|
|
1110
561
|
}
|
|
1111
562
|
catch (err) {
|
|
1112
563
|
warnings.push(`orphan purge failed: ${errMessage(err)}`);
|
|
1113
564
|
}
|
|
1114
|
-
return { orphansPurged, warnings };
|
|
1115
|
-
}
|
|
1116
|
-
/**
|
|
1117
|
-
* Phase 6B (Advantage D6b): expire pending proposals that have aged past
|
|
1118
|
-
* the retention window. Runs AFTER orphan purge so we never double-archive
|
|
1119
|
-
* a proposal that orphan-purge already moved. `expireStaleProposals` emits
|
|
1120
|
-
* its own per-proposal `proposal_expired` events; we additionally emit a
|
|
1121
|
-
* single roll-up event here for parity with the orphan-purge surface.
|
|
1122
|
-
*/
|
|
1123
|
-
function runProposalExpirationPass(ctx) {
|
|
1124
|
-
const { primaryStashDir, config, eventsCtx } = ctx;
|
|
1125
|
-
const warnings = [];
|
|
1126
|
-
let proposalsExpired = 0;
|
|
1127
565
|
try {
|
|
1128
|
-
const
|
|
1129
|
-
proposalsExpired =
|
|
1130
|
-
if (
|
|
1131
|
-
info(`[improve] expiration: ${
|
|
1132
|
-
`(retention=${
|
|
566
|
+
const expiry = expireStaleProposals(primaryStashDir, ctx.config);
|
|
567
|
+
proposalsExpired = expiry.expired;
|
|
568
|
+
if (expiry.expired > 0) {
|
|
569
|
+
info(`[improve] expiration: ${expiry.expired}/${expiry.checked} pending proposals expired ` +
|
|
570
|
+
`(retention=${expiry.retentionDays}d, ${expiry.durationMs}ms)`);
|
|
1133
571
|
}
|
|
1134
572
|
appendEvent({
|
|
1135
573
|
eventType: "proposal_expiration_pass",
|
|
1136
574
|
ref: "proposals/_expiration",
|
|
1137
575
|
metadata: {
|
|
1138
|
-
checked:
|
|
1139
|
-
expired:
|
|
1140
|
-
durationMs:
|
|
1141
|
-
retentionDays:
|
|
1142
|
-
expiredProposals:
|
|
576
|
+
checked: expiry.checked,
|
|
577
|
+
expired: expiry.expired,
|
|
578
|
+
durationMs: expiry.durationMs,
|
|
579
|
+
retentionDays: expiry.retentionDays,
|
|
580
|
+
expiredProposals: expiry.expiredProposals,
|
|
1143
581
|
},
|
|
1144
582
|
}, eventsCtx);
|
|
1145
583
|
}
|
|
1146
584
|
catch (err) {
|
|
1147
585
|
warnings.push(`proposal expiration failed: ${errMessage(err)}`);
|
|
1148
586
|
}
|
|
1149
|
-
return { proposalsExpired, warnings };
|
|
587
|
+
return { orphansPurged, proposalsExpired, warnings };
|
|
1150
588
|
}
|
|
1151
589
|
/**
|
|
1152
|
-
*
|
|
1153
|
-
*
|
|
1154
|
-
*
|
|
1155
|
-
*
|
|
1156
|
-
*
|
|
1157
|
-
*
|
|
1158
|
-
* the index handle the other passes use).
|
|
590
|
+
* Trim the observability data that grows append-only — state.db events and
|
|
591
|
+
* improve_runs, logs.db task_logs, and the per-run task log files — to
|
|
592
|
+
* `improve.eventRetentionDays` (default 90; 0 disables), then VACUUM state.db
|
|
593
|
+
* when enough pages are free. Each store fails on its own. state.db work
|
|
594
|
+
* borrows the run's long-lived handle: a second writer on the same WAL file
|
|
595
|
+
* locks (#585).
|
|
1159
596
|
*/
|
|
1160
597
|
export function runRetentionPurgePass(ctx) {
|
|
1161
598
|
const { config, eventsCtx } = ctx;
|
|
1162
599
|
const warnings = [];
|
|
1163
600
|
const retentionDays = typeof config.improve?.eventRetentionDays === "number" ? config.improve.eventRetentionDays : 90;
|
|
1164
|
-
if (retentionDays
|
|
1165
|
-
|
|
1166
|
-
|
|
1167
|
-
|
|
1168
|
-
|
|
1169
|
-
|
|
1170
|
-
|
|
1171
|
-
|
|
1172
|
-
|
|
1173
|
-
|
|
1174
|
-
|
|
1175
|
-
|
|
1176
|
-
|
|
1177
|
-
}
|
|
1178
|
-
|
|
1179
|
-
|
|
1180
|
-
|
|
1181
|
-
|
|
1182
|
-
|
|
1183
|
-
|
|
1184
|
-
|
|
1185
|
-
|
|
1186
|
-
|
|
1187
|
-
|
|
1188
|
-
|
|
1189
|
-
|
|
1190
|
-
|
|
1191
|
-
|
|
1192
|
-
eventType: "improve_runs_purged",
|
|
1193
|
-
ref: "improve_runs/_purge",
|
|
1194
|
-
metadata: { purgedCount: improveRunsPurged, retentionDays },
|
|
1195
|
-
}, eventsCtx);
|
|
1196
|
-
// R5: improve_cycle_metrics has its OWN retention window
|
|
1197
|
-
// (default 365d — a slow collapse needs a longer trend than
|
|
1198
|
-
// the 90d events window). canary_queries rows are never purged.
|
|
1199
|
-
const cycleRetention = config.improve?.collapseDetector?.retentionDays ?? CYCLE_METRICS_RETENTION_DAYS;
|
|
1200
|
-
const cycleMetricsPurged = purgeOldCycleMetrics(stateDb, cycleRetention);
|
|
1201
|
-
if (cycleMetricsPurged > 0) {
|
|
1202
|
-
info(`[improve] cycle-metrics purge: ${cycleMetricsPurged} row(s) older than ${cycleRetention}d removed from state.db`);
|
|
1203
|
-
appendEvent({
|
|
1204
|
-
// Dedicated type (mirrors improve_runs_purged) so consumers
|
|
1205
|
-
// never have to disambiguate purge targets via the ref string.
|
|
1206
|
-
eventType: "improve_cycle_metrics_purged",
|
|
1207
|
-
ref: "improve_cycle_metrics/_purge",
|
|
1208
|
-
metadata: { purgedCount: cycleMetricsPurged, retentionDays: cycleRetention },
|
|
1209
|
-
}, eventsCtx);
|
|
1210
|
-
}
|
|
1211
|
-
// R0 step 3: opportunistic post-purge VACUUM. Reads the freelist
|
|
1212
|
-
// off this same connection (no second state.db handle) and only
|
|
1213
|
-
// runs when reclaimable space crosses STATE_DB_FREELIST_WARN_RATIO.
|
|
1214
|
-
const vacuumOutcome = vacuumStateDbIfReclaimable(stateDb, readFreelistInfo(stateDb), eventsCtx);
|
|
1215
|
-
if (vacuumOutcome.ran) {
|
|
1216
|
-
info(`[improve] state.db vacuum: ${vacuumOutcome.pagesBefore} -> ${vacuumOutcome.pagesAfter} pages`);
|
|
1217
|
-
}
|
|
1218
|
-
}, { path: eventsCtx?.dbPath, borrowed: eventsCtx?.db });
|
|
1219
|
-
}
|
|
1220
|
-
catch (err) {
|
|
1221
|
-
warnings.push(`events purge failed: ${errMessage(err)}`);
|
|
1222
|
-
}
|
|
1223
|
-
// task_logs in logs.db (#579) shares the same retention window as
|
|
1224
|
-
// events/improve_runs — all three are observability data governed by
|
|
1225
|
-
// the single improve.eventRetentionDays knob. Separate try/finally
|
|
1226
|
-
// because logs.db is a different file: a locked/missing logs.db must
|
|
1227
|
-
// not block the state.db purges above.
|
|
1228
|
-
let logsDb;
|
|
1229
|
-
try {
|
|
1230
|
-
logsDb = openLogsDatabase();
|
|
1231
|
-
const taskLogsPurged = purgeOldTaskLogs(logsDb, retentionDays);
|
|
1232
|
-
if (taskLogsPurged > 0) {
|
|
1233
|
-
info(`[improve] task_logs purge: ${taskLogsPurged} log line(s) older than ${retentionDays}d removed from logs.db`);
|
|
1234
|
-
}
|
|
1235
|
-
appendEvent({
|
|
1236
|
-
eventType: "task_logs_purged",
|
|
1237
|
-
ref: "task_logs/_purge",
|
|
1238
|
-
metadata: { purgedCount: taskLogsPurged, retentionDays },
|
|
1239
|
-
}, eventsCtx);
|
|
1240
|
-
}
|
|
1241
|
-
catch (err) {
|
|
1242
|
-
warnings.push(`task_logs purge failed: ${errMessage(err)}`);
|
|
1243
|
-
}
|
|
1244
|
-
finally {
|
|
1245
|
-
if (logsDb) {
|
|
1246
|
-
try {
|
|
1247
|
-
logsDb.close();
|
|
1248
|
-
}
|
|
1249
|
-
catch {
|
|
1250
|
-
// best-effort
|
|
1251
|
-
}
|
|
1252
|
-
}
|
|
1253
|
-
}
|
|
1254
|
-
// Per-run flat log files under getTaskLogDir() (#951): logs.db above is
|
|
1255
|
-
// the durable record and already retention-purged, so the transitional
|
|
1256
|
-
// `<taskId>/<timestamp>.log` tail files can be deleted on the same
|
|
1257
|
-
// window without losing anything. A separate try/catch — a filesystem
|
|
1258
|
-
// problem here must not block the DB purges above.
|
|
601
|
+
if (retentionDays <= 0)
|
|
602
|
+
return { warnings };
|
|
603
|
+
const report = (eventType, ref, purgedCount, what) => {
|
|
604
|
+
if (purgedCount > 0)
|
|
605
|
+
info(`[improve] ${eventType}: ${purgedCount} ${what} older than ${retentionDays}d removed`);
|
|
606
|
+
appendEvent({ eventType, ref, metadata: { purgedCount, retentionDays } }, eventsCtx);
|
|
607
|
+
};
|
|
608
|
+
try {
|
|
609
|
+
withStateDb((stateDb) => {
|
|
610
|
+
report("events_purged", "events/_purge", purgeOldEvents(stateDb, retentionDays), "event(s)");
|
|
611
|
+
report("improve_runs_purged", "improve_runs/_purge", purgeOldImproveRuns(stateDb, retentionDays), "run(s)");
|
|
612
|
+
const vacuum = vacuumIfReclaimable(stateDb, readFreelistInfo(stateDb), { eventType: STATE_DB_VACUUMED_EVENT }, eventsCtx);
|
|
613
|
+
if (vacuum.ran)
|
|
614
|
+
info(`[improve] state.db vacuum: ${vacuum.pagesBefore} -> ${vacuum.pagesAfter} pages`);
|
|
615
|
+
}, { path: eventsCtx?.dbPath, borrowed: eventsCtx?.db });
|
|
616
|
+
}
|
|
617
|
+
catch (err) {
|
|
618
|
+
warnings.push(`events purge failed: ${errMessage(err)}`);
|
|
619
|
+
}
|
|
620
|
+
let logsDb;
|
|
621
|
+
try {
|
|
622
|
+
logsDb = openLogsDatabase();
|
|
623
|
+
report("task_logs_purged", "task_logs/_purge", purgeOldTaskLogs(logsDb, retentionDays), "log line(s)");
|
|
624
|
+
}
|
|
625
|
+
catch (err) {
|
|
626
|
+
warnings.push(`task_logs purge failed: ${errMessage(err)}`);
|
|
627
|
+
}
|
|
628
|
+
finally {
|
|
1259
629
|
try {
|
|
1260
|
-
|
|
1261
|
-
if (taskLogFilesPurged > 0) {
|
|
1262
|
-
info(`[improve] task log files purge: ${taskLogFilesPurged} file(s) older than ${retentionDays}d removed from ${getTaskLogDir()}`);
|
|
1263
|
-
}
|
|
1264
|
-
appendEvent({
|
|
1265
|
-
eventType: "task_log_files_purged",
|
|
1266
|
-
ref: "task_log_files/_purge",
|
|
1267
|
-
metadata: { purgedCount: taskLogFilesPurged, retentionDays },
|
|
1268
|
-
}, eventsCtx);
|
|
630
|
+
logsDb?.close();
|
|
1269
631
|
}
|
|
1270
|
-
catch
|
|
1271
|
-
|
|
632
|
+
catch {
|
|
633
|
+
// best-effort
|
|
1272
634
|
}
|
|
1273
635
|
}
|
|
636
|
+
try {
|
|
637
|
+
report("task_log_files_purged", "task_log_files/_purge", purgeOldTaskLogFiles(undefined, retentionDays), `file(s) under ${getTaskLogDir()}`);
|
|
638
|
+
}
|
|
639
|
+
catch (err) {
|
|
640
|
+
warnings.push(`task log files purge failed: ${errMessage(err)}`);
|
|
641
|
+
}
|
|
1274
642
|
return { warnings };
|
|
1275
643
|
}
|
|
1276
|
-
// ── #733 — orphan-state GC pass (Workstream C) ──────────────────────────────
|
|
1277
|
-
//
|
|
1278
|
-
// Deliberately lean: one maintenance pass, one additive migration (021), one
|
|
1279
|
-
// event type (asset_state_gc), one config gate (improve.stateGc.collect,
|
|
1280
|
-
// default false). See docs/architecture/specs/0.9.0-close-out-plan.md
|
|
1281
|
-
// Workstream C for the full design rationale. No quarantine archive, no
|
|
1282
|
-
// circuit breaker, no health-advisory plumbing, no new tables.
|
|
1283
644
|
/**
|
|
1284
|
-
* Grace window
|
|
1285
|
-
*
|
|
1286
|
-
* A named constant, not a config knob (owner ruling — see the close-out
|
|
1287
|
-
* plan's Workstream C). Mirrors `TXN_SWEEP_GRACE_MS` (src/core/fs-txn.ts:298).
|
|
645
|
+
* Grace window before an unresolved salience/outcome row may be deleted (only
|
|
646
|
+
* when `improve.stateGc.collect` is true).
|
|
1288
647
|
*/
|
|
1289
648
|
export const STATE_GC_GRACE_MS = daysToMs(7);
|
|
1290
649
|
/**
|
|
1291
|
-
*
|
|
1292
|
-
*
|
|
1293
|
-
*
|
|
1294
|
-
* predicate (see the pass doc comment below), so this checks the same two
|
|
1295
|
-
* spellings `getEntryByRef` (index-entries-repository.ts) resolves — an exact
|
|
1296
|
-
* bundle-qualified item_ref, or a bare conceptId matched by suffix across all
|
|
1297
|
-
* bundles — but against a prebuilt {@link LiveRefSnapshot}
|
|
1298
|
-
* (`getLiveRefSnapshot`) instead of a database round trip per row: with up to
|
|
1299
|
-
* a few thousand pending rows per run, one probe per row was the dominant
|
|
1300
|
-
* cost R78 (tier1-0917).
|
|
1301
|
-
*
|
|
1302
|
-
* On top of that, falls back to the BARE conceptId form (`bareImproveRef` —
|
|
1303
|
-
* the same primitive `preparation.ts`'s `normalizeStoredKey` map is built
|
|
1304
|
-
* from via `improveStateReadRefs`) when the stored ref carries a bundle
|
|
1305
|
-
* prefix that no longer matches exactly. This is the legacy-spelling
|
|
1306
|
-
* normalization trap: a naive `asset_ref NOT IN (SELECT item_ref FROM
|
|
1307
|
-
* entries)` would treat a live asset whose row predates bundle-qualification
|
|
1308
|
-
* (or whose bundle prefix is stale) as an orphan and delete it. Preferring
|
|
1309
|
-
* "never delete a live row" over "never miss a genuinely dead one" mirrors
|
|
1310
|
-
* `getEntryByRef`'s own bare-conceptId suffix-match trade-off.
|
|
650
|
+
* A stored state ref is live when it, or its bare conceptId, resolves in the
|
|
651
|
+
* index. The bare fallback keeps a live asset whose row predates
|
|
652
|
+
* bundle-qualification from being collected.
|
|
1311
653
|
*/
|
|
1312
654
|
function isStateRefLive(snapshot, storedRef) {
|
|
1313
655
|
if (isRefLiveInSnapshot(snapshot, storedRef))
|
|
1314
656
|
return true;
|
|
1315
|
-
const bare =
|
|
657
|
+
const bare = stripBundle(storedRef);
|
|
1316
658
|
return bare !== storedRef && isRefLiveInSnapshot(snapshot, bare);
|
|
1317
659
|
}
|
|
1318
660
|
/**
|
|
1319
|
-
*
|
|
1320
|
-
*
|
|
1321
|
-
*
|
|
1322
|
-
*
|
|
1323
|
-
*
|
|
1324
|
-
*
|
|
1325
|
-
*
|
|
1326
|
-
*/
|
|
1327
|
-
function gcOneStateTable(args) {
|
|
1328
|
-
const { refRows, liveRefs, now, collect, stamp, clear, deleteOlderThan, countPending } = args;
|
|
1329
|
-
const toStamp = [];
|
|
1330
|
-
const toClear = [];
|
|
1331
|
-
for (const row of refRows) {
|
|
1332
|
-
const live = isStateRefLive(liveRefs, row.asset_ref);
|
|
1333
|
-
if (!live && row.missing_since == null)
|
|
1334
|
-
toStamp.push(row.asset_ref);
|
|
1335
|
-
else if (live && row.missing_since != null)
|
|
1336
|
-
toClear.push(row.asset_ref);
|
|
1337
|
-
}
|
|
1338
|
-
if (toStamp.length > 0)
|
|
1339
|
-
stamp(toStamp, now);
|
|
1340
|
-
if (toClear.length > 0)
|
|
1341
|
-
clear(toClear);
|
|
1342
|
-
const collected = collect ? deleteOlderThan(now - STATE_GC_GRACE_MS) : 0;
|
|
1343
|
-
const pending = countPending();
|
|
1344
|
-
return { pending, collected };
|
|
1345
|
-
}
|
|
1346
|
-
/**
|
|
1347
|
-
* Orphan-state GC — #733 (Workstream C). For each of the two per-asset state
|
|
1348
|
-
* tables (`asset_salience`, `asset_outcome`), stamps `missing_since` on refs
|
|
1349
|
-
* that no longer resolve against `entries.item_ref` in index.db, clears the
|
|
1350
|
-
* stamp on refs that resolve again, and — only when `improve.stateGc.collect`
|
|
1351
|
-
* is true — deletes rows whose stamp is older than {@link STATE_GC_GRACE_MS}.
|
|
1352
|
-
*
|
|
1353
|
-
* "ref not present in entries.item_ref" IS the authoritative-deletion
|
|
1354
|
-
* predicate: "absent ≠ deleted" is inherited from the indexer, not
|
|
1355
|
-
* re-implemented here — an incomplete or failed source scan preserves that
|
|
1356
|
-
* source's last-known-good `entries` rows (indexer.ts ~1195-1199), and
|
|
1357
|
-
* mass-wipe is already gated upstream (`preserveExistingIndex` +
|
|
1358
|
-
* `fullDelete && scanComplete`). A temporarily unreachable source therefore
|
|
1359
|
-
* never surfaces candidates; no separate scan-status tracking is needed.
|
|
1360
|
-
*
|
|
1361
|
-
* Runs under the SAME index-writer lease / borrowed-state.db-connection
|
|
1362
|
-
* discipline as the neighboring maintenance passes: `dbCell.current` supplies
|
|
1363
|
-
* the already-open index.db handle (#584), and state.db access goes through
|
|
1364
|
-
* `withStateDb(..., { borrowed: eventsCtx?.db })` so this never opens a
|
|
1365
|
-
* second live writer alongside a long-lived `eventsCtx.db` connection — see
|
|
1366
|
-
* the #585 comment on {@link runRetentionPurgePass}'s events-purge call for
|
|
1367
|
-
* why that matters ("database is locked").
|
|
1368
|
-
*
|
|
1369
|
-
* Exported for direct test coverage (tests/integration/commands/improve/
|
|
1370
|
-
* state-gc.test.ts), mirroring the `runMemoryInferenceMaintenancePass` /
|
|
1371
|
-
* `runGraphExtractionMaintenancePass` / `runRetentionPurgePass` precedent;
|
|
1372
|
-
* production callers reach it only through `runMaintenancePassesUnderLease`.
|
|
661
|
+
* Orphan-state GC (#733) over `asset_salience` and `asset_outcome`: stamp
|
|
662
|
+
* `missing_since` on refs that no longer resolve in index.db, clear it on refs
|
|
663
|
+
* that resolve again, and — only with `improve.stateGc.collect` — delete rows
|
|
664
|
+
* stamped longer ago than {@link STATE_GC_GRACE_MS}. An unreachable source keeps
|
|
665
|
+
* its last-known index rows, so it never surfaces candidates. `pending` is the
|
|
666
|
+
* backlog after the sweep; the event is emitted only when there is something to
|
|
667
|
+
* report.
|
|
1373
668
|
*/
|
|
1374
669
|
export function runOrphanStateGcPass(ctx, dbCell) {
|
|
1375
|
-
const { eventsCtx
|
|
1376
|
-
const warnings = [];
|
|
670
|
+
const { eventsCtx } = ctx;
|
|
1377
671
|
const indexDb = dbCell.current;
|
|
1378
|
-
if (!indexDb)
|
|
1379
|
-
warnings
|
|
1380
|
-
|
|
1381
|
-
}
|
|
1382
|
-
const collect = config.improve?.stateGc?.collect === true;
|
|
672
|
+
if (!indexDb)
|
|
673
|
+
return { pending: 0, collected: 0, warnings: ["orphan state GC skipped: no index.db handle available"] };
|
|
674
|
+
const collect = ctx.config.improve?.stateGc?.collect === true;
|
|
1383
675
|
const now = Date.now();
|
|
1384
676
|
let pending = 0;
|
|
1385
677
|
let collected = 0;
|
|
1386
678
|
try {
|
|
1387
|
-
// R78 (tier1-0917): one query for every live item_ref, shared by both tables' sweeps
|
|
1388
|
-
// below — replaces a `getEntryByRef` round trip per pending row. Inside
|
|
1389
|
-
// the try so a schema mismatch (e.g. a DB version upgrade that dropped
|
|
1390
|
-
// `entries`) degrades to the "orphan state GC failed" warning below
|
|
1391
|
-
// instead of escaping this pass and failing the whole maintenance run.
|
|
1392
679
|
const liveRefs = getLiveRefSnapshot(indexDb);
|
|
680
|
+
const sweep = (table) => {
|
|
681
|
+
const toStamp = [];
|
|
682
|
+
const toClear = [];
|
|
683
|
+
for (const row of table.rows) {
|
|
684
|
+
const live = isStateRefLive(liveRefs, row.asset_ref);
|
|
685
|
+
if (!live && row.missing_since == null)
|
|
686
|
+
toStamp.push(row.asset_ref);
|
|
687
|
+
else if (live && row.missing_since != null)
|
|
688
|
+
toClear.push(row.asset_ref);
|
|
689
|
+
}
|
|
690
|
+
if (toStamp.length > 0)
|
|
691
|
+
table.stamp(toStamp, now);
|
|
692
|
+
if (toClear.length > 0)
|
|
693
|
+
table.clear(toClear);
|
|
694
|
+
const removed = collect ? table.deleteOlderThan(now - STATE_GC_GRACE_MS) : 0;
|
|
695
|
+
return { pending: table.countPending(), collected: removed };
|
|
696
|
+
};
|
|
1393
697
|
withStateDb((stateDb) => {
|
|
1394
|
-
|
|
1395
|
-
|
|
1396
|
-
|
|
1397
|
-
|
|
1398
|
-
|
|
1399
|
-
|
|
1400
|
-
|
|
1401
|
-
|
|
1402
|
-
|
|
1403
|
-
|
|
1404
|
-
|
|
1405
|
-
|
|
1406
|
-
|
|
1407
|
-
|
|
1408
|
-
|
|
1409
|
-
|
|
1410
|
-
|
|
1411
|
-
|
|
1412
|
-
|
|
1413
|
-
|
|
1414
|
-
// Table-name-shaped keys ("salience"/"outcome", not the guarded
|
|
1415
|
-
// "asset_salience"/"asset_outcome" table names) — this file sits
|
|
1416
|
-
// outside src/storage/repositories/**, where the state-table-sql
|
|
1417
|
-
// lint rule (#672) forbids naming those tables even in a log string
|
|
1418
|
-
// or an object key, not just in raw SQL.
|
|
1419
|
-
const byTable = { salience: salienceResult, outcome: outcomeResult };
|
|
1420
|
-
pending = salienceResult.pending + outcomeResult.pending;
|
|
1421
|
-
collected = salienceResult.collected + outcomeResult.collected;
|
|
698
|
+
// Keys avoid the state table names: the state-table-sql lint rule (#672)
|
|
699
|
+
// forbids them outside the repositories.
|
|
700
|
+
const byTable = {
|
|
701
|
+
salience: sweep({
|
|
702
|
+
rows: listAssetSalienceMissingState(stateDb),
|
|
703
|
+
stamp: (refs, ts) => stampAssetSalienceMissing(stateDb, refs, ts),
|
|
704
|
+
clear: (refs) => clearAssetSalienceMissing(stateDb, refs),
|
|
705
|
+
deleteOlderThan: (cutoff) => deleteAssetSalienceMissingBefore(stateDb, cutoff),
|
|
706
|
+
countPending: () => countAssetSalienceMissing(stateDb),
|
|
707
|
+
}),
|
|
708
|
+
outcome: sweep({
|
|
709
|
+
rows: listAssetOutcomeMissingState(stateDb),
|
|
710
|
+
stamp: (refs, ts) => stampAssetOutcomeMissing(stateDb, refs, ts),
|
|
711
|
+
clear: (refs) => clearAssetOutcomeMissing(stateDb, refs),
|
|
712
|
+
deleteOlderThan: (cutoff) => deleteAssetOutcomeMissingBefore(stateDb, cutoff),
|
|
713
|
+
countPending: () => countAssetOutcomeMissing(stateDb),
|
|
714
|
+
}),
|
|
715
|
+
};
|
|
716
|
+
pending = byTable.salience.pending + byTable.outcome.pending;
|
|
717
|
+
collected = byTable.salience.collected + byTable.outcome.collected;
|
|
1422
718
|
if (pending > 0 || collected > 0) {
|
|
1423
719
|
info(`[improve] orphan state GC: ${pending} pending, ${collected} collected ` +
|
|
1424
|
-
`(salience ${
|
|
1425
|
-
`outcome ${
|
|
1426
|
-
|
|
1427
|
-
// snapshot (`pending`) plus this run's deletions (`collected`);
|
|
1428
|
-
// emitted only when there is something to report (mirrors the
|
|
1429
|
-
// rekey script's no-op-stays-silent precedent) so a perpetually
|
|
1430
|
-
// clean stash never accumulates events.
|
|
1431
|
-
appendEvent({
|
|
1432
|
-
eventType: "asset_state_gc",
|
|
1433
|
-
ref: "asset_state/_gc",
|
|
1434
|
-
metadata: { pending, collected, byTable },
|
|
1435
|
-
}, eventsCtx);
|
|
720
|
+
`(salience ${byTable.salience.pending}/${byTable.salience.collected}, ` +
|
|
721
|
+
`outcome ${byTable.outcome.pending}/${byTable.outcome.collected})`);
|
|
722
|
+
appendEvent({ eventType: "asset_state_gc", ref: "asset_state/_gc", metadata: { pending, collected, byTable } }, eventsCtx);
|
|
1436
723
|
}
|
|
1437
724
|
}, { path: eventsCtx?.dbPath, borrowed: eventsCtx?.db });
|
|
1438
725
|
}
|
|
1439
726
|
catch (err) {
|
|
1440
|
-
warnings
|
|
727
|
+
return { pending, collected, warnings: [`orphan state GC failed: ${errMessage(err)}`] };
|
|
1441
728
|
}
|
|
1442
|
-
return { pending, collected, warnings };
|
|
729
|
+
return { pending, collected, warnings: [] };
|
|
1443
730
|
}
|