akm-cli 0.9.17-alpha.2 → 0.9.17-alpha.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +756 -0
- package/dist/akm +94 -196
- package/dist/cli/shared.js +6 -2
- package/dist/cli.js +22 -9
- package/dist/commands/agent/agent-dispatch.js +1 -1
- package/dist/commands/command/command-execution.js +24 -62
- package/dist/commands/feedback-cli.js +0 -1
- package/dist/commands/health/accept-rate.js +2 -2
- package/dist/commands/health/checks.js +30 -75
- package/dist/commands/health/config-skew.js +38 -0
- package/dist/commands/health/egress.js +54 -0
- package/dist/commands/health/html-report.js +0 -38
- package/dist/commands/health/improve-metrics.js +123 -562
- package/dist/commands/health/plugin-staleness.js +53 -3
- package/dist/commands/health/renderers.js +12 -4
- package/dist/commands/health/report-view-model.js +11 -106
- package/dist/commands/health/types-improve.js +4 -19
- package/dist/commands/health/windows.js +64 -73
- package/dist/commands/health.js +122 -143
- package/dist/commands/improve/consolidate/chunking.js +25 -100
- package/dist/commands/improve/consolidate/sanitize.js +54 -149
- package/dist/commands/improve/consolidate.js +538 -1075
- package/dist/commands/improve/content-hash.js +16 -24
- package/dist/commands/improve/distill/content-repair.js +18 -100
- package/dist/commands/improve/distill-guards.js +20 -81
- package/dist/commands/improve/distill-promotion-policy.js +23 -243
- package/dist/commands/improve/distill.js +608 -1075
- package/dist/commands/improve/eligibility.js +126 -400
- package/dist/commands/improve/execution.js +3 -5
- package/dist/commands/improve/extract.js +487 -1046
- package/dist/commands/improve/feedback-valence.js +0 -25
- package/dist/commands/improve/improve-cli.js +29 -166
- package/dist/commands/improve/improve-result-file.js +10 -66
- package/dist/commands/improve/improve-strategies.js +12 -7
- package/dist/commands/improve/improve-usage-report.js +18 -64
- package/dist/commands/improve/improve.js +443 -1063
- package/dist/commands/improve/ledger.js +114 -0
- package/dist/commands/improve/locks.js +2 -8
- package/dist/commands/improve/loop-stages.js +459 -1172
- package/dist/commands/improve/memory/derived-ref.js +12 -77
- package/dist/commands/improve/memory/memory-belief.js +14 -118
- package/dist/commands/improve/memory/memory-improve.js +4 -3
- package/dist/commands/improve/outcome-loop.js +28 -156
- package/dist/commands/improve/planner.js +5 -10
- package/dist/commands/improve/preparation.js +851 -2339
- package/dist/commands/improve/proactive-maintenance.js +34 -101
- package/dist/commands/improve/reflect-noise.js +104 -280
- package/dist/commands/improve/reflect.js +621 -1367
- package/dist/commands/improve/salience.js +46 -232
- package/dist/commands/improve/session-asset.js +19 -100
- package/dist/commands/improve/stage.js +323 -0
- package/dist/commands/proposal/drain.js +251 -644
- package/dist/commands/proposal/proposal-cli.js +3 -18
- package/dist/commands/proposal/proposal-types.js +20 -41
- package/dist/commands/proposal/proposal.js +1 -2
- package/dist/commands/proposal/propose.js +134 -160
- package/dist/commands/proposal/repository.js +502 -1487
- package/dist/commands/proposal/validators/proposal-quality-validators.js +71 -174
- package/dist/commands/proposal/validators/proposal-validators.js +1 -1
- package/dist/commands/proposal/validators/proposals.js +13 -89
- package/dist/commands/read/curate.js +63 -413
- package/dist/commands/read/search-cli.js +16 -33
- package/dist/commands/read/search.js +17 -23
- package/dist/commands/read/show.js +2 -13
- package/dist/commands/sources/bundle-cli.js +25 -2
- package/dist/commands/sources/bundle-config-ops.js +7 -0
- package/dist/commands/sources/dangerous-env-audit.js +1 -2
- package/dist/commands/sources/info.js +2 -11
- package/dist/commands/sources/installed-stashes.js +197 -746
- package/dist/commands/sources/schema-repair.js +98 -129
- package/dist/commands/sources/source-add.js +62 -12
- package/dist/commands/sources/stash-cli.js +1 -1
- package/dist/commands/tasks/explain.js +10 -13
- package/dist/commands/tasks/tasks-cli.js +9 -8
- package/dist/commands/tasks/tasks.js +326 -930
- package/dist/commands/tasks/validate.js +42 -21
- package/dist/commands/workflow/plan.js +22 -29
- package/dist/commands/workflow-cli.js +4 -4
- package/dist/core/adapter/adapters/akm-adapter.js +0 -1
- package/dist/core/adapter/adapters/akm-lint.js +2 -3
- package/dist/core/adapter/adapters/akm-metadata.js +11 -12
- package/dist/core/adapter/adapters/akm-workflow-adapter.js +1 -1
- package/dist/core/adapter/execution-source.js +17 -29
- package/dist/core/asset/resolve-ref.js +1 -1
- package/dist/core/bundle-id.js +42 -5
- package/dist/core/bundle-rename.js +291 -0
- package/dist/core/config/config-io.js +1 -2
- package/dist/core/config/config-schema.js +1 -33
- package/dist/core/config/config-walker.js +1 -1
- package/dist/core/config/config.js +163 -68
- package/dist/core/config/legacy-source-shape-shim.js +38 -9
- package/dist/core/config/schema/embedding.js +20 -5
- package/dist/core/config/schema/engines.js +5 -0
- package/dist/core/config/schema/execution.js +1 -1
- package/dist/core/config/schema/experimental.js +1 -1
- package/dist/core/config/schema/improve-processes.js +21 -95
- package/dist/core/config/schema/improve.js +4 -42
- package/dist/core/config/schema/scheduler.js +12 -12
- package/dist/core/config/schema/search.js +6 -22
- package/dist/core/env-secret-ref.js +0 -1
- package/dist/core/errors.js +8 -9
- package/dist/core/file-lock.js +76 -173
- package/dist/core/logs-db.js +2 -2
- package/dist/core/paths.js +0 -27
- package/dist/core/redaction.js +109 -2
- package/dist/core/run-lock.js +2 -5
- package/dist/core/spawn-env.js +1 -1
- package/dist/core/state/migrations.js +108 -61
- package/dist/core/state-db-scope.js +2 -4
- package/dist/core/state-db.js +126 -692
- package/dist/core/type-presentation.js +1 -9
- package/dist/core/write-source.js +293 -1012
- package/dist/execution/input-contract.js +1 -1
- package/dist/execution/resolved-request.js +135 -689
- package/dist/execution/source.js +63 -257
- package/dist/execution/target-ref.js +1 -1
- package/dist/indexer/bundle-identity-guard.js +2 -2
- package/dist/indexer/db/graph-db.js +106 -46
- package/dist/indexer/ensure-index.js +44 -85
- package/dist/indexer/graph/graph-extraction.js +340 -562
- package/dist/indexer/graph/graph-related.js +130 -0
- package/dist/indexer/index-rebuild-lock.js +3 -11
- package/dist/indexer/index-writer-lock.js +8 -17
- package/dist/indexer/index-written-assets.js +139 -151
- package/dist/indexer/indexer.js +524 -846
- package/dist/indexer/materialize-embeddings.js +60 -397
- package/dist/indexer/passes/memory-inference.js +81 -90
- package/dist/indexer/passes/metadata.js +132 -200
- package/dist/indexer/read-preflight.js +0 -7
- package/dist/indexer/scan/doc-to-entry.js +1 -3
- package/dist/indexer/scan/drain-dir.js +1 -1
- package/dist/indexer/search/db-search.js +181 -590
- package/dist/indexer/search/fts-query.js +30 -41
- package/dist/indexer/search/ranking.js +28 -154
- package/dist/indexer/search/search-attribution.js +12 -32
- package/dist/indexer/search/search-fields.js +11 -15
- package/dist/indexer/search/search-hit-enrichers.js +54 -85
- package/dist/indexer/search/search-source.js +1 -4
- package/dist/indexer/usage/usage-events.js +2 -7
- package/dist/integrations/agent/engine-fallback.js +23 -40
- package/dist/integrations/agent/engine-resolution.js +93 -183
- package/dist/integrations/agent/execution.js +507 -0
- package/dist/integrations/agent/model-map.js +28 -156
- package/dist/integrations/agent/request-lowering.js +66 -141
- package/dist/integrations/agent/runner-dispatch.js +143 -321
- package/dist/integrations/agent/runner.js +54 -14
- package/dist/integrations/lockfile.js +53 -101
- package/dist/llm/embedders/deterministic.js +2 -3
- package/dist/llm/embedders/profile.js +71 -0
- package/dist/llm/embedders/remote.js +10 -15
- package/dist/llm/graph-extract.js +3 -12
- package/dist/llm/index-passes.js +3 -5
- package/dist/llm/memory-infer.js +1 -2
- package/dist/llm/metadata-enhance.js +1 -2
- package/dist/llm/structured-call.js +5 -24
- package/dist/output/generic-render.js +23 -11
- package/dist/output/html-render.js +13 -10
- package/dist/output/render-registry.js +3 -32
- package/dist/output/shapes/helpers.js +2 -34
- package/dist/output/shapes/passthrough.js +1 -9
- package/dist/{indexer/search/ranking-types.js → output/text/bundle-rename.js} +4 -1
- package/dist/output/text/command-format.js +60 -23
- package/dist/output/text/helpers.js +1 -1
- package/dist/output/text/migrate.js +5 -14
- package/dist/output/text/proposal-format.js +1 -2
- package/dist/output/text/workflow-format.js +0 -32
- package/dist/output/text.js +2 -0
- package/dist/registry/factory.js +4 -19
- package/dist/registry/network.js +66 -220
- package/dist/registry/providers/index.js +0 -2
- package/dist/registry/providers/skills-sh.js +3 -14
- package/dist/registry/providers/static-index.js +24 -26
- package/dist/registry/resolve.js +55 -131
- package/dist/scripts/akm-migrate-node.js +43937 -93313
- package/dist/scripts/akm-migrate.js +43697 -93071
- package/dist/setup/registry-stash-loader.js +4 -13
- package/dist/setup/semantic-assets.js +3 -44
- package/dist/setup/setup.js +1 -1
- package/dist/setup/steps/tasks.js +25 -15
- package/dist/sources/provider-factory.js +17 -18
- package/dist/sources/providers/filesystem.js +2 -3
- package/dist/sources/providers/git-install.js +7 -1
- package/dist/sources/providers/git-provider.js +0 -3
- package/dist/sources/providers/git-stash.js +0 -17
- package/dist/sources/providers/npm.js +2 -4
- package/dist/sources/providers/provider-utils.js +5 -10
- package/dist/sources/providers/website.js +0 -2
- package/dist/sources/snapshot-fetchers/website-ingest.js +1 -1
- package/dist/sources/website-url.js +2 -2
- package/dist/storage/database.js +9 -35
- package/dist/storage/repositories/improve-ledger-repository.js +168 -0
- package/dist/storage/repositories/index-connection.js +34 -70
- package/dist/storage/repositories/index-entries-repository.js +69 -111
- package/dist/storage/repositories/index-entry-mapper.js +1 -2
- package/dist/storage/repositories/index-entry-schema.js +83 -269
- package/dist/storage/repositories/index-fts-repository.js +86 -256
- package/dist/storage/repositories/index-llm-cache-repository.js +17 -0
- package/dist/storage/repositories/index-meta-repository.js +6 -4
- package/dist/storage/repositories/index-schema.js +192 -220
- package/dist/storage/repositories/index-utility-repository.js +8 -29
- package/dist/storage/repositories/index-vec-repository.js +133 -414
- package/dist/storage/repositories/outcome-repository.js +2 -1
- package/dist/storage/repositories/proposals-repository.js +35 -0
- package/dist/storage/repositories/registry-index-cache-repository.js +100 -0
- package/dist/storage/repositories/task-history-repository.js +26 -4
- package/dist/storage/repositories/workflow-runs-repository.js +53 -244
- package/dist/storage/sqlite-migrations.js +136 -0
- package/dist/storage/sqlite-pragmas.js +11 -9
- package/dist/storage/sqlite-transaction.js +170 -0
- package/dist/storage/state-db-integrity.js +34 -27
- package/dist/tasks/activation-config.js +134 -62
- package/dist/tasks/backends/cron.js +129 -277
- package/dist/tasks/backends/exec-utils.js +2 -5
- package/dist/tasks/backends/launchd.js +125 -745
- package/dist/tasks/backends/schtasks.js +101 -620
- package/dist/tasks/prepare/prepare-support.js +5 -15
- package/dist/tasks/prepare/prepare.js +0 -2
- package/dist/tasks/resolve-akm-bin.js +20 -79
- package/dist/tasks/run/attempt-lifecycle.js +0 -1
- package/dist/tasks/scheduler-binding.js +18 -238
- package/dist/tasks/scheduler-invocation.js +52 -52
- package/dist/tasks/scheduler-lock.js +53 -0
- package/dist/tasks/scheduler-sync.js +363 -679
- package/dist/tasks/source/parse-task-source.js +160 -10
- package/dist/tasks/source/task-source-v3-frozen.js +3 -4
- package/dist/tasks/source/task-to-v4.js +2 -2
- package/dist/workflows/authoring/authoring.js +3 -12
- package/dist/workflows/compile.js +211 -0
- package/dist/workflows/concurrency-policy.js +13 -74
- package/dist/workflows/exec/child-invocation.js +3 -17
- package/dist/workflows/exec/child-workflow.js +32 -141
- package/dist/workflows/exec/dispatch-redaction.js +13 -53
- package/dist/workflows/exec/environment.js +98 -0
- package/dist/workflows/exec/exec-unit.js +33 -140
- package/dist/workflows/exec/frozen-judge.js +7 -59
- package/dist/workflows/exec/native-executor.js +82 -341
- package/dist/workflows/exec/param-secrets.js +29 -47
- package/dist/workflows/exec/run-workflow.js +154 -387
- package/dist/workflows/exec/scheduler.js +9 -36
- package/dist/workflows/exec/step-work.js +127 -430
- package/dist/workflows/exec/unit-dispatch.js +11 -63
- package/dist/workflows/exec/unit-writer.js +8 -52
- package/dist/workflows/exec/worktree.js +39 -273
- package/dist/workflows/freeze/child-output-references.js +4 -15
- package/dist/workflows/freeze/environment.js +99 -92
- package/dist/workflows/freeze/freeze.js +172 -0
- package/dist/workflows/freeze/step-values.js +19 -21
- package/dist/workflows/freeze/targets/child-workflow.js +23 -92
- package/dist/workflows/freeze/targets/command.js +10 -33
- package/dist/workflows/freeze/targets/script.js +5 -12
- package/dist/workflows/freeze/targets/shell.js +3 -6
- package/dist/workflows/freeze/targets/task.js +25 -80
- package/dist/workflows/freeze/task-bindings.js +20 -67
- package/dist/workflows/{source-ir/github-yaml.js → github-yaml.js} +88 -206
- package/dist/workflows/ir/params.js +6 -51
- package/dist/workflows/ir/plan-hash.js +2 -34
- package/dist/workflows/parser.js +140 -43
- package/dist/{commands/improve/consolidate/types.js → workflows/plan.js} +2 -1
- package/dist/workflows/renderer.js +36 -69
- package/dist/workflows/resource-limits.js +12 -120
- package/dist/workflows/runtime/agent-identity.js +8 -40
- package/dist/workflows/runtime/run-outputs.js +3 -6
- package/dist/workflows/runtime/run-plan.js +316 -0
- package/dist/workflows/runtime/runs.js +48 -200
- package/dist/workflows/runtime/workflow-asset-loader.js +24 -57
- package/dist/workflows/{source-ir/semantics.js → source-semantics.js} +16 -20
- package/dist/workflows/validate-summary.js +2 -7
- package/docs/integration/bundling-akm.md +49 -42
- package/docs/migration/README.md +1 -0
- package/docs/migration/release-notes/0.9.17.md +41 -0
- package/docs/migration/v0.9.1-to-v0.9.2.md +19 -7
- package/docs/reference/cli.md +182 -125
- package/docs/reference/configuration.md +49 -56
- package/docs/reference/data-and-telemetry.md +19 -20
- package/docs/reference/tasks.md +86 -38
- package/docs/reference/workflow-schema.md +14 -18
- package/docs/reference/workflows.md +6 -9
- package/package.json +1 -1
- package/schemas/akm-config.json +87 -406
- package/dist/commands/health/advisories.js +0 -150
- package/dist/commands/health/metrics.js +0 -329
- package/dist/commands/health/surfaces.js +0 -102
- package/dist/commands/improve/anti-collapse.js +0 -83
- package/dist/commands/improve/collapse-detector.js +0 -432
- package/dist/commands/improve/consolidate/eligibility.js +0 -48
- package/dist/commands/improve/consolidate/merge.js +0 -146
- package/dist/commands/improve/distill/promote-memory.js +0 -329
- package/dist/commands/improve/distill/quality-gate.js +0 -500
- package/dist/commands/improve/memory/memory-contradiction-detect.js +0 -291
- package/dist/commands/improve/proposal-envelope.js +0 -31
- package/dist/commands/improve/run-context.js +0 -123
- package/dist/commands/improve/shared.js +0 -21
- package/dist/commands/improve/source-identity.js +0 -28
- package/dist/commands/improve/triage.js +0 -96
- package/dist/commands/proposal/drain-policies.js +0 -151
- package/dist/commands/sources/update-transaction.js +0 -220
- package/dist/core/action-contributors.js +0 -28
- package/dist/core/config/config-version-shim.js +0 -101
- package/dist/core/config/retired-experimental-keys-shim.js +0 -62
- package/dist/core/fs-txn.js +0 -405
- package/dist/core/lexical-score.js +0 -25
- package/dist/core/maintenance-barrier.js +0 -167
- package/dist/execution/executable-identity.js +0 -105
- package/dist/execution/guarded-source.js +0 -427
- package/dist/indexer/graph/graph-boost.js +0 -427
- package/dist/indexer/graph/graph-dedup.js +0 -95
- package/dist/indexer/search/name-match.js +0 -35
- package/dist/indexer/search/ranking-contributors.js +0 -515
- package/dist/indexer/walk/project-context.js +0 -192
- package/dist/integrations/agent/execution-cascade.js +0 -566
- package/dist/integrations/agent/execution-definitions.js +0 -202
- package/dist/integrations/agent/execution-lowering.js +0 -841
- package/dist/integrations/agent/execution-preparation.js +0 -98
- package/dist/integrations/agent/inline-execution.js +0 -74
- package/dist/registry/create-provider-registry.js +0 -29
- package/dist/registry/pinned-request-helper.js +0 -247
- package/dist/registry/pinned-transport.js +0 -717
- package/dist/sources/providers/index.js +0 -14
- package/dist/storage/engines/sqlite-migrations.js +0 -271
- package/dist/storage/repositories/canaries-repository.js +0 -107
- package/dist/storage/repositories/embedding-salvage-repository.js +0 -184
- package/dist/storage/repositories/registry-cache.js +0 -113
- package/dist/tasks/scheduler-sync-preview.js +0 -52
- package/dist/workflows/freeze/resolve-steps.js +0 -86
- package/dist/workflows/freeze/source-freeze.js +0 -64
- package/dist/workflows/ir/compile.js +0 -321
- package/dist/workflows/ir/environment-v4.js +0 -330
- package/dist/workflows/ir/freeze-v4.js +0 -153
- package/dist/workflows/ir/schema-v4.js +0 -745
- package/dist/workflows/ir/schema.js +0 -354
- package/dist/workflows/program/schema.js +0 -78
- package/dist/workflows/runtime/checkin.js +0 -57
- package/dist/workflows/runtime/plan-classifier.js +0 -196
- package/dist/workflows/runtime/unit-checkin.js +0 -45
- package/dist/workflows/runtime/unit-phases.js +0 -20
- package/dist/workflows/schema.js +0 -4
- package/dist/workflows/source-ir/compile.js +0 -200
- package/dist/workflows/source-ir/program.js +0 -50
- package/dist/workflows/source-ir/result.js +0 -26
- package/dist/workflows/source-ir/schema.js +0 -786
- package/dist/workflows/source-ir/triggers.js +0 -79
- package/dist/workflows/source-ir/uses.js +0 -40
- package/dist/workflows/validator.js +0 -60
|
@@ -7,10 +7,8 @@
|
|
|
7
7
|
* Walks the primary stash for `memory:` and `knowledge:` assets, asks the
|
|
8
8
|
* configured LLM to extract entities and relations from each one, and
|
|
9
9
|
* persists the result to stash-local SQLite graph tables keyed by stash root.
|
|
10
|
-
* The artifact
|
|
11
|
-
*
|
|
12
|
-
* inside the existing FTS5+boosts loop — there is NO second SearchHit
|
|
13
|
-
* scorer and no parallel ranking track.
|
|
10
|
+
* The artifact backs `akm show`'s `related` list and curate's support refs
|
|
11
|
+
* (`src/indexer/graph/graph-related.ts`); it plays no part in search ranking.
|
|
14
12
|
*
|
|
15
13
|
* Disabling — three preconditions must ALL hold for the pass to run:
|
|
16
14
|
* 1. An LLM profile must be configured (no provider = no extraction). When
|
|
@@ -24,7 +22,7 @@
|
|
|
24
22
|
* Set to `false` to skip just this pass while leaving other passes
|
|
25
23
|
* that share the same LLM profile enabled.
|
|
26
24
|
* Toggling any one off does NOT delete the existing persisted graph — the
|
|
27
|
-
* user keeps the
|
|
25
|
+
* user keeps the related links they already have, they just stop
|
|
28
26
|
* refreshing.
|
|
29
27
|
*
|
|
30
28
|
* Locked v1 contract:
|
|
@@ -47,39 +45,46 @@ import { concurrentMap } from "../../core/concurrent.js";
|
|
|
47
45
|
import { getIndexPassConfig, loadConfig, resolveBatchSize } from "../../core/config/config.js";
|
|
48
46
|
import { ConfigError, rethrowIfTestIsolationError } from "../../core/errors.js";
|
|
49
47
|
import { warn, warnVerbose } from "../../core/warn.js";
|
|
50
|
-
import {
|
|
48
|
+
import { assertRunnerCredentials } from "../../integrations/agent/runner-dispatch.js";
|
|
51
49
|
import { isProcessEnabled } from "../../llm/feature-gate.js";
|
|
52
50
|
import * as graphExtract from "../../llm/graph-extract.js";
|
|
53
51
|
import { resolveIndexPassExecution } from "../../llm/index-passes.js";
|
|
54
|
-
import { preflightStructuredLlmRunner } from "../../llm/structured-call.js";
|
|
55
52
|
import { computeBodyHash, getLlmCacheEntriesByRefs, upsertLlmCacheEntry, } from "../../storage/repositories/index-llm-cache-repository.js";
|
|
56
53
|
import { GRAPH_SCHEMA_VERSION } from "../../storage/repositories/index-schema.js";
|
|
57
|
-
import { acknowledgeExtractionQueueEntry, enqueueGraphExtraction, loadStoredGraphSnapshot, peekExtractionQueue, replaceStoredGraph, } from "../db/graph-db.js";
|
|
54
|
+
import { acknowledgeExtractionQueueEntry, enqueueGraphExtraction, loadStoredGraphMeta, loadStoredGraphSnapshot, peekExtractionQueue, replaceStoredGraph, } from "../db/graph-db.js";
|
|
58
55
|
import { walkMarkdownFiles } from "../walk/walker.js";
|
|
59
|
-
import { deduplicateGraph } from "./graph-dedup.js";
|
|
60
56
|
/** Schema version for the persisted artifact — bumps trigger a full rebuild. */
|
|
61
57
|
export const GRAPH_FILE_SCHEMA_VERSION = GRAPH_SCHEMA_VERSION;
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
|
|
58
|
+
/**
|
|
59
|
+
* The frozen execution a graph call runs under: the invocation's own runner
|
|
60
|
+
* when the caller passed one (it already passed its own gates), otherwise the
|
|
61
|
+
* configured `graph` pass's. Undefined when the feature gate closes the pass.
|
|
62
|
+
*/
|
|
63
|
+
function selectGraphExecution(holder, config) {
|
|
64
|
+
if (Object.hasOwn(holder, "llmRunner")) {
|
|
65
|
+
return {
|
|
66
|
+
execution: Object.freeze({ runner: holder.llmRunner ?? undefined, notices: Object.freeze([]) }),
|
|
67
|
+
featureConfig: { ...config, index: { ...config.index, graph: { ...config.index?.graph, enabled: true } } },
|
|
68
|
+
};
|
|
65
69
|
}
|
|
66
|
-
|
|
70
|
+
if (!isProcessEnabled("index", "graph_extraction", config))
|
|
71
|
+
return undefined;
|
|
72
|
+
return { execution: resolveIndexPassExecution("graph", config), featureConfig: config };
|
|
67
73
|
}
|
|
68
|
-
const EMPTY_QUALITY = {
|
|
69
|
-
consideredFiles: 0,
|
|
70
|
-
extractedFiles: 0,
|
|
71
|
-
entityCount: 0,
|
|
72
|
-
relationCount: 0,
|
|
73
|
-
extractionCoverage: 0,
|
|
74
|
-
density: 0,
|
|
75
|
-
};
|
|
76
74
|
const EMPTY_RESULT = {
|
|
77
75
|
considered: 0,
|
|
78
76
|
extracted: 0,
|
|
79
77
|
totalEntities: 0,
|
|
80
78
|
totalRelations: 0,
|
|
81
79
|
written: false,
|
|
82
|
-
quality: {
|
|
80
|
+
quality: {
|
|
81
|
+
consideredFiles: 0,
|
|
82
|
+
extractedFiles: 0,
|
|
83
|
+
entityCount: 0,
|
|
84
|
+
relationCount: 0,
|
|
85
|
+
extractionCoverage: 0,
|
|
86
|
+
density: 0,
|
|
87
|
+
},
|
|
83
88
|
telemetry: {
|
|
84
89
|
cacheHits: 0,
|
|
85
90
|
cacheMisses: 0,
|
|
@@ -89,22 +94,6 @@ const EMPTY_RESULT = {
|
|
|
89
94
|
},
|
|
90
95
|
warnings: [],
|
|
91
96
|
};
|
|
92
|
-
function roundMetric(value) {
|
|
93
|
-
return Number(value.toFixed(4));
|
|
94
|
-
}
|
|
95
|
-
function computeGraphQualityTelemetry(consideredFiles, extractedFiles, entityCount, relationCount) {
|
|
96
|
-
const extractionCoverage = consideredFiles > 0 ? extractedFiles / consideredFiles : 0;
|
|
97
|
-
const maxEdges = entityCount > 1 ? (entityCount * (entityCount - 1)) / 2 : 0;
|
|
98
|
-
const density = maxEdges > 0 ? relationCount / maxEdges : 0;
|
|
99
|
-
return {
|
|
100
|
-
consideredFiles,
|
|
101
|
-
extractedFiles,
|
|
102
|
-
entityCount,
|
|
103
|
-
relationCount,
|
|
104
|
-
extractionCoverage: roundMetric(extractionCoverage),
|
|
105
|
-
density: roundMetric(density),
|
|
106
|
-
};
|
|
107
|
-
}
|
|
108
97
|
export const DEFAULT_GRAPH_EXTRACTION_INCLUDE_TYPES = ["memory", "knowledge"];
|
|
109
98
|
/**
|
|
110
99
|
* Max number of lazy-extraction queue rows drained per pass (#624-P3). Bounds
|
|
@@ -245,8 +234,6 @@ function isFailedExtractionStatus(status) {
|
|
|
245
234
|
return status === "failed";
|
|
246
235
|
}
|
|
247
236
|
function loadGraphFile(stashRoot, db) {
|
|
248
|
-
if (!db)
|
|
249
|
-
return { files: [] };
|
|
250
237
|
const graph = loadStoredGraphSnapshot(stashRoot, db);
|
|
251
238
|
if (!graph)
|
|
252
239
|
return { files: [] };
|
|
@@ -272,87 +259,93 @@ function loadGraphFile(stashRoot, db) {
|
|
|
272
259
|
...(graph.telemetry ? { telemetry: graph.telemetry } : {}),
|
|
273
260
|
};
|
|
274
261
|
}
|
|
275
|
-
|
|
276
|
-
|
|
277
|
-
|
|
262
|
+
/**
|
|
263
|
+
* The stored graph after a run: each refreshed node replaces the stored node
|
|
264
|
+
* for its path, and every other stored node is kept as it was — files a
|
|
265
|
+
* scoped run (`candidatePaths`, `topN`) did not select, and files an aborted
|
|
266
|
+
* run never reached. With `keptPaths`, stored nodes outside it are dropped
|
|
267
|
+
* (the file left the eligible set); without it, nothing is dropped.
|
|
268
|
+
*/
|
|
269
|
+
function mergeGraphNodes(previousNodes, refreshedNodes, keptPaths) {
|
|
278
270
|
const refreshedByPath = new Map(refreshedNodes.map((node) => [node.path, node]));
|
|
279
271
|
const merged = [];
|
|
280
272
|
for (const node of previousNodes) {
|
|
281
|
-
|
|
282
|
-
|
|
283
|
-
|
|
273
|
+
const refreshed = refreshedByPath.get(node.path);
|
|
274
|
+
if (refreshed) {
|
|
275
|
+
merged.push(refreshed);
|
|
276
|
+
refreshedByPath.delete(node.path);
|
|
277
|
+
}
|
|
278
|
+
else if (!keptPaths || keptPaths.has(node.path)) {
|
|
279
|
+
merged.push(node);
|
|
280
|
+
}
|
|
284
281
|
}
|
|
285
|
-
|
|
286
|
-
merged.push(refreshedByPath.get(node.path) ?? node);
|
|
282
|
+
merged.push(...refreshedByPath.values());
|
|
287
283
|
return merged;
|
|
288
284
|
}
|
|
285
|
+
/** A previous node (validated by {@link loadGraphFile}) for this exact body, unless it failed. */
|
|
289
286
|
function reuseGraphNode(previousNodes, candidate, bodyHash) {
|
|
290
287
|
const node = previousNodes.get(candidate.absPath);
|
|
291
|
-
if (!node)
|
|
292
|
-
return undefined;
|
|
293
|
-
if (node.type !== candidate.type)
|
|
294
|
-
return undefined;
|
|
295
|
-
if (typeof node.bodyHash !== "string" || node.bodyHash.length === 0)
|
|
296
|
-
return undefined;
|
|
297
|
-
if (node.bodyHash !== bodyHash)
|
|
288
|
+
if (!node || node.type !== candidate.type || node.bodyHash !== bodyHash)
|
|
298
289
|
return undefined;
|
|
299
290
|
if (isFailedExtractionStatus(node.status))
|
|
300
291
|
return undefined;
|
|
301
|
-
const validated = validateGraphCacheShape({ entities: node.entities, relations: node.relations });
|
|
302
|
-
if (!validated)
|
|
303
|
-
return undefined;
|
|
304
292
|
return {
|
|
305
|
-
entities:
|
|
306
|
-
relations:
|
|
307
|
-
confidence:
|
|
293
|
+
entities: node.entities,
|
|
294
|
+
relations: node.relations,
|
|
295
|
+
confidence: node.confidence,
|
|
308
296
|
...(node.status ? { status: node.status } : {}),
|
|
309
297
|
...(node.reason ? { reason: node.reason } : {}),
|
|
310
298
|
};
|
|
311
299
|
}
|
|
300
|
+
/**
|
|
301
|
+
* A file is a cache hit only through `llm_enrichment_cache`, whose variant is
|
|
302
|
+
* the extractor id. A stored graph node is never reused here: the graph keeps
|
|
303
|
+
* nodes that older extractors wrote (files a run did not reach), and reusing
|
|
304
|
+
* one would record another extractor's output as this one's.
|
|
305
|
+
*/
|
|
312
306
|
function planEligibleGraphExtractions(args) {
|
|
313
|
-
const { eligible, db, reEnrich, cacheVariant
|
|
314
|
-
const
|
|
315
|
-
|
|
316
|
-
|
|
317
|
-
|
|
318
|
-
|
|
319
|
-
|
|
320
|
-
|
|
321
|
-
|
|
322
|
-
|
|
323
|
-
|
|
324
|
-
|
|
325
|
-
|
|
326
|
-
|
|
327
|
-
}
|
|
328
|
-
}
|
|
329
|
-
catch {
|
|
330
|
-
// Corrupt cache rows are immutable model plans for this pass.
|
|
307
|
+
const { eligible, db, reEnrich, cacheVariant } = args;
|
|
308
|
+
const cacheEntries = reEnrich
|
|
309
|
+
? new Map()
|
|
310
|
+
: getLlmCacheEntriesByRefs(db, eligible.map((candidate) => candidate.absPath), cacheVariant);
|
|
311
|
+
return eligible.map((candidate) => {
|
|
312
|
+
const bodyHash = computeBodyHash(candidate.body);
|
|
313
|
+
if (reEnrich)
|
|
314
|
+
return { kind: "model", candidate, bodyHash };
|
|
315
|
+
const entry = cacheEntries.get(candidate.absPath);
|
|
316
|
+
if (entry?.bodyHash === bodyHash) {
|
|
317
|
+
try {
|
|
318
|
+
const cached = validateGraphCacheShape(JSON.parse(entry.resultJson));
|
|
319
|
+
if (cached && !isFailedExtractionStatus(cached.status)) {
|
|
320
|
+
return { kind: "cache-hit", candidate, bodyHash, cached };
|
|
331
321
|
}
|
|
332
322
|
}
|
|
333
|
-
|
|
334
|
-
|
|
335
|
-
|
|
336
|
-
if (cached)
|
|
337
|
-
return { kind: "cache-hit", candidate, bodyHash, cached, persistCache: Boolean(db) };
|
|
323
|
+
catch {
|
|
324
|
+
// A corrupt cache row is a miss.
|
|
325
|
+
}
|
|
338
326
|
}
|
|
339
327
|
return { kind: "model", candidate, bodyHash };
|
|
340
328
|
});
|
|
341
329
|
}
|
|
342
|
-
function
|
|
330
|
+
function extractionRecord(candidate, bodyHash, shape) {
|
|
343
331
|
return {
|
|
344
|
-
absPath:
|
|
345
|
-
type:
|
|
346
|
-
bodyHash
|
|
347
|
-
entities:
|
|
348
|
-
relations:
|
|
349
|
-
...(
|
|
350
|
-
...(
|
|
351
|
-
...(
|
|
332
|
+
absPath: candidate.absPath,
|
|
333
|
+
type: candidate.type,
|
|
334
|
+
bodyHash,
|
|
335
|
+
entities: shape.entities,
|
|
336
|
+
relations: shape.relations,
|
|
337
|
+
...(shape.confidence !== undefined ? { confidence: shape.confidence } : {}),
|
|
338
|
+
...(shape.status ? { status: shape.status } : {}),
|
|
339
|
+
...(shape.reason ? { reason: shape.reason } : {}),
|
|
352
340
|
};
|
|
353
341
|
}
|
|
342
|
+
/**
|
|
343
|
+
* Run the planned extractions in chunks of `batchSize`: cache hits are taken
|
|
344
|
+
* as-is, and each chunk's model plans go to the provider in one
|
|
345
|
+
* `extractGraphFromBodies` call (a one-body call is the per-asset path).
|
|
346
|
+
*/
|
|
354
347
|
async function extractGraphBatches(args) {
|
|
355
|
-
const { plans, batchSize, signal, db, cacheVariant, telemetry, llmRunner,
|
|
348
|
+
const { plans, batchSize, signal, db, cacheVariant, telemetry, llmRunner, featureConfig, onFallback, abortState, batchState, runtimeTelemetry, onNotices, reportProgress, maxChunksPerAsset, } = args;
|
|
356
349
|
const results = new Array(plans.length).fill(undefined);
|
|
357
350
|
const chunkStarts = [];
|
|
358
351
|
for (let start = 0; start < plans.length; start += batchSize)
|
|
@@ -363,23 +356,18 @@ async function extractGraphBatches(args) {
|
|
|
363
356
|
return;
|
|
364
357
|
const chunk = plans.slice(start, start + batchSize);
|
|
365
358
|
const reportChunkProgress = () => {
|
|
366
|
-
for (
|
|
367
|
-
|
|
368
|
-
if (plan)
|
|
369
|
-
reportProgress(plan.candidate.absPath, results[start + j]);
|
|
370
|
-
}
|
|
359
|
+
for (const [j, plan] of chunk.entries())
|
|
360
|
+
reportProgress(plan.candidate.absPath, results[start + j]);
|
|
371
361
|
};
|
|
372
|
-
|
|
373
|
-
|
|
374
|
-
if (
|
|
362
|
+
const modelPlans = [];
|
|
363
|
+
for (const [offset, plan] of chunk.entries()) {
|
|
364
|
+
if (plan.kind === "model") {
|
|
365
|
+
modelPlans.push({ plan, offset });
|
|
375
366
|
continue;
|
|
376
|
-
telemetry.cacheHits += 1;
|
|
377
|
-
results[start + index] = graphRecordFromCachePlan(plan);
|
|
378
|
-
if (db && plan.persistCache && !isFailedExtractionStatus(plan.cached.status)) {
|
|
379
|
-
upsertLlmCacheEntry(db, plan.candidate.absPath, plan.bodyHash, JSON.stringify(plan.cached), cacheVariant);
|
|
380
367
|
}
|
|
368
|
+
telemetry.cacheHits += 1;
|
|
369
|
+
results[start + offset] = extractionRecord(plan.candidate, plan.bodyHash, plan.cached);
|
|
381
370
|
}
|
|
382
|
-
const modelPlans = chunk.filter((plan) => plan.kind === "model");
|
|
383
371
|
if (modelPlans.length === 0 || abortState.aborted) {
|
|
384
372
|
reportChunkProgress();
|
|
385
373
|
return;
|
|
@@ -387,11 +375,10 @@ async function extractGraphBatches(args) {
|
|
|
387
375
|
telemetry.cacheMisses += modelPlans.length;
|
|
388
376
|
let batchExtractions;
|
|
389
377
|
try {
|
|
390
|
-
batchExtractions = await graphExtract.extractGraphFromBodies(llmRunner, modelPlans.map((plan) => plan.candidate.body), signal, featureConfig, onFallback, {
|
|
378
|
+
batchExtractions = await graphExtract.extractGraphFromBodies(llmRunner, modelPlans.map(({ plan }) => plan.candidate.body), signal, featureConfig, onFallback, {
|
|
391
379
|
batchState,
|
|
392
380
|
telemetry: runtimeTelemetry,
|
|
393
381
|
onNotices,
|
|
394
|
-
...(lease ? { lease } : {}),
|
|
395
382
|
...(maxChunksPerAsset != null ? { maxChunksPerAsset } : {}),
|
|
396
383
|
});
|
|
397
384
|
}
|
|
@@ -402,14 +389,10 @@ async function extractGraphBatches(args) {
|
|
|
402
389
|
}
|
|
403
390
|
throw error;
|
|
404
391
|
}
|
|
405
|
-
let llmIndex = 0;
|
|
406
392
|
let dispatchHadResult = false;
|
|
407
393
|
let dispatchAllFailed = true;
|
|
408
|
-
for (
|
|
409
|
-
const
|
|
410
|
-
if (!plan || plan.kind !== "model")
|
|
411
|
-
continue;
|
|
412
|
-
const extraction = batchExtractions[llmIndex++];
|
|
394
|
+
for (const [i, { plan, offset }] of modelPlans.entries()) {
|
|
395
|
+
const extraction = batchExtractions[i];
|
|
413
396
|
if (!extraction)
|
|
414
397
|
continue;
|
|
415
398
|
const cacheShape = {
|
|
@@ -420,17 +403,11 @@ async function extractGraphBatches(args) {
|
|
|
420
403
|
...(extraction.reason ? { reason: extraction.reason } : {}),
|
|
421
404
|
};
|
|
422
405
|
dispatchHadResult = true;
|
|
423
|
-
if (!isFailedExtractionStatus(cacheShape.status))
|
|
406
|
+
if (!isFailedExtractionStatus(cacheShape.status)) {
|
|
424
407
|
dispatchAllFailed = false;
|
|
425
|
-
if (db && !isFailedExtractionStatus(cacheShape.status)) {
|
|
426
408
|
upsertLlmCacheEntry(db, plan.candidate.absPath, plan.bodyHash, JSON.stringify(cacheShape), cacheVariant);
|
|
427
409
|
}
|
|
428
|
-
results[start +
|
|
429
|
-
absPath: plan.candidate.absPath,
|
|
430
|
-
type: plan.candidate.type,
|
|
431
|
-
bodyHash: plan.bodyHash,
|
|
432
|
-
...cacheShape,
|
|
433
|
-
};
|
|
410
|
+
results[start + offset] = extractionRecord(plan.candidate, plan.bodyHash, cacheShape);
|
|
434
411
|
}
|
|
435
412
|
// One attempt per `extractGraphFromBodies` dispatch (this chunk's batch
|
|
436
413
|
// call), not one per file it covers — mirrors consolidate.ts's
|
|
@@ -455,101 +432,54 @@ function readCurrentGraphBodyHash(filePath) {
|
|
|
455
432
|
}
|
|
456
433
|
function planQueuedGraphExtractions(args) {
|
|
457
434
|
const { db, stashRoot, previousNodes, signal, reEnrich } = args;
|
|
458
|
-
if (!db)
|
|
459
|
-
return [];
|
|
460
435
|
return peekExtractionQueue(db, stashRoot, GRAPH_EXTRACTION_QUEUE_DRAIN_LIMIT).map((queued) => {
|
|
461
|
-
|
|
462
|
-
|
|
463
|
-
|
|
464
|
-
|
|
465
|
-
|
|
466
|
-
|
|
467
|
-
};
|
|
468
|
-
}
|
|
469
|
-
let raw;
|
|
470
|
-
try {
|
|
471
|
-
raw = fs.readFileSync(queued.filePath, "utf8");
|
|
472
|
-
}
|
|
473
|
-
catch {
|
|
474
|
-
return { kind: "discard", filePath: queued.filePath, queuedBodyHash: queued.bodyHash, priority: queued.priority };
|
|
475
|
-
}
|
|
476
|
-
const body = parseFrontmatter(raw).content.trim();
|
|
477
|
-
if (!body) {
|
|
478
|
-
return { kind: "discard", filePath: queued.filePath, queuedBodyHash: queued.bodyHash, priority: queued.priority };
|
|
479
|
-
}
|
|
480
|
-
const currentBodyHash = computeBodyHash(body);
|
|
436
|
+
const base = { filePath: queued.filePath, queuedBodyHash: queued.bodyHash, priority: queued.priority };
|
|
437
|
+
if (signal?.aborted)
|
|
438
|
+
return { kind: "deferred", ...base };
|
|
439
|
+
const currentBodyHash = readCurrentGraphBodyHash(queued.filePath);
|
|
440
|
+
if (!currentBodyHash)
|
|
441
|
+
return { kind: "discard", ...base };
|
|
481
442
|
const type = inferGraphTypeForPath(stashRoot, queued.filePath) ?? "memory";
|
|
482
|
-
|
|
483
|
-
|
|
484
|
-
kind: "hit",
|
|
485
|
-
filePath: queued.filePath,
|
|
486
|
-
queuedBodyHash: queued.bodyHash,
|
|
487
|
-
currentBodyHash,
|
|
488
|
-
priority: queued.priority,
|
|
489
|
-
};
|
|
490
|
-
}
|
|
491
|
-
return {
|
|
492
|
-
kind: "model",
|
|
493
|
-
filePath: queued.filePath,
|
|
494
|
-
queuedBodyHash: queued.bodyHash,
|
|
495
|
-
currentBodyHash,
|
|
496
|
-
priority: queued.priority,
|
|
497
|
-
};
|
|
443
|
+
const hit = !reEnrich && reuseGraphNode(previousNodes, { absPath: queued.filePath, type }, currentBodyHash);
|
|
444
|
+
return { kind: hit ? "hit" : "model", ...base, currentBodyHash };
|
|
498
445
|
});
|
|
499
446
|
}
|
|
500
447
|
async function executeQueuedGraphPlans(args) {
|
|
501
|
-
const { plans, db, stashRoot, featureConfig, signal, llmRunner,
|
|
502
|
-
if (!db)
|
|
503
|
-
return { graphChanged: false, acknowledgements: [] };
|
|
448
|
+
const { plans, db, stashRoot, featureConfig, signal, llmRunner, onNotices } = args;
|
|
504
449
|
let graphChanged = false;
|
|
505
450
|
const acknowledgements = [];
|
|
506
451
|
for (const plan of plans) {
|
|
507
452
|
if (signal?.aborted || plan.kind === "deferred")
|
|
508
453
|
break;
|
|
509
|
-
|
|
510
|
-
|
|
511
|
-
if (currentBodyHash) {
|
|
512
|
-
enqueueGraphExtraction(db, stashRoot, plan.filePath, currentBodyHash, plan.priority);
|
|
513
|
-
continue;
|
|
514
|
-
}
|
|
515
|
-
acknowledgements.push({ filePath: plan.filePath, queuedBodyHash: plan.queuedBodyHash });
|
|
516
|
-
continue;
|
|
517
|
-
}
|
|
518
|
-
if (plan.kind === "hit") {
|
|
519
|
-
const currentBodyHash = readCurrentGraphBodyHash(plan.filePath);
|
|
520
|
-
if (currentBodyHash !== plan.currentBodyHash) {
|
|
521
|
-
if (currentBodyHash)
|
|
522
|
-
enqueueGraphExtraction(db, stashRoot, plan.filePath, currentBodyHash, plan.priority);
|
|
523
|
-
continue;
|
|
524
|
-
}
|
|
525
|
-
acknowledgements.push({ filePath: plan.filePath, queuedBodyHash: plan.queuedBodyHash });
|
|
526
|
-
continue;
|
|
527
|
-
}
|
|
454
|
+
// The body this plan settled: none for a discard (the file was gone or empty).
|
|
455
|
+
let settledBodyHash;
|
|
528
456
|
if (plan.kind === "model") {
|
|
529
457
|
const outcome = await extractGraphForSingleFileRevision(db, stashRoot, plan.filePath, {
|
|
530
458
|
config: featureConfig,
|
|
531
459
|
signal,
|
|
532
460
|
llmRunner,
|
|
533
|
-
lease,
|
|
534
461
|
onNotices,
|
|
535
462
|
});
|
|
536
463
|
if (!outcome.written)
|
|
537
464
|
continue;
|
|
538
465
|
graphChanged = true;
|
|
539
|
-
|
|
540
|
-
|
|
541
|
-
|
|
542
|
-
|
|
543
|
-
|
|
544
|
-
|
|
466
|
+
settledBodyHash = outcome.bodyHash;
|
|
467
|
+
}
|
|
468
|
+
else if (plan.kind === "hit") {
|
|
469
|
+
settledBodyHash = plan.currentBodyHash;
|
|
470
|
+
}
|
|
471
|
+
// A body that changed since it was planned goes back on the queue.
|
|
472
|
+
const currentBodyHash = readCurrentGraphBodyHash(plan.filePath);
|
|
473
|
+
if (currentBodyHash !== settledBodyHash) {
|
|
474
|
+
if (currentBodyHash)
|
|
475
|
+
enqueueGraphExtraction(db, stashRoot, plan.filePath, currentBodyHash, plan.priority);
|
|
476
|
+
continue;
|
|
545
477
|
}
|
|
546
478
|
acknowledgements.push({ filePath: plan.filePath, queuedBodyHash: plan.queuedBodyHash });
|
|
547
479
|
}
|
|
548
480
|
return { graphChanged, acknowledgements };
|
|
549
481
|
}
|
|
550
482
|
function acknowledgeQueuedGraphPlans(db, stashRoot, execution) {
|
|
551
|
-
if (!db)
|
|
552
|
-
return;
|
|
553
483
|
for (const intent of execution.acknowledgements) {
|
|
554
484
|
acknowledgeExtractionQueueEntry(db, stashRoot, intent.filePath, intent.queuedBodyHash);
|
|
555
485
|
}
|
|
@@ -571,18 +501,16 @@ function acknowledgeQueuedGraphPlans(db, stashRoot, execution) {
|
|
|
571
501
|
* If any of the three is missing or `false`, this function short-circuits
|
|
572
502
|
* to an empty no-op result, leaving any existing persisted graph untouched.
|
|
573
503
|
*
|
|
574
|
-
*
|
|
575
|
-
*
|
|
576
|
-
*
|
|
577
|
-
* preserves existing behaviour, fully opt-in).
|
|
504
|
+
* Eligible files are chunked by the resolved batch size
|
|
505
|
+
* (`graphExtractionBatchSize`) and each chunk is one `extractGraphFromBodies`
|
|
506
|
+
* call; a batch size of 1 is one call per asset.
|
|
578
507
|
*/
|
|
579
508
|
export async function runGraphExtractionPass(ctx) {
|
|
580
509
|
const { config, sources, signal, db, reEnrich, onProgress, options = {} } = ctx;
|
|
581
|
-
|
|
582
|
-
//
|
|
583
|
-
|
|
584
|
-
|
|
585
|
-
if (!invocationOwnsRunner && !isProcessEnabled("index", "graph_extraction", config))
|
|
510
|
+
// Gate 1 — the feature gate (selected strategy's
|
|
511
|
+
// processes.graphExtraction.enabled, default enabled).
|
|
512
|
+
const selection = selectGraphExecution(ctx, config);
|
|
513
|
+
if (!selection)
|
|
586
514
|
return { ...EMPTY_RESULT };
|
|
587
515
|
const noticesByKey = new Map();
|
|
588
516
|
const onNotices = (notices) => {
|
|
@@ -595,9 +523,8 @@ export async function runGraphExtractionPass(ctx) {
|
|
|
595
523
|
});
|
|
596
524
|
// Gate 2 — per-pass opt-out (#208). Retain the whole frozen resolution so
|
|
597
525
|
// selection-time lowering notices cannot be separated from the runner.
|
|
598
|
-
|
|
599
|
-
|
|
600
|
-
const llmRunner = execution.runner;
|
|
526
|
+
onNotices(selection.execution.notices);
|
|
527
|
+
const llmRunner = selection.execution.runner;
|
|
601
528
|
if (!llmRunner) {
|
|
602
529
|
const reason = getIndexPassConfig(config.index, "graph")?.enabled === false
|
|
603
530
|
? "index.graph.enabled is false"
|
|
@@ -605,9 +532,7 @@ export async function runGraphExtractionPass(ctx) {
|
|
|
605
532
|
warnVerbose(`graph extraction: skipped because ${reason}.`);
|
|
606
533
|
return emptyResult();
|
|
607
534
|
}
|
|
608
|
-
const featureConfig =
|
|
609
|
-
? { ...config, index: { ...config.index, graph: { ...config.index?.graph, enabled: true } } }
|
|
610
|
-
: config;
|
|
535
|
+
const { featureConfig } = selection;
|
|
611
536
|
// The pass only writes to the primary (working) stash. Read-only caches
|
|
612
537
|
// (git, npm, website) are deliberately untouched — the graph artifact for
|
|
613
538
|
// those sources would be clobbered by the next sync().
|
|
@@ -616,13 +541,15 @@ export async function runGraphExtractionPass(ctx) {
|
|
|
616
541
|
warnVerbose("graph extraction: skipped because no primary stash source is available.");
|
|
617
542
|
return emptyResult();
|
|
618
543
|
}
|
|
544
|
+
if (!db) {
|
|
545
|
+
warn("graph extraction: no database handle available; skipping graph persistence.");
|
|
546
|
+
return emptyResult();
|
|
547
|
+
}
|
|
619
548
|
const includeTypes = options.includeTypes ?? getGraphExtractionIncludeTypes(config);
|
|
620
549
|
let previousGraph = loadGraphFile(primary.path, db);
|
|
621
550
|
const previousNodes = new Map(previousGraph.files.map((node) => [node.path, node]));
|
|
622
551
|
const batchSize = resolveBatchSize(options.batchSize ?? getIndexPassConfig(config.index, "graph")?.graphExtractionBatchSize, llmRunner.connection.contextLength);
|
|
623
552
|
const extractorId = getGraphExtractorId({ model: llmRunner.connection.model, batchSize, includeTypes });
|
|
624
|
-
const cacheVariant = extractorId;
|
|
625
|
-
const canReusePreviousGraph = previousGraph.telemetry?.extractorId === extractorId;
|
|
626
553
|
const queuePlans = planQueuedGraphExtractions({
|
|
627
554
|
db,
|
|
628
555
|
stashRoot: primary.path,
|
|
@@ -631,286 +558,193 @@ export async function runGraphExtractionPass(ctx) {
|
|
|
631
558
|
reEnrich,
|
|
632
559
|
});
|
|
633
560
|
const queuedPaths = new Set(queuePlans.map((plan) => plan.filePath));
|
|
634
|
-
|
|
635
|
-
//
|
|
636
|
-
//
|
|
637
|
-
//
|
|
638
|
-
//
|
|
639
|
-
//
|
|
640
|
-
|
|
641
|
-
|
|
561
|
+
const scan = collectEligibleFiles(primary.path, includeTypes);
|
|
562
|
+
// The stored nodes this run keeps without touching them: every eligible file
|
|
563
|
+
// (outside candidatePaths or topN, or never reached before an abort) and every
|
|
564
|
+
// file the queue handled. Only a node whose file left the eligible set — gone,
|
|
565
|
+
// emptied, inferred, or of a type no longer included — is dropped, and an
|
|
566
|
+
// incomplete scan drops nothing.
|
|
567
|
+
const keptPaths = scan.complete
|
|
568
|
+
? new Set([
|
|
569
|
+
...scan.files.map((file) => file.absPath),
|
|
570
|
+
...queuePlans.filter((plan) => plan.kind !== "discard").map((plan) => plan.filePath),
|
|
571
|
+
])
|
|
572
|
+
: undefined;
|
|
573
|
+
let eligible = scan.files.filter((candidate) => (!options.candidatePaths || options.candidatePaths.has(candidate.absPath)) && !queuedPaths.has(candidate.absPath));
|
|
574
|
+
// P2 (#624): when topN is set, rank the (already candidate-filtered)
|
|
575
|
+
// eligible set by utility_scores DESC and keep only the top-N. Unset issues
|
|
576
|
+
// no ranking query. Ranking composes WITH the candidatePaths filter:
|
|
577
|
+
// scoped-then-ranked-then-sliced.
|
|
578
|
+
if (options.topN != null && options.topN >= 0) {
|
|
579
|
+
eligible = rankCandidatesByUtility(db, eligible).slice(0, options.topN);
|
|
642
580
|
}
|
|
643
581
|
const considered = eligible.length;
|
|
644
|
-
const eligiblePlans = planEligibleGraphExtractions({
|
|
645
|
-
eligible,
|
|
646
|
-
db,
|
|
647
|
-
reEnrich,
|
|
648
|
-
cacheVariant,
|
|
649
|
-
previousNodes,
|
|
650
|
-
canReusePreviousGraph,
|
|
651
|
-
});
|
|
652
|
-
const queueNeedsModel = queuePlans.some((plan) => plan.kind === "model");
|
|
653
|
-
const eligibleNeedsModel = !signal?.aborted && eligiblePlans.some((plan) => plan.kind === "model");
|
|
582
|
+
const eligiblePlans = planEligibleGraphExtractions({ eligible, db, reEnrich, cacheVariant: extractorId });
|
|
654
583
|
if (signal?.aborted)
|
|
655
584
|
return emptyResult();
|
|
656
585
|
// Validate exactly once iff classification found real model work. Queue
|
|
657
586
|
// acknowledgements, cache writes, and graph replacement all happen after
|
|
658
587
|
// this boundary, so a missing credential cannot partially mutate a batch.
|
|
659
|
-
|
|
660
|
-
|
|
661
|
-
|
|
662
|
-
|
|
663
|
-
|
|
664
|
-
|
|
665
|
-
|
|
666
|
-
|
|
667
|
-
|
|
668
|
-
|
|
669
|
-
|
|
670
|
-
|
|
671
|
-
|
|
672
|
-
|
|
673
|
-
|
|
674
|
-
if (considered === 0) {
|
|
675
|
-
acknowledgeQueuedGraphPlans(db, primary.path, queueExecution);
|
|
676
|
-
const scoped = options.candidatePaths ? ` matching ${options.candidatePaths.size} candidate path(s)` : "";
|
|
677
|
-
warnVerbose(`graph extraction: skipped because no eligible files${scoped} were found under ${primary.path}. ` +
|
|
678
|
-
`includeTypes=${includeTypes.join(",")}`);
|
|
679
|
-
return emptyResult();
|
|
680
|
-
}
|
|
681
|
-
const nodes = [];
|
|
682
|
-
let totalEntities = 0;
|
|
683
|
-
let totalRelations = 0;
|
|
684
|
-
let processed = 0;
|
|
685
|
-
let extracted = 0;
|
|
686
|
-
onProgress?.({ processed, total: considered, extracted, totalEntities, totalRelations });
|
|
687
|
-
const reportProgress = (currentPath, result) => {
|
|
688
|
-
processed += 1;
|
|
689
|
-
if (result) {
|
|
690
|
-
if (result.entities.length > 0)
|
|
691
|
-
extracted += 1;
|
|
692
|
-
totalEntities += result.entities.length;
|
|
693
|
-
totalRelations += result.relations.length;
|
|
694
|
-
}
|
|
695
|
-
onProgress?.({
|
|
696
|
-
processed,
|
|
697
|
-
total: considered,
|
|
698
|
-
extracted,
|
|
699
|
-
totalEntities,
|
|
700
|
-
totalRelations,
|
|
701
|
-
currentPath,
|
|
702
|
-
});
|
|
703
|
-
};
|
|
704
|
-
const extractionRunId = crypto.randomUUID();
|
|
705
|
-
const telemetry = {
|
|
706
|
-
extractorId,
|
|
707
|
-
extractionRunId,
|
|
708
|
-
model: llmRunner.connection.model,
|
|
709
|
-
promptVersion: graphExtract.GRAPH_EXTRACT_PROMPT_VERSION,
|
|
710
|
-
batchSize,
|
|
711
|
-
cacheHits: 0,
|
|
712
|
-
cacheMisses: 0,
|
|
713
|
-
truncationCount: 0,
|
|
714
|
-
failureCount: 0,
|
|
715
|
-
htmlErrorCount: 0,
|
|
716
|
-
retryAttempts: 0,
|
|
717
|
-
nonArrayBatchFailures: 0,
|
|
718
|
-
};
|
|
719
|
-
const runtimeTelemetry = {
|
|
720
|
-
truncationCount: 0,
|
|
721
|
-
failureCount: 0,
|
|
722
|
-
htmlErrorCount: 0,
|
|
723
|
-
retryAttempts: 0,
|
|
724
|
-
filteredGenericEntities: 0,
|
|
725
|
-
filteredInvalidRelations: 0,
|
|
726
|
-
filteredLowConfidenceRelations: 0,
|
|
727
|
-
contextBatchRetries: 0,
|
|
728
|
-
nonArrayBatchFailures: 0,
|
|
729
|
-
};
|
|
730
|
-
const batchState = {
|
|
731
|
-
batchingDisabled: false,
|
|
732
|
-
nonArrayBatchFailures: 0,
|
|
733
|
-
};
|
|
734
|
-
const abortState = { attempts: 0, failures: 0, aborted: false };
|
|
735
|
-
warnVerbose(`graph extraction: starting for ${considered} eligible file(s) under ${primary.path}; ` +
|
|
736
|
-
`includeTypes=${includeTypes.join(",")}, batchSize=${batchSize}, concurrency=${llmRunner.connection.concurrency ?? 1}, ` +
|
|
737
|
-
`reEnrich=${reEnrich === true}, candidateScoped=${options.candidatePaths ? "true" : "false"}.`);
|
|
738
|
-
const onFallback = (evt) => {
|
|
739
|
-
warn(`[akm] LLM fallback for ${evt.feature}: ${evt.reason}`);
|
|
740
|
-
};
|
|
741
|
-
let extractionResults;
|
|
742
|
-
let configFailure;
|
|
743
|
-
if (batchSize <= 1) {
|
|
744
|
-
// ── Original per-asset path (with incremental cache) ─────────────────
|
|
745
|
-
extractionResults = await concurrentMap(eligiblePlans, async (plan) => {
|
|
746
|
-
const { candidate, bodyHash } = plan;
|
|
747
|
-
if (signal?.aborted) {
|
|
748
|
-
reportProgress(candidate.absPath, undefined);
|
|
749
|
-
return undefined;
|
|
750
|
-
}
|
|
751
|
-
let cached;
|
|
752
|
-
if (plan.kind === "cache-hit") {
|
|
753
|
-
telemetry.cacheHits += 1;
|
|
754
|
-
cached = plan.cached;
|
|
755
|
-
if (db && plan.persistCache && !isFailedExtractionStatus(cached.status)) {
|
|
756
|
-
upsertLlmCacheEntry(db, candidate.absPath, bodyHash, JSON.stringify(cached), cacheVariant);
|
|
757
|
-
}
|
|
758
|
-
}
|
|
759
|
-
else {
|
|
760
|
-
if (abortState.aborted) {
|
|
761
|
-
reportProgress(candidate.absPath, undefined);
|
|
762
|
-
return undefined;
|
|
763
|
-
}
|
|
764
|
-
telemetry.cacheMisses += 1;
|
|
765
|
-
let extraction;
|
|
766
|
-
try {
|
|
767
|
-
extraction = await graphExtract.extractGraphFromBody(llmRunner, candidate.body, signal, featureConfig, onFallback, {
|
|
768
|
-
batchState,
|
|
769
|
-
telemetry: runtimeTelemetry,
|
|
770
|
-
onNotices,
|
|
771
|
-
...(dispatchLease ? { lease: dispatchLease } : {}),
|
|
772
|
-
...(options.maxChunksPerAsset != null ? { maxChunksPerAsset: options.maxChunksPerAsset } : {}),
|
|
773
|
-
});
|
|
774
|
-
}
|
|
775
|
-
catch (err) {
|
|
776
|
-
if (err instanceof ConfigError) {
|
|
777
|
-
configFailure ??= err;
|
|
778
|
-
return undefined;
|
|
779
|
-
}
|
|
780
|
-
throw err;
|
|
781
|
-
}
|
|
782
|
-
cached = {
|
|
783
|
-
entities: extraction.entities,
|
|
784
|
-
relations: extraction.relations,
|
|
785
|
-
...(extraction.confidence !== undefined ? { confidence: extraction.confidence } : {}),
|
|
786
|
-
...(extraction.status ? { status: extraction.status } : {}),
|
|
787
|
-
...(extraction.reason ? { reason: extraction.reason } : {}),
|
|
788
|
-
};
|
|
789
|
-
recordGraphExtractionAttempt(abortState, isFailedExtractionStatus(cached.status));
|
|
790
|
-
if (db && !isFailedExtractionStatus(cached.status)) {
|
|
791
|
-
upsertLlmCacheEntry(db, candidate.absPath, bodyHash, JSON.stringify(cached), cacheVariant);
|
|
792
|
-
}
|
|
793
|
-
}
|
|
794
|
-
const result = {
|
|
795
|
-
absPath: candidate.absPath,
|
|
796
|
-
type: candidate.type,
|
|
797
|
-
bodyHash,
|
|
798
|
-
entities: cached.entities,
|
|
799
|
-
relations: cached.relations,
|
|
800
|
-
...(cached.confidence !== undefined ? { confidence: cached.confidence } : {}),
|
|
801
|
-
...(cached.status ? { status: cached.status } : {}),
|
|
802
|
-
...(cached.reason ? { reason: cached.reason } : {}),
|
|
803
|
-
};
|
|
804
|
-
reportProgress(candidate.absPath, result);
|
|
805
|
-
return result;
|
|
806
|
-
},
|
|
807
|
-
// Caller-set connection concurrency or 1: `resolveLlmEngineUse` does
|
|
808
|
-
// not forward `engines.<name>.concurrency`, so config cannot raise this.
|
|
809
|
-
llmRunner.connection.concurrency ?? 1);
|
|
810
|
-
}
|
|
811
|
-
else {
|
|
812
|
-
const batch = await extractGraphBatches({
|
|
813
|
-
plans: eligiblePlans,
|
|
814
|
-
batchSize,
|
|
815
|
-
signal,
|
|
816
|
-
db,
|
|
817
|
-
cacheVariant,
|
|
818
|
-
telemetry,
|
|
819
|
-
llmRunner,
|
|
820
|
-
lease: dispatchLease,
|
|
821
|
-
featureConfig,
|
|
822
|
-
onFallback,
|
|
823
|
-
batchState,
|
|
824
|
-
runtimeTelemetry,
|
|
825
|
-
abortState,
|
|
826
|
-
onNotices,
|
|
827
|
-
reportProgress,
|
|
828
|
-
...(options.maxChunksPerAsset != null ? { maxChunksPerAsset: options.maxChunksPerAsset } : {}),
|
|
829
|
-
});
|
|
830
|
-
extractionResults = batch.results;
|
|
831
|
-
configFailure ??= batch.configFailure;
|
|
832
|
-
}
|
|
833
|
-
if (configFailure)
|
|
834
|
-
throw configFailure;
|
|
588
|
+
if ([...queuePlans, ...eligiblePlans].some((plan) => plan.kind === "model"))
|
|
589
|
+
assertRunnerCredentials(llmRunner);
|
|
590
|
+
const queueExecution = await executeQueuedGraphPlans({
|
|
591
|
+
plans: queuePlans,
|
|
592
|
+
db,
|
|
593
|
+
stashRoot: primary.path,
|
|
594
|
+
featureConfig,
|
|
595
|
+
signal,
|
|
596
|
+
llmRunner,
|
|
597
|
+
onNotices,
|
|
598
|
+
});
|
|
599
|
+
if (queueExecution.graphChanged) {
|
|
600
|
+
previousGraph = loadGraphFile(primary.path, db);
|
|
601
|
+
}
|
|
602
|
+
if (considered === 0) {
|
|
835
603
|
acknowledgeQueuedGraphPlans(db, primary.path, queueExecution);
|
|
836
|
-
|
|
837
|
-
|
|
838
|
-
|
|
839
|
-
|
|
840
|
-
|
|
841
|
-
|
|
842
|
-
|
|
843
|
-
|
|
844
|
-
|
|
845
|
-
|
|
846
|
-
|
|
847
|
-
|
|
848
|
-
|
|
849
|
-
|
|
850
|
-
|
|
851
|
-
|
|
852
|
-
|
|
853
|
-
.filter((relation) => relation.from && relation.to),
|
|
854
|
-
...(normalizeConfidence(result.confidence) !== undefined
|
|
855
|
-
? { confidence: normalizeConfidence(result.confidence) }
|
|
856
|
-
: {}),
|
|
857
|
-
status: result.status ?? (result.entities.length > 0 ? "extracted" : "empty"),
|
|
858
|
-
reason: result.reason ?? (result.entities.length > 0 ? "none" : "no_graph_content"),
|
|
859
|
-
extractionRunId,
|
|
860
|
-
});
|
|
604
|
+
const scoped = options.candidatePaths ? ` matching ${options.candidatePaths.size} candidate path(s)` : "";
|
|
605
|
+
warnVerbose(`graph extraction: skipped because no eligible files${scoped} were found under ${primary.path}. ` +
|
|
606
|
+
`includeTypes=${includeTypes.join(",")}`);
|
|
607
|
+
return emptyResult();
|
|
608
|
+
}
|
|
609
|
+
let totalEntities = 0;
|
|
610
|
+
let totalRelations = 0;
|
|
611
|
+
let processed = 0;
|
|
612
|
+
let extracted = 0;
|
|
613
|
+
onProgress?.({ processed, total: considered, extracted, totalEntities, totalRelations });
|
|
614
|
+
const reportProgress = (currentPath, result) => {
|
|
615
|
+
processed += 1;
|
|
616
|
+
if (result) {
|
|
617
|
+
if (result.entities.length > 0)
|
|
618
|
+
extracted += 1;
|
|
619
|
+
totalEntities += result.entities.length;
|
|
620
|
+
totalRelations += result.relations.length;
|
|
861
621
|
}
|
|
862
|
-
|
|
863
|
-
|
|
864
|
-
:
|
|
865
|
-
!nodes.some((candidate) => candidate.path === node.path));
|
|
866
|
-
const mergedNodes = mergeGraphNodes(previousGraph.files, [...queuedNodes, ...nodes], options.candidatePaths);
|
|
867
|
-
const assetRefs = mergedNodes.map((node) => node.path);
|
|
868
|
-
const deduped = deduplicateGraph(mergedNodes.map((node) => ({ entities: node.entities, relations: node.relations })), assetRefs);
|
|
869
|
-
telemetry.truncationCount = runtimeTelemetry.truncationCount ?? 0;
|
|
870
|
-
telemetry.truncatedChunks = runtimeTelemetry.truncatedChunks ?? 0;
|
|
871
|
-
telemetry.failureCount = runtimeTelemetry.failureCount ?? 0;
|
|
872
|
-
telemetry.htmlErrorCount = runtimeTelemetry.htmlErrorCount ?? 0;
|
|
873
|
-
telemetry.retryAttempts = runtimeTelemetry.retryAttempts ?? 0;
|
|
874
|
-
telemetry.nonArrayBatchFailures = runtimeTelemetry.nonArrayBatchFailures ?? 0;
|
|
875
|
-
telemetry.aborted = abortState.aborted;
|
|
876
|
-
const qualityConsidered = mergedNodes.length;
|
|
877
|
-
const qualityExtracted = mergedNodes.filter((node) => node.status === "extracted" && node.entities.length > 0).length;
|
|
878
|
-
const quality = computeGraphQualityTelemetry(qualityConsidered, qualityExtracted, deduped.entities.length, deduped.relations.length);
|
|
879
|
-
const warnings = buildLowQualityWarnings(quality, telemetry);
|
|
880
|
-
if (abortState.message)
|
|
881
|
-
warnings.push(abortState.message);
|
|
882
|
-
for (const warning of warnings)
|
|
883
|
-
warnVerbose(`graph extraction quality: ${warning}`);
|
|
884
|
-
const graph = {
|
|
885
|
-
schemaVersion: GRAPH_FILE_SCHEMA_VERSION,
|
|
886
|
-
generatedAt: new Date().toISOString(),
|
|
887
|
-
stashRoot: primary.path,
|
|
888
|
-
files: mergedNodes,
|
|
889
|
-
entities: deduped.entities,
|
|
890
|
-
relations: deduped.relations,
|
|
891
|
-
quality,
|
|
892
|
-
telemetry,
|
|
893
|
-
};
|
|
894
|
-
const written = writeGraphFile(primary.path, graph, db);
|
|
895
|
-
warnVerbose(`graph extraction: ${written ? "persisted" : "did not persist"} graph for ${primary.path}; ` +
|
|
896
|
-
`considered=${considered}, extractedThisRun=${extracted}, storedFiles=${mergedNodes.length}, ` +
|
|
897
|
-
`entities=${deduped.entities.length}, relations=${deduped.relations.length}, coverage=${quality.extractionCoverage}.`);
|
|
898
|
-
return {
|
|
899
|
-
considered,
|
|
622
|
+
onProgress?.({
|
|
623
|
+
processed,
|
|
624
|
+
total: considered,
|
|
900
625
|
extracted,
|
|
901
626
|
totalEntities,
|
|
902
627
|
totalRelations,
|
|
903
|
-
|
|
904
|
-
|
|
905
|
-
|
|
906
|
-
|
|
907
|
-
|
|
908
|
-
|
|
909
|
-
|
|
910
|
-
|
|
911
|
-
|
|
912
|
-
|
|
913
|
-
|
|
628
|
+
currentPath,
|
|
629
|
+
});
|
|
630
|
+
};
|
|
631
|
+
const extractionRunId = crypto.randomUUID();
|
|
632
|
+
const telemetry = {
|
|
633
|
+
extractorId,
|
|
634
|
+
extractionRunId,
|
|
635
|
+
model: llmRunner.connection.model,
|
|
636
|
+
promptVersion: graphExtract.GRAPH_EXTRACT_PROMPT_VERSION,
|
|
637
|
+
batchSize,
|
|
638
|
+
cacheHits: 0,
|
|
639
|
+
cacheMisses: 0,
|
|
640
|
+
truncationCount: 0,
|
|
641
|
+
failureCount: 0,
|
|
642
|
+
htmlErrorCount: 0,
|
|
643
|
+
retryAttempts: 0,
|
|
644
|
+
nonArrayBatchFailures: 0,
|
|
645
|
+
};
|
|
646
|
+
const runtimeTelemetry = {
|
|
647
|
+
truncationCount: 0,
|
|
648
|
+
failureCount: 0,
|
|
649
|
+
htmlErrorCount: 0,
|
|
650
|
+
retryAttempts: 0,
|
|
651
|
+
filteredGenericEntities: 0,
|
|
652
|
+
filteredInvalidRelations: 0,
|
|
653
|
+
filteredLowConfidenceRelations: 0,
|
|
654
|
+
contextBatchRetries: 0,
|
|
655
|
+
nonArrayBatchFailures: 0,
|
|
656
|
+
};
|
|
657
|
+
const abortState = { attempts: 0, failures: 0, aborted: false };
|
|
658
|
+
warnVerbose(`graph extraction: starting for ${considered} eligible file(s) under ${primary.path}; ` +
|
|
659
|
+
`includeTypes=${includeTypes.join(",")}, batchSize=${batchSize}, concurrency=${llmRunner.connection.concurrency ?? 1}, ` +
|
|
660
|
+
`reEnrich=${reEnrich === true}, candidateScoped=${options.candidatePaths ? "true" : "false"}.`);
|
|
661
|
+
const { results, configFailure } = await extractGraphBatches({
|
|
662
|
+
plans: eligiblePlans,
|
|
663
|
+
batchSize,
|
|
664
|
+
signal,
|
|
665
|
+
db,
|
|
666
|
+
cacheVariant: extractorId,
|
|
667
|
+
telemetry,
|
|
668
|
+
llmRunner,
|
|
669
|
+
featureConfig,
|
|
670
|
+
onFallback: (evt) => warn(`[akm] LLM fallback for ${evt.feature}: ${evt.reason}`),
|
|
671
|
+
batchState: { batchingDisabled: false, nonArrayBatchFailures: 0 },
|
|
672
|
+
runtimeTelemetry,
|
|
673
|
+
abortState,
|
|
674
|
+
onNotices,
|
|
675
|
+
reportProgress,
|
|
676
|
+
...(options.maxChunksPerAsset != null ? { maxChunksPerAsset: options.maxChunksPerAsset } : {}),
|
|
677
|
+
});
|
|
678
|
+
if (configFailure)
|
|
679
|
+
throw configFailure;
|
|
680
|
+
acknowledgeQueuedGraphPlans(db, primary.path, queueExecution);
|
|
681
|
+
// A failed attempt says nothing about the file, so a stored node for it stays
|
|
682
|
+
// as it was; only a file with no stored node records the failure.
|
|
683
|
+
const storedPaths = new Set(previousGraph.files.map((node) => node.path));
|
|
684
|
+
const nodes = results.flatMap((result) => !result || (isFailedExtractionStatus(result.status) && storedPaths.has(result.absPath))
|
|
685
|
+
? []
|
|
686
|
+
: [toGraphNode(result, extractionRunId)]);
|
|
687
|
+
telemetry.truncationCount = runtimeTelemetry.truncationCount ?? 0;
|
|
688
|
+
telemetry.truncatedChunks = runtimeTelemetry.truncatedChunks ?? 0;
|
|
689
|
+
telemetry.failureCount = runtimeTelemetry.failureCount ?? 0;
|
|
690
|
+
telemetry.htmlErrorCount = runtimeTelemetry.htmlErrorCount ?? 0;
|
|
691
|
+
telemetry.retryAttempts = runtimeTelemetry.retryAttempts ?? 0;
|
|
692
|
+
telemetry.nonArrayBatchFailures = runtimeTelemetry.nonArrayBatchFailures ?? 0;
|
|
693
|
+
telemetry.aborted = abortState.aborted;
|
|
694
|
+
const graph = buildGraphFile(primary.path, mergeGraphNodes(previousGraph.files, nodes, keptPaths), telemetry);
|
|
695
|
+
const written = writeGraphFile(db, graph);
|
|
696
|
+
const quality = loadStoredGraphMeta(primary.path, db)?.quality ?? EMPTY_RESULT.quality;
|
|
697
|
+
const warnings = buildLowQualityWarnings(quality, telemetry);
|
|
698
|
+
if (abortState.message)
|
|
699
|
+
warnings.push(abortState.message);
|
|
700
|
+
for (const warning of warnings)
|
|
701
|
+
warnVerbose(`graph extraction quality: ${warning}`);
|
|
702
|
+
warnVerbose(`graph extraction: ${written ? "persisted" : "did not persist"} graph for ${primary.path}; ` +
|
|
703
|
+
`considered=${considered}, extractedThisRun=${extracted}, storedFiles=${quality.consideredFiles}, ` +
|
|
704
|
+
`entities=${quality.entityCount}, relations=${quality.relationCount}, coverage=${quality.extractionCoverage}.`);
|
|
705
|
+
return {
|
|
706
|
+
considered,
|
|
707
|
+
extracted,
|
|
708
|
+
totalEntities,
|
|
709
|
+
totalRelations,
|
|
710
|
+
written,
|
|
711
|
+
quality,
|
|
712
|
+
telemetry,
|
|
713
|
+
warnings,
|
|
714
|
+
...(noticesByKey.size > 0 ? { notices: Object.freeze([...noticesByKey.values()]) } : {}),
|
|
715
|
+
};
|
|
716
|
+
}
|
|
717
|
+
/** The persisted node for one extraction outcome (entities and relations trimmed, deduplicated). */
|
|
718
|
+
function toGraphNode(record, extractionRunId) {
|
|
719
|
+
const confidence = normalizeConfidence(record.confidence);
|
|
720
|
+
return {
|
|
721
|
+
path: record.absPath,
|
|
722
|
+
type: record.type,
|
|
723
|
+
bodyHash: record.bodyHash,
|
|
724
|
+
entities: [...new Set(record.entities.map((entity) => entity.trim()).filter(Boolean))],
|
|
725
|
+
relations: record.relations
|
|
726
|
+
.map((r) => ({
|
|
727
|
+
from: r.from.trim(),
|
|
728
|
+
to: r.to.trim(),
|
|
729
|
+
...(r.type ? { type: r.type.trim() } : {}),
|
|
730
|
+
...(normalizeConfidence(r.confidence) !== undefined ? { confidence: normalizeConfidence(r.confidence) } : {}),
|
|
731
|
+
}))
|
|
732
|
+
.filter((relation) => relation.from && relation.to),
|
|
733
|
+
...(confidence !== undefined ? { confidence } : {}),
|
|
734
|
+
status: record.status ?? (record.entities.length > 0 ? "extracted" : "empty"),
|
|
735
|
+
reason: record.reason ?? (record.entities.length > 0 ? "none" : "no_graph_content"),
|
|
736
|
+
extractionRunId,
|
|
737
|
+
};
|
|
738
|
+
}
|
|
739
|
+
/** The graph snapshot to store for `files`; its counts are derived from the stored rows on write. */
|
|
740
|
+
function buildGraphFile(stashRoot, files, telemetry) {
|
|
741
|
+
return {
|
|
742
|
+
schemaVersion: GRAPH_FILE_SCHEMA_VERSION,
|
|
743
|
+
generatedAt: new Date().toISOString(),
|
|
744
|
+
stashRoot,
|
|
745
|
+
files,
|
|
746
|
+
...(telemetry ? { telemetry } : {}),
|
|
747
|
+
};
|
|
914
748
|
}
|
|
915
749
|
/**
|
|
916
750
|
* Infer the asset type (`memory`, `knowledge`, …) for a path from the stash
|
|
@@ -944,7 +778,6 @@ function inferGraphTypeForPath(stashRoot, absPath) {
|
|
|
944
778
|
* skip (missing file, empty body, unknown type, no model, or extraction error).
|
|
945
779
|
*/
|
|
946
780
|
async function extractGraphForSingleFileRevision(db, stashRoot, filePath, opts) {
|
|
947
|
-
let ownedLease;
|
|
948
781
|
try {
|
|
949
782
|
// Re-read from disk — never trust a stale queued body.
|
|
950
783
|
let raw;
|
|
@@ -963,83 +796,34 @@ async function extractGraphForSingleFileRevision(db, stashRoot, filePath, opts)
|
|
|
963
796
|
// Extract — via the injected seam, or the real per-asset path.
|
|
964
797
|
let extraction;
|
|
965
798
|
if (opts?.llmOverride) {
|
|
966
|
-
|
|
967
|
-
extraction = {
|
|
968
|
-
entities: out.entities,
|
|
969
|
-
relations: out.relations,
|
|
970
|
-
...(out.confidence !== undefined ? { confidence: out.confidence } : {}),
|
|
971
|
-
};
|
|
799
|
+
extraction = await opts.llmOverride(body);
|
|
972
800
|
}
|
|
973
801
|
else {
|
|
974
|
-
const
|
|
975
|
-
|
|
976
|
-
if (!invocationOwnsRunner && !isProcessEnabled("index", "graph_extraction", config))
|
|
802
|
+
const selection = selectGraphExecution(opts ?? {}, opts?.config ?? loadConfig());
|
|
803
|
+
if (!selection)
|
|
977
804
|
return { written: false };
|
|
978
|
-
|
|
979
|
-
|
|
980
|
-
: resolveIndexPassExecution("graph", config);
|
|
981
|
-
opts?.onNotices?.(execution.notices);
|
|
982
|
-
const llmRunner = execution.runner;
|
|
805
|
+
opts?.onNotices?.(selection.execution.notices);
|
|
806
|
+
const llmRunner = selection.execution.runner;
|
|
983
807
|
if (!llmRunner)
|
|
984
808
|
return { written: false }; // model-available guard
|
|
985
|
-
|
|
986
|
-
? { ...config, index: { ...config.index, graph: { ...config.index?.graph, enabled: true } } }
|
|
987
|
-
: config;
|
|
988
|
-
const lease = opts?.lease ?? (await preflightStructuredLlmRunner(llmRunner));
|
|
989
|
-
if (!opts?.lease)
|
|
990
|
-
ownedLease = lease;
|
|
991
|
-
const result = await graphExtract.extractGraphFromBody(llmRunner, body, opts?.signal, featureConfig, undefined, {
|
|
992
|
-
...(opts?.onNotices ? { onNotices: opts.onNotices } : {}),
|
|
993
|
-
lease,
|
|
994
|
-
});
|
|
995
|
-
extraction = {
|
|
996
|
-
entities: result.entities,
|
|
997
|
-
relations: result.relations,
|
|
998
|
-
...(result.confidence !== undefined ? { confidence: result.confidence } : {}),
|
|
999
|
-
};
|
|
809
|
+
extraction = await graphExtract.extractGraphFromBody(llmRunner, body, opts?.signal, selection.featureConfig, undefined, { ...(opts?.onNotices ? { onNotices: opts.onNotices } : {}) });
|
|
1000
810
|
}
|
|
811
|
+
// A single-file refresh records only entities, relations and confidence;
|
|
812
|
+
// its status follows from the entities that survive trimming.
|
|
1001
813
|
const entities = [...new Set(extraction.entities.map((e) => e.trim()).filter(Boolean))];
|
|
1002
|
-
const
|
|
1003
|
-
|
|
1004
|
-
from: r.from.trim(),
|
|
1005
|
-
to: r.to.trim(),
|
|
1006
|
-
...(r.type ? { type: r.type.trim() } : {}),
|
|
1007
|
-
...(normalizeConfidence(r.confidence) !== undefined ? { confidence: normalizeConfidence(r.confidence) } : {}),
|
|
1008
|
-
}))
|
|
1009
|
-
.filter((r) => r.from && r.to);
|
|
1010
|
-
const node = {
|
|
1011
|
-
path: filePath,
|
|
814
|
+
const node = toGraphNode({
|
|
815
|
+
absPath: filePath,
|
|
1012
816
|
type,
|
|
1013
817
|
bodyHash: effectiveHash,
|
|
1014
818
|
entities,
|
|
1015
|
-
relations,
|
|
1016
|
-
...(
|
|
1017
|
-
|
|
1018
|
-
|
|
1019
|
-
|
|
1020
|
-
reason: entities.length > 0 ? "none" : "no_graph_content",
|
|
1021
|
-
extractionRunId: crypto.randomUUID(),
|
|
1022
|
-
};
|
|
1023
|
-
// Merge with the previously-stored nodes, scoping the refresh to JUST this
|
|
1024
|
-
// path so other files' rows are preserved (and graph_meta counts refresh).
|
|
819
|
+
relations: extraction.relations,
|
|
820
|
+
...(extraction.confidence !== undefined ? { confidence: extraction.confidence } : {}),
|
|
821
|
+
}, crypto.randomUUID());
|
|
822
|
+
// Merge with the previously-stored nodes, replacing JUST this path so other
|
|
823
|
+
// files' rows are preserved (and graph_meta counts refresh).
|
|
1025
824
|
const previousGraph = loadGraphFile(stashRoot, db);
|
|
1026
|
-
const
|
|
1027
|
-
|
|
1028
|
-
const assetRefs = mergedNodes.map((n) => n.path);
|
|
1029
|
-
const deduped = deduplicateGraph(mergedNodes.map((n) => ({ entities: n.entities, relations: n.relations })), assetRefs);
|
|
1030
|
-
const qualityExtracted = mergedNodes.filter((n) => n.status === "extracted" && n.entities.length > 0).length;
|
|
1031
|
-
const quality = computeGraphQualityTelemetry(mergedNodes.length, qualityExtracted, deduped.entities.length, deduped.relations.length);
|
|
1032
|
-
const graph = {
|
|
1033
|
-
schemaVersion: GRAPH_FILE_SCHEMA_VERSION,
|
|
1034
|
-
generatedAt: new Date().toISOString(),
|
|
1035
|
-
stashRoot,
|
|
1036
|
-
files: mergedNodes,
|
|
1037
|
-
entities: deduped.entities,
|
|
1038
|
-
relations: deduped.relations,
|
|
1039
|
-
quality,
|
|
1040
|
-
...(previousGraph.telemetry ? { telemetry: previousGraph.telemetry } : {}),
|
|
1041
|
-
};
|
|
1042
|
-
return writeGraphFile(stashRoot, graph, db) ? { written: true, bodyHash: effectiveHash } : { written: false };
|
|
825
|
+
const graph = buildGraphFile(stashRoot, mergeGraphNodes(previousGraph.files, [node]), previousGraph.telemetry);
|
|
826
|
+
return writeGraphFile(db, graph) ? { written: true, bodyHash: effectiveHash } : { written: false };
|
|
1043
827
|
}
|
|
1044
828
|
catch (err) {
|
|
1045
829
|
if (err instanceof ConfigError)
|
|
@@ -1052,12 +836,8 @@ async function extractGraphForSingleFileRevision(db, stashRoot, filePath, opts)
|
|
|
1052
836
|
warn(`graph extraction: failed to extract graph for ${filePath}: ${err instanceof Error ? err.message : String(err)}`);
|
|
1053
837
|
return { written: false };
|
|
1054
838
|
}
|
|
1055
|
-
finally {
|
|
1056
|
-
if (ownedLease)
|
|
1057
|
-
disposeLoweredExecutionDispatchLease(ownedLease);
|
|
1058
|
-
}
|
|
1059
839
|
}
|
|
1060
|
-
export async function extractGraphForSingleFile(db, stashRoot, filePath,
|
|
840
|
+
export async function extractGraphForSingleFile(db, stashRoot, filePath, opts) {
|
|
1061
841
|
return (await extractGraphForSingleFileRevision(db, stashRoot, filePath, opts)).written;
|
|
1062
842
|
}
|
|
1063
843
|
// ── Eligible-file detection ─────────────────────────────────────────────────
|
|
@@ -1068,9 +848,7 @@ export async function extractGraphForSingleFile(db, stashRoot, filePath, _bodyHa
|
|
|
1068
848
|
* The join is READ-ONLY (`entries.file_path = candidate.absPath`, then
|
|
1069
849
|
* `entries.id -> utility_scores.entry_id`) and does NOT re-couple the graph
|
|
1070
850
|
* rows to `entries`. It reads the GLOBAL `utility_scores` table (not the
|
|
1071
|
-
* per-scope `utility_scores_scoped`), so ranking is corpus-wide
|
|
1072
|
-
* is accepted for call-site symmetry/future scoping but is not used to filter
|
|
1073
|
-
* (the global table has no `stash_root` column).
|
|
851
|
+
* per-scope `utility_scores_scoped`), so ranking is corpus-wide.
|
|
1074
852
|
*
|
|
1075
853
|
* Candidates with no matching `entries` row, or an entry with no
|
|
1076
854
|
* `utility_scores` row, get an effective utility of 0 (LEFT JOIN + COALESCE)
|
|
@@ -1083,9 +861,7 @@ export async function extractGraphForSingleFile(db, stashRoot, filePath, _bodyHa
|
|
|
1083
861
|
*
|
|
1084
862
|
* Exported for direct unit testing.
|
|
1085
863
|
*/
|
|
1086
|
-
export function rankCandidatesByUtility(db, candidates
|
|
1087
|
-
// Cannot rank without a DB → return the input unranked rather than throw.
|
|
1088
|
-
// Keeps the DB-less code path (reuse-from-memory) working when topN is set.
|
|
864
|
+
export function rankCandidatesByUtility(db, candidates) {
|
|
1089
865
|
if (!db || candidates.length === 0)
|
|
1090
866
|
return candidates;
|
|
1091
867
|
const utilityByPath = new Map();
|
|
@@ -1122,10 +898,14 @@ export function rankCandidatesByUtility(db, candidates, _stashRoot) {
|
|
|
1122
898
|
* are already derived summaries, with no additional internal graph structure worth
|
|
1123
899
|
* extracting.
|
|
1124
900
|
*
|
|
901
|
+
* `complete` is false when a directory or a candidate file could not be read,
|
|
902
|
+
* so the result may be missing eligible files.
|
|
903
|
+
*
|
|
1125
904
|
* Exported for direct unit testing.
|
|
1126
905
|
*/
|
|
1127
906
|
export function collectEligibleFiles(stashRoot, includeTypes = [...DEFAULT_GRAPH_EXTRACTION_INCLUDE_TYPES]) {
|
|
1128
907
|
const out = [];
|
|
908
|
+
let complete = true;
|
|
1129
909
|
for (const rawType of includeTypes) {
|
|
1130
910
|
const type = rawType.trim().toLowerCase();
|
|
1131
911
|
if (!SUPPORTED_GRAPH_EXTRACTION_INCLUDE_TYPES.has(type))
|
|
@@ -1138,6 +918,7 @@ export function collectEligibleFiles(stashRoot, includeTypes = [...DEFAULT_GRAPH
|
|
|
1138
918
|
continue;
|
|
1139
919
|
const walked = walkMarkdownFiles(dir);
|
|
1140
920
|
if (!walked.complete) {
|
|
921
|
+
complete = false;
|
|
1141
922
|
warn(`graph extraction: directory scan under ${dir} is incomplete — some files may be missing`);
|
|
1142
923
|
}
|
|
1143
924
|
for (const filePath of walked.files) {
|
|
@@ -1146,6 +927,7 @@ export function collectEligibleFiles(stashRoot, includeTypes = [...DEFAULT_GRAPH
|
|
|
1146
927
|
raw = fs.readFileSync(filePath, "utf8");
|
|
1147
928
|
}
|
|
1148
929
|
catch (err) {
|
|
930
|
+
complete = false;
|
|
1149
931
|
warn(`graph extraction: failed to read candidate file ${filePath}: ${err instanceof Error ? err.message : String(err)}`);
|
|
1150
932
|
continue;
|
|
1151
933
|
}
|
|
@@ -1160,23 +942,19 @@ export function collectEligibleFiles(stashRoot, includeTypes = [...DEFAULT_GRAPH
|
|
|
1160
942
|
out.push({ absPath: filePath, type, body });
|
|
1161
943
|
}
|
|
1162
944
|
}
|
|
1163
|
-
return out;
|
|
945
|
+
return { files: out, complete };
|
|
1164
946
|
}
|
|
1165
947
|
// ── Persistence ─────────────────────────────────────────────────────────────
|
|
1166
948
|
/**
|
|
1167
949
|
* Persist graph rows into the SQLite index DB.
|
|
1168
950
|
*/
|
|
1169
|
-
function writeGraphFile(
|
|
1170
|
-
if (!db) {
|
|
1171
|
-
warn("graph extraction: no database handle available; skipping graph persistence.");
|
|
1172
|
-
return false;
|
|
1173
|
-
}
|
|
951
|
+
function writeGraphFile(db, graph) {
|
|
1174
952
|
try {
|
|
1175
953
|
replaceStoredGraph(db, graph);
|
|
1176
954
|
return true;
|
|
1177
955
|
}
|
|
1178
956
|
catch (err) {
|
|
1179
|
-
warn(`graph extraction: failed to persist graph for ${stashRoot}: ${err instanceof Error ? err.message : String(err)}`);
|
|
957
|
+
warn(`graph extraction: failed to persist graph for ${graph.stashRoot}: ${err instanceof Error ? err.message : String(err)}`);
|
|
1180
958
|
return false;
|
|
1181
959
|
}
|
|
1182
960
|
}
|