akm-cli 0.9.17-alpha.2 → 0.9.17-alpha.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +756 -0
- package/dist/akm +94 -196
- package/dist/cli/shared.js +6 -2
- package/dist/cli.js +22 -9
- package/dist/commands/agent/agent-dispatch.js +1 -1
- package/dist/commands/command/command-execution.js +24 -62
- package/dist/commands/feedback-cli.js +0 -1
- package/dist/commands/health/accept-rate.js +2 -2
- package/dist/commands/health/checks.js +30 -75
- package/dist/commands/health/config-skew.js +38 -0
- package/dist/commands/health/egress.js +54 -0
- package/dist/commands/health/html-report.js +0 -38
- package/dist/commands/health/improve-metrics.js +123 -562
- package/dist/commands/health/plugin-staleness.js +53 -3
- package/dist/commands/health/renderers.js +12 -4
- package/dist/commands/health/report-view-model.js +11 -106
- package/dist/commands/health/types-improve.js +4 -19
- package/dist/commands/health/windows.js +64 -73
- package/dist/commands/health.js +122 -143
- package/dist/commands/improve/consolidate/chunking.js +25 -100
- package/dist/commands/improve/consolidate/sanitize.js +54 -149
- package/dist/commands/improve/consolidate.js +538 -1075
- package/dist/commands/improve/content-hash.js +16 -24
- package/dist/commands/improve/distill/content-repair.js +18 -100
- package/dist/commands/improve/distill-guards.js +20 -81
- package/dist/commands/improve/distill-promotion-policy.js +23 -243
- package/dist/commands/improve/distill.js +608 -1075
- package/dist/commands/improve/eligibility.js +126 -400
- package/dist/commands/improve/execution.js +3 -5
- package/dist/commands/improve/extract.js +487 -1046
- package/dist/commands/improve/feedback-valence.js +0 -25
- package/dist/commands/improve/improve-cli.js +29 -166
- package/dist/commands/improve/improve-result-file.js +10 -66
- package/dist/commands/improve/improve-strategies.js +12 -7
- package/dist/commands/improve/improve-usage-report.js +18 -64
- package/dist/commands/improve/improve.js +443 -1063
- package/dist/commands/improve/ledger.js +114 -0
- package/dist/commands/improve/locks.js +2 -8
- package/dist/commands/improve/loop-stages.js +459 -1172
- package/dist/commands/improve/memory/derived-ref.js +12 -77
- package/dist/commands/improve/memory/memory-belief.js +14 -118
- package/dist/commands/improve/memory/memory-improve.js +4 -3
- package/dist/commands/improve/outcome-loop.js +28 -156
- package/dist/commands/improve/planner.js +5 -10
- package/dist/commands/improve/preparation.js +851 -2339
- package/dist/commands/improve/proactive-maintenance.js +34 -101
- package/dist/commands/improve/reflect-noise.js +104 -280
- package/dist/commands/improve/reflect.js +621 -1367
- package/dist/commands/improve/salience.js +46 -232
- package/dist/commands/improve/session-asset.js +19 -100
- package/dist/commands/improve/stage.js +323 -0
- package/dist/commands/proposal/drain.js +251 -644
- package/dist/commands/proposal/proposal-cli.js +3 -18
- package/dist/commands/proposal/proposal-types.js +20 -41
- package/dist/commands/proposal/proposal.js +1 -2
- package/dist/commands/proposal/propose.js +134 -160
- package/dist/commands/proposal/repository.js +502 -1487
- package/dist/commands/proposal/validators/proposal-quality-validators.js +71 -174
- package/dist/commands/proposal/validators/proposal-validators.js +1 -1
- package/dist/commands/proposal/validators/proposals.js +13 -89
- package/dist/commands/read/curate.js +63 -413
- package/dist/commands/read/search-cli.js +16 -33
- package/dist/commands/read/search.js +17 -23
- package/dist/commands/read/show.js +2 -13
- package/dist/commands/sources/bundle-cli.js +25 -2
- package/dist/commands/sources/bundle-config-ops.js +7 -0
- package/dist/commands/sources/dangerous-env-audit.js +1 -2
- package/dist/commands/sources/info.js +2 -11
- package/dist/commands/sources/installed-stashes.js +197 -746
- package/dist/commands/sources/schema-repair.js +98 -129
- package/dist/commands/sources/source-add.js +62 -12
- package/dist/commands/sources/stash-cli.js +1 -1
- package/dist/commands/tasks/explain.js +10 -13
- package/dist/commands/tasks/tasks-cli.js +9 -8
- package/dist/commands/tasks/tasks.js +326 -930
- package/dist/commands/tasks/validate.js +42 -21
- package/dist/commands/workflow/plan.js +22 -29
- package/dist/commands/workflow-cli.js +4 -4
- package/dist/core/adapter/adapters/akm-adapter.js +0 -1
- package/dist/core/adapter/adapters/akm-lint.js +2 -3
- package/dist/core/adapter/adapters/akm-metadata.js +11 -12
- package/dist/core/adapter/adapters/akm-workflow-adapter.js +1 -1
- package/dist/core/adapter/execution-source.js +17 -29
- package/dist/core/asset/resolve-ref.js +1 -1
- package/dist/core/bundle-id.js +42 -5
- package/dist/core/bundle-rename.js +291 -0
- package/dist/core/config/config-io.js +1 -2
- package/dist/core/config/config-schema.js +1 -33
- package/dist/core/config/config-walker.js +1 -1
- package/dist/core/config/config.js +163 -68
- package/dist/core/config/legacy-source-shape-shim.js +38 -9
- package/dist/core/config/schema/embedding.js +20 -5
- package/dist/core/config/schema/engines.js +5 -0
- package/dist/core/config/schema/execution.js +1 -1
- package/dist/core/config/schema/experimental.js +1 -1
- package/dist/core/config/schema/improve-processes.js +21 -95
- package/dist/core/config/schema/improve.js +4 -42
- package/dist/core/config/schema/scheduler.js +12 -12
- package/dist/core/config/schema/search.js +6 -22
- package/dist/core/env-secret-ref.js +0 -1
- package/dist/core/errors.js +8 -9
- package/dist/core/file-lock.js +76 -173
- package/dist/core/logs-db.js +2 -2
- package/dist/core/paths.js +0 -27
- package/dist/core/redaction.js +109 -2
- package/dist/core/run-lock.js +2 -5
- package/dist/core/spawn-env.js +1 -1
- package/dist/core/state/migrations.js +108 -61
- package/dist/core/state-db-scope.js +2 -4
- package/dist/core/state-db.js +126 -692
- package/dist/core/type-presentation.js +1 -9
- package/dist/core/write-source.js +293 -1012
- package/dist/execution/input-contract.js +1 -1
- package/dist/execution/resolved-request.js +135 -689
- package/dist/execution/source.js +63 -257
- package/dist/execution/target-ref.js +1 -1
- package/dist/indexer/bundle-identity-guard.js +2 -2
- package/dist/indexer/db/graph-db.js +106 -46
- package/dist/indexer/ensure-index.js +44 -85
- package/dist/indexer/graph/graph-extraction.js +340 -562
- package/dist/indexer/graph/graph-related.js +130 -0
- package/dist/indexer/index-rebuild-lock.js +3 -11
- package/dist/indexer/index-writer-lock.js +8 -17
- package/dist/indexer/index-written-assets.js +139 -151
- package/dist/indexer/indexer.js +524 -846
- package/dist/indexer/materialize-embeddings.js +60 -397
- package/dist/indexer/passes/memory-inference.js +81 -90
- package/dist/indexer/passes/metadata.js +132 -200
- package/dist/indexer/read-preflight.js +0 -7
- package/dist/indexer/scan/doc-to-entry.js +1 -3
- package/dist/indexer/scan/drain-dir.js +1 -1
- package/dist/indexer/search/db-search.js +181 -590
- package/dist/indexer/search/fts-query.js +30 -41
- package/dist/indexer/search/ranking.js +28 -154
- package/dist/indexer/search/search-attribution.js +12 -32
- package/dist/indexer/search/search-fields.js +11 -15
- package/dist/indexer/search/search-hit-enrichers.js +54 -85
- package/dist/indexer/search/search-source.js +1 -4
- package/dist/indexer/usage/usage-events.js +2 -7
- package/dist/integrations/agent/engine-fallback.js +23 -40
- package/dist/integrations/agent/engine-resolution.js +93 -183
- package/dist/integrations/agent/execution.js +507 -0
- package/dist/integrations/agent/model-map.js +28 -156
- package/dist/integrations/agent/request-lowering.js +66 -141
- package/dist/integrations/agent/runner-dispatch.js +143 -321
- package/dist/integrations/agent/runner.js +54 -14
- package/dist/integrations/lockfile.js +53 -101
- package/dist/llm/embedders/deterministic.js +2 -3
- package/dist/llm/embedders/profile.js +71 -0
- package/dist/llm/embedders/remote.js +10 -15
- package/dist/llm/graph-extract.js +3 -12
- package/dist/llm/index-passes.js +3 -5
- package/dist/llm/memory-infer.js +1 -2
- package/dist/llm/metadata-enhance.js +1 -2
- package/dist/llm/structured-call.js +5 -24
- package/dist/output/generic-render.js +23 -11
- package/dist/output/html-render.js +13 -10
- package/dist/output/render-registry.js +3 -32
- package/dist/output/shapes/helpers.js +2 -34
- package/dist/output/shapes/passthrough.js +1 -9
- package/dist/{indexer/search/ranking-types.js → output/text/bundle-rename.js} +4 -1
- package/dist/output/text/command-format.js +60 -23
- package/dist/output/text/helpers.js +1 -1
- package/dist/output/text/migrate.js +5 -14
- package/dist/output/text/proposal-format.js +1 -2
- package/dist/output/text/workflow-format.js +0 -32
- package/dist/output/text.js +2 -0
- package/dist/registry/factory.js +4 -19
- package/dist/registry/network.js +66 -220
- package/dist/registry/providers/index.js +0 -2
- package/dist/registry/providers/skills-sh.js +3 -14
- package/dist/registry/providers/static-index.js +24 -26
- package/dist/registry/resolve.js +55 -131
- package/dist/scripts/akm-migrate-node.js +43937 -93313
- package/dist/scripts/akm-migrate.js +43697 -93071
- package/dist/setup/registry-stash-loader.js +4 -13
- package/dist/setup/semantic-assets.js +3 -44
- package/dist/setup/setup.js +1 -1
- package/dist/setup/steps/tasks.js +25 -15
- package/dist/sources/provider-factory.js +17 -18
- package/dist/sources/providers/filesystem.js +2 -3
- package/dist/sources/providers/git-install.js +7 -1
- package/dist/sources/providers/git-provider.js +0 -3
- package/dist/sources/providers/git-stash.js +0 -17
- package/dist/sources/providers/npm.js +2 -4
- package/dist/sources/providers/provider-utils.js +5 -10
- package/dist/sources/providers/website.js +0 -2
- package/dist/sources/snapshot-fetchers/website-ingest.js +1 -1
- package/dist/sources/website-url.js +2 -2
- package/dist/storage/database.js +9 -35
- package/dist/storage/repositories/improve-ledger-repository.js +168 -0
- package/dist/storage/repositories/index-connection.js +34 -70
- package/dist/storage/repositories/index-entries-repository.js +69 -111
- package/dist/storage/repositories/index-entry-mapper.js +1 -2
- package/dist/storage/repositories/index-entry-schema.js +83 -269
- package/dist/storage/repositories/index-fts-repository.js +86 -256
- package/dist/storage/repositories/index-llm-cache-repository.js +17 -0
- package/dist/storage/repositories/index-meta-repository.js +6 -4
- package/dist/storage/repositories/index-schema.js +192 -220
- package/dist/storage/repositories/index-utility-repository.js +8 -29
- package/dist/storage/repositories/index-vec-repository.js +133 -414
- package/dist/storage/repositories/outcome-repository.js +2 -1
- package/dist/storage/repositories/proposals-repository.js +35 -0
- package/dist/storage/repositories/registry-index-cache-repository.js +100 -0
- package/dist/storage/repositories/task-history-repository.js +26 -4
- package/dist/storage/repositories/workflow-runs-repository.js +53 -244
- package/dist/storage/sqlite-migrations.js +136 -0
- package/dist/storage/sqlite-pragmas.js +11 -9
- package/dist/storage/sqlite-transaction.js +170 -0
- package/dist/storage/state-db-integrity.js +34 -27
- package/dist/tasks/activation-config.js +134 -62
- package/dist/tasks/backends/cron.js +129 -277
- package/dist/tasks/backends/exec-utils.js +2 -5
- package/dist/tasks/backends/launchd.js +125 -745
- package/dist/tasks/backends/schtasks.js +101 -620
- package/dist/tasks/prepare/prepare-support.js +5 -15
- package/dist/tasks/prepare/prepare.js +0 -2
- package/dist/tasks/resolve-akm-bin.js +20 -79
- package/dist/tasks/run/attempt-lifecycle.js +0 -1
- package/dist/tasks/scheduler-binding.js +18 -238
- package/dist/tasks/scheduler-invocation.js +52 -52
- package/dist/tasks/scheduler-lock.js +53 -0
- package/dist/tasks/scheduler-sync.js +363 -679
- package/dist/tasks/source/parse-task-source.js +160 -10
- package/dist/tasks/source/task-source-v3-frozen.js +3 -4
- package/dist/tasks/source/task-to-v4.js +2 -2
- package/dist/workflows/authoring/authoring.js +3 -12
- package/dist/workflows/compile.js +211 -0
- package/dist/workflows/concurrency-policy.js +13 -74
- package/dist/workflows/exec/child-invocation.js +3 -17
- package/dist/workflows/exec/child-workflow.js +32 -141
- package/dist/workflows/exec/dispatch-redaction.js +13 -53
- package/dist/workflows/exec/environment.js +98 -0
- package/dist/workflows/exec/exec-unit.js +33 -140
- package/dist/workflows/exec/frozen-judge.js +7 -59
- package/dist/workflows/exec/native-executor.js +82 -341
- package/dist/workflows/exec/param-secrets.js +29 -47
- package/dist/workflows/exec/run-workflow.js +154 -387
- package/dist/workflows/exec/scheduler.js +9 -36
- package/dist/workflows/exec/step-work.js +127 -430
- package/dist/workflows/exec/unit-dispatch.js +11 -63
- package/dist/workflows/exec/unit-writer.js +8 -52
- package/dist/workflows/exec/worktree.js +39 -273
- package/dist/workflows/freeze/child-output-references.js +4 -15
- package/dist/workflows/freeze/environment.js +99 -92
- package/dist/workflows/freeze/freeze.js +172 -0
- package/dist/workflows/freeze/step-values.js +19 -21
- package/dist/workflows/freeze/targets/child-workflow.js +23 -92
- package/dist/workflows/freeze/targets/command.js +10 -33
- package/dist/workflows/freeze/targets/script.js +5 -12
- package/dist/workflows/freeze/targets/shell.js +3 -6
- package/dist/workflows/freeze/targets/task.js +25 -80
- package/dist/workflows/freeze/task-bindings.js +20 -67
- package/dist/workflows/{source-ir/github-yaml.js → github-yaml.js} +88 -206
- package/dist/workflows/ir/params.js +6 -51
- package/dist/workflows/ir/plan-hash.js +2 -34
- package/dist/workflows/parser.js +140 -43
- package/dist/{commands/improve/consolidate/types.js → workflows/plan.js} +2 -1
- package/dist/workflows/renderer.js +36 -69
- package/dist/workflows/resource-limits.js +12 -120
- package/dist/workflows/runtime/agent-identity.js +8 -40
- package/dist/workflows/runtime/run-outputs.js +3 -6
- package/dist/workflows/runtime/run-plan.js +316 -0
- package/dist/workflows/runtime/runs.js +48 -200
- package/dist/workflows/runtime/workflow-asset-loader.js +24 -57
- package/dist/workflows/{source-ir/semantics.js → source-semantics.js} +16 -20
- package/dist/workflows/validate-summary.js +2 -7
- package/docs/integration/bundling-akm.md +49 -42
- package/docs/migration/README.md +1 -0
- package/docs/migration/release-notes/0.9.17.md +41 -0
- package/docs/migration/v0.9.1-to-v0.9.2.md +19 -7
- package/docs/reference/cli.md +182 -125
- package/docs/reference/configuration.md +49 -56
- package/docs/reference/data-and-telemetry.md +19 -20
- package/docs/reference/tasks.md +86 -38
- package/docs/reference/workflow-schema.md +14 -18
- package/docs/reference/workflows.md +6 -9
- package/package.json +1 -1
- package/schemas/akm-config.json +87 -406
- package/dist/commands/health/advisories.js +0 -150
- package/dist/commands/health/metrics.js +0 -329
- package/dist/commands/health/surfaces.js +0 -102
- package/dist/commands/improve/anti-collapse.js +0 -83
- package/dist/commands/improve/collapse-detector.js +0 -432
- package/dist/commands/improve/consolidate/eligibility.js +0 -48
- package/dist/commands/improve/consolidate/merge.js +0 -146
- package/dist/commands/improve/distill/promote-memory.js +0 -329
- package/dist/commands/improve/distill/quality-gate.js +0 -500
- package/dist/commands/improve/memory/memory-contradiction-detect.js +0 -291
- package/dist/commands/improve/proposal-envelope.js +0 -31
- package/dist/commands/improve/run-context.js +0 -123
- package/dist/commands/improve/shared.js +0 -21
- package/dist/commands/improve/source-identity.js +0 -28
- package/dist/commands/improve/triage.js +0 -96
- package/dist/commands/proposal/drain-policies.js +0 -151
- package/dist/commands/sources/update-transaction.js +0 -220
- package/dist/core/action-contributors.js +0 -28
- package/dist/core/config/config-version-shim.js +0 -101
- package/dist/core/config/retired-experimental-keys-shim.js +0 -62
- package/dist/core/fs-txn.js +0 -405
- package/dist/core/lexical-score.js +0 -25
- package/dist/core/maintenance-barrier.js +0 -167
- package/dist/execution/executable-identity.js +0 -105
- package/dist/execution/guarded-source.js +0 -427
- package/dist/indexer/graph/graph-boost.js +0 -427
- package/dist/indexer/graph/graph-dedup.js +0 -95
- package/dist/indexer/search/name-match.js +0 -35
- package/dist/indexer/search/ranking-contributors.js +0 -515
- package/dist/indexer/walk/project-context.js +0 -192
- package/dist/integrations/agent/execution-cascade.js +0 -566
- package/dist/integrations/agent/execution-definitions.js +0 -202
- package/dist/integrations/agent/execution-lowering.js +0 -841
- package/dist/integrations/agent/execution-preparation.js +0 -98
- package/dist/integrations/agent/inline-execution.js +0 -74
- package/dist/registry/create-provider-registry.js +0 -29
- package/dist/registry/pinned-request-helper.js +0 -247
- package/dist/registry/pinned-transport.js +0 -717
- package/dist/sources/providers/index.js +0 -14
- package/dist/storage/engines/sqlite-migrations.js +0 -271
- package/dist/storage/repositories/canaries-repository.js +0 -107
- package/dist/storage/repositories/embedding-salvage-repository.js +0 -184
- package/dist/storage/repositories/registry-cache.js +0 -113
- package/dist/tasks/scheduler-sync-preview.js +0 -52
- package/dist/workflows/freeze/resolve-steps.js +0 -86
- package/dist/workflows/freeze/source-freeze.js +0 -64
- package/dist/workflows/ir/compile.js +0 -321
- package/dist/workflows/ir/environment-v4.js +0 -330
- package/dist/workflows/ir/freeze-v4.js +0 -153
- package/dist/workflows/ir/schema-v4.js +0 -745
- package/dist/workflows/ir/schema.js +0 -354
- package/dist/workflows/program/schema.js +0 -78
- package/dist/workflows/runtime/checkin.js +0 -57
- package/dist/workflows/runtime/plan-classifier.js +0 -196
- package/dist/workflows/runtime/unit-checkin.js +0 -45
- package/dist/workflows/runtime/unit-phases.js +0 -20
- package/dist/workflows/schema.js +0 -4
- package/dist/workflows/source-ir/compile.js +0 -200
- package/dist/workflows/source-ir/program.js +0 -50
- package/dist/workflows/source-ir/result.js +0 -26
- package/dist/workflows/source-ir/schema.js +0 -786
- package/dist/workflows/source-ir/triggers.js +0 -79
- package/dist/workflows/source-ir/uses.js +0 -40
- package/dist/workflows/validator.js +0 -60
|
@@ -1,427 +0,0 @@
|
|
|
1
|
-
// This Source Code Form is subject to the terms of the Mozilla Public
|
|
2
|
-
// License, v. 2.0. If a copy of the MPL was not distributed with this
|
|
3
|
-
// file, You can obtain one at https://mozilla.org/MPL/2.0/.
|
|
4
|
-
import { loadStoredGraphMeta, loadStoredGraphSnapshot } from "../db/graph-db.js";
|
|
5
|
-
function normalizeGraphName(value) {
|
|
6
|
-
return value.trim().toLowerCase();
|
|
7
|
-
}
|
|
8
|
-
let cachedParsedGraph;
|
|
9
|
-
/**
|
|
10
|
-
* Clear the module-level parsed-graph cache.
|
|
11
|
-
*
|
|
12
|
-
* The cache keeps the most recently parsed `ParsedGraphContext` keyed by
|
|
13
|
-
* (stashPath, generatedAt) tuples so back-to-back search invocations within
|
|
14
|
-
* the same process don't re-read the SQLite snapshot from disk. The cache
|
|
15
|
-
* persists across calls — which is the desired behaviour in production but
|
|
16
|
-
* pathological for tests that swap the underlying stash directory between
|
|
17
|
-
* test cases without bumping `generatedAt`. Such tests can observe stale
|
|
18
|
-
* graph nodes from a previous test's stash.
|
|
19
|
-
*
|
|
20
|
-
* Tests (and any tooling that swaps stash backings) should call this between
|
|
21
|
-
* setups to guarantee the next `loadGraphBoostContext` reads fresh state.
|
|
22
|
-
* Recommended placement: `beforeEach` for test files that mutate graph
|
|
23
|
-
* state, or after `deleteStoredGraph` / `replaceStoredGraph` calls that
|
|
24
|
-
* intentionally invalidate the cache.
|
|
25
|
-
*/
|
|
26
|
-
export function resetGraphBoostCache() {
|
|
27
|
-
cachedParsedGraph = undefined;
|
|
28
|
-
}
|
|
29
|
-
function resolveGraphBoostWeights(config) {
|
|
30
|
-
const configured = config?.search?.graphBoost;
|
|
31
|
-
return {
|
|
32
|
-
directBoostPerEntity: configured?.directBoostPerEntity ?? GRAPH_DIRECT_BOOST_PER_ENTITY,
|
|
33
|
-
directBoostCap: configured?.directBoostCap ?? GRAPH_DIRECT_BOOST_CAP,
|
|
34
|
-
hopBoostPerEntity: configured?.hopBoostPerEntity ?? GRAPH_HOP_BOOST_PER_ENTITY,
|
|
35
|
-
hopBoostCap: configured?.hopBoostCap ?? GRAPH_HOP_BOOST_CAP,
|
|
36
|
-
maxHops: Math.min(Math.max(configured?.maxHops ?? GRAPH_MAX_HOPS, 1), GRAPH_MAX_HOPS_HARD_CAP),
|
|
37
|
-
confidenceMode: configured?.confidenceMode ?? GRAPH_CONFIDENCE_MODE,
|
|
38
|
-
confidenceWeight: configured?.confidenceWeight ?? GRAPH_CONFIDENCE_WEIGHT,
|
|
39
|
-
};
|
|
40
|
-
}
|
|
41
|
-
/**
|
|
42
|
-
* Per-entry weights, exposed as constants so tests can read them and so the
|
|
43
|
-
* single-source-of-truth for "how much does the graph contribute" is here
|
|
44
|
-
* rather than inlined into `db-search.ts`. Kept conservative — the goal is
|
|
45
|
-
* a useful tiebreaker, not domination of the lexical signal.
|
|
46
|
-
*/
|
|
47
|
-
export const GRAPH_DIRECT_BOOST_PER_ENTITY = 0.25;
|
|
48
|
-
export const GRAPH_DIRECT_BOOST_CAP = 0.75;
|
|
49
|
-
export const GRAPH_HOP_BOOST_PER_ENTITY = 0.1;
|
|
50
|
-
export const GRAPH_HOP_BOOST_CAP = 0.3;
|
|
51
|
-
export const GRAPH_MAX_HOPS = 1;
|
|
52
|
-
export const GRAPH_CONFIDENCE_MODE = "blend";
|
|
53
|
-
export const GRAPH_CONFIDENCE_WEIGHT = 0.2;
|
|
54
|
-
const GRAPH_MAX_HOPS_HARD_CAP = 3;
|
|
55
|
-
function normalizeConfidence(raw) {
|
|
56
|
-
if (typeof raw !== "number" || !Number.isFinite(raw))
|
|
57
|
-
return undefined;
|
|
58
|
-
return Math.max(0, Math.min(1, raw));
|
|
59
|
-
}
|
|
60
|
-
function combineConfidence(...parts) {
|
|
61
|
-
let out;
|
|
62
|
-
for (const part of parts) {
|
|
63
|
-
const value = normalizeConfidence(part);
|
|
64
|
-
if (value === undefined)
|
|
65
|
-
continue;
|
|
66
|
-
out = out === undefined ? value : out * value;
|
|
67
|
-
}
|
|
68
|
-
return out;
|
|
69
|
-
}
|
|
70
|
-
function toConfidenceMultiplier(rawConfidence, weights) {
|
|
71
|
-
const confidence = normalizeConfidence(rawConfidence) ?? 1;
|
|
72
|
-
const blendWeight = Math.max(0, Math.min(1, weights.confidenceWeight));
|
|
73
|
-
return 1 - blendWeight + blendWeight * confidence;
|
|
74
|
-
}
|
|
75
|
-
/**
|
|
76
|
-
* Load the graph file for a stash root and pre-compute everything that's
|
|
77
|
-
* shared across all entries scored for one query. Returns `null` when:
|
|
78
|
-
* - No graph snapshot exists in SQLite.
|
|
79
|
-
* - The query produces no token-level entity matches (no boost is
|
|
80
|
-
* possible, so we skip the per-entry overhead entirely).
|
|
81
|
-
*/
|
|
82
|
-
export function loadGraphBoostContext(stashRoot, query, config, db) {
|
|
83
|
-
const stashRoots = Array.isArray(stashRoot) ? stashRoot : [stashRoot];
|
|
84
|
-
const parsed = readParsedGraphContext(stashRoots, db);
|
|
85
|
-
if (!parsed)
|
|
86
|
-
return null;
|
|
87
|
-
const weights = resolveGraphBoostWeights(config);
|
|
88
|
-
const queryTokens = query
|
|
89
|
-
.toLowerCase()
|
|
90
|
-
.split(/[\s\-_/]+/)
|
|
91
|
-
.filter((t) => t.length >= 2);
|
|
92
|
-
if (queryTokens.length === 0)
|
|
93
|
-
return null;
|
|
94
|
-
// Build a flat union of all extracted entities across the corpus. This
|
|
95
|
-
// is small (capped per-asset at extract time) and lets the per-entry
|
|
96
|
-
// path do a single set membership test.
|
|
97
|
-
const allEntities = new Set();
|
|
98
|
-
for (const node of parsed.graph.files) {
|
|
99
|
-
for (const entity of node.entities)
|
|
100
|
-
allEntities.add(entity);
|
|
101
|
-
}
|
|
102
|
-
// An entity matches the query when any of its sub-tokens equals or
|
|
103
|
-
// contains a query token. Matching is case-insensitive; the graph keeps
|
|
104
|
-
// canonical display strings and we normalize only for comparisons here.
|
|
105
|
-
const matchedEntities = new Set();
|
|
106
|
-
for (const entity of allEntities) {
|
|
107
|
-
const normalizedEntity = normalizeGraphName(entity);
|
|
108
|
-
const entityTokens = normalizedEntity.split(/[\s\-_/]+/).filter(Boolean);
|
|
109
|
-
for (const qt of queryTokens) {
|
|
110
|
-
if (normalizedEntity === qt ||
|
|
111
|
-
normalizedEntity.includes(qt) ||
|
|
112
|
-
entityTokens.some((et) => et === qt || et.includes(qt))) {
|
|
113
|
-
matchedEntities.add(entity);
|
|
114
|
-
break;
|
|
115
|
-
}
|
|
116
|
-
}
|
|
117
|
-
}
|
|
118
|
-
if (matchedEntities.size === 0)
|
|
119
|
-
return null;
|
|
120
|
-
const connectedEntities = new Set();
|
|
121
|
-
const connectedConfidence = new Map();
|
|
122
|
-
const visited = new Set();
|
|
123
|
-
let frontier = new Map();
|
|
124
|
-
for (const entity of matchedEntities) {
|
|
125
|
-
const seed = parsed.entityConfidence.get(entity) ?? 1;
|
|
126
|
-
frontier.set(entity, seed);
|
|
127
|
-
visited.add(entity);
|
|
128
|
-
}
|
|
129
|
-
for (let hop = 1; hop <= weights.maxHops; hop += 1) {
|
|
130
|
-
const next = new Map();
|
|
131
|
-
for (const [entity, pathConfidence] of frontier.entries()) {
|
|
132
|
-
const neighbors = parsed.adjacency.get(entity);
|
|
133
|
-
if (!neighbors)
|
|
134
|
-
continue;
|
|
135
|
-
for (const [neighbor, edgeConfidence] of neighbors.entries()) {
|
|
136
|
-
const neighborPathConfidence = Math.max(0, Math.min(1, pathConfidence * edgeConfidence));
|
|
137
|
-
const currentBest = connectedConfidence.get(neighbor) ?? 0;
|
|
138
|
-
if (neighborPathConfidence > currentBest)
|
|
139
|
-
connectedConfidence.set(neighbor, neighborPathConfidence);
|
|
140
|
-
if (visited.has(neighbor))
|
|
141
|
-
continue;
|
|
142
|
-
visited.add(neighbor);
|
|
143
|
-
next.set(neighbor, Math.max(next.get(neighbor) ?? 0, neighborPathConfidence));
|
|
144
|
-
connectedEntities.add(neighbor);
|
|
145
|
-
}
|
|
146
|
-
}
|
|
147
|
-
if (next.size === 0)
|
|
148
|
-
break;
|
|
149
|
-
frontier = next;
|
|
150
|
-
}
|
|
151
|
-
return {
|
|
152
|
-
graph: parsed.graph,
|
|
153
|
-
nodesByPath: parsed.nodesByPath,
|
|
154
|
-
matchedEntities,
|
|
155
|
-
connectedEntities,
|
|
156
|
-
connectedConfidence,
|
|
157
|
-
entityConfidence: parsed.entityConfidence,
|
|
158
|
-
adjacency: parsed.adjacency,
|
|
159
|
-
weights,
|
|
160
|
-
};
|
|
161
|
-
}
|
|
162
|
-
/**
|
|
163
|
-
* Compute the graph-boost contribution for a single scored entry.
|
|
164
|
-
*
|
|
165
|
-
* The return value is added directly into `boostSum` in `searchDatabase`'s
|
|
166
|
-
* existing scoring loop — same units, same cap policy. Returns `0` when
|
|
167
|
-
* the entry's file isn't in the graph or when no entity overlap exists.
|
|
168
|
-
*/
|
|
169
|
-
export function computeGraphBoost(context, filePath) {
|
|
170
|
-
const node = context.nodesByPath.get(filePath);
|
|
171
|
-
if (!node)
|
|
172
|
-
return 0;
|
|
173
|
-
let directBoostRaw = 0;
|
|
174
|
-
let hopBoostRaw = 0;
|
|
175
|
-
for (const entity of node.entities) {
|
|
176
|
-
if (context.matchedEntities.has(entity)) {
|
|
177
|
-
const directConfidence = combineConfidence(node.confidence, context.entityConfidence.get(entity));
|
|
178
|
-
directBoostRaw +=
|
|
179
|
-
context.weights.directBoostPerEntity * toConfidenceMultiplier(directConfidence, context.weights);
|
|
180
|
-
}
|
|
181
|
-
else if (context.connectedEntities.has(entity)) {
|
|
182
|
-
const hopConfidence = combineConfidence(node.confidence, context.entityConfidence.get(entity), context.connectedConfidence.get(entity));
|
|
183
|
-
hopBoostRaw += context.weights.hopBoostPerEntity * toConfidenceMultiplier(hopConfidence, context.weights);
|
|
184
|
-
}
|
|
185
|
-
}
|
|
186
|
-
const directBoost = Math.min(context.weights.directBoostCap, directBoostRaw);
|
|
187
|
-
const hopBoost = Math.min(context.weights.hopBoostCap, hopBoostRaw);
|
|
188
|
-
return directBoost + hopBoost;
|
|
189
|
-
}
|
|
190
|
-
export function collectGraphRelatedHit(context, filePath) {
|
|
191
|
-
const node = context.nodesByPath.get(filePath);
|
|
192
|
-
if (!node)
|
|
193
|
-
return null;
|
|
194
|
-
const entities = [];
|
|
195
|
-
for (const entity of node.entities) {
|
|
196
|
-
if (context.matchedEntities.has(entity)) {
|
|
197
|
-
entities.push({
|
|
198
|
-
name: entity,
|
|
199
|
-
kind: "matched",
|
|
200
|
-
...(context.entityConfidence.get(entity) !== undefined
|
|
201
|
-
? { confidence: context.entityConfidence.get(entity) }
|
|
202
|
-
: {}),
|
|
203
|
-
});
|
|
204
|
-
continue;
|
|
205
|
-
}
|
|
206
|
-
if (context.connectedEntities.has(entity)) {
|
|
207
|
-
entities.push({
|
|
208
|
-
name: entity,
|
|
209
|
-
kind: "connected",
|
|
210
|
-
...(context.connectedConfidence.get(entity) !== undefined
|
|
211
|
-
? { confidence: context.connectedConfidence.get(entity) }
|
|
212
|
-
: {}),
|
|
213
|
-
});
|
|
214
|
-
}
|
|
215
|
-
}
|
|
216
|
-
if (entities.length === 0)
|
|
217
|
-
return null;
|
|
218
|
-
const relatedNames = new Set(entities.map((entity) => entity.name));
|
|
219
|
-
const relations = node.relations
|
|
220
|
-
.filter((relation) => relatedNames.has(relation.from) || relatedNames.has(relation.to))
|
|
221
|
-
.map((relation) => ({
|
|
222
|
-
from: relation.from,
|
|
223
|
-
to: relation.to,
|
|
224
|
-
...(relation.type ? { type: relation.type } : {}),
|
|
225
|
-
...(normalizeConfidence(relation.confidence) !== undefined
|
|
226
|
-
? { confidence: normalizeConfidence(relation.confidence) }
|
|
227
|
-
: {}),
|
|
228
|
-
}));
|
|
229
|
-
return {
|
|
230
|
-
path: filePath,
|
|
231
|
-
type: node.type,
|
|
232
|
-
entities: entities.sort((a, b) => a.name.localeCompare(b.name)),
|
|
233
|
-
relations,
|
|
234
|
-
};
|
|
235
|
-
}
|
|
236
|
-
/**
|
|
237
|
-
* Find graph files that share entities with the given file.
|
|
238
|
-
*
|
|
239
|
-
* Implementation: SQL self-join on graph_file_entities, scoped by stash_root,
|
|
240
|
-
* grouped by file_path, ordered by shared-entity count desc. Touches ~50-200
|
|
241
|
-
* rows instead of loading the entire snapshot into memory. Cold-call latency
|
|
242
|
-
* drops from ~30-60ms (full snapshot parse) to ~2-5ms on typical stashes.
|
|
243
|
-
*
|
|
244
|
-
* #624-P1: the graph tables are keyed on (stash_root, file_path, body_hash) —
|
|
245
|
-
* NOT entries.id — so candidates are identified by file_path (the unique index
|
|
246
|
-
* idx_graph_files_path guarantees one graph_files row per path).
|
|
247
|
-
*
|
|
248
|
-
* The returned `ref` field carries the canonical indexed `concept_id`, never a
|
|
249
|
-
* value re-derived from presentation fields. It is undefined
|
|
250
|
-
* when the graph row has no matching entry with current indexed provenance.
|
|
251
|
-
*/
|
|
252
|
-
export function listRelatedPathsForFile(stashRoot, filePath, limit = 5, db) {
|
|
253
|
-
if (!db) {
|
|
254
|
-
// Fallback: opening a transient DB here is not currently a use case (all
|
|
255
|
-
// callers pass a handle), so degrade to empty rather than reopening.
|
|
256
|
-
return [];
|
|
257
|
-
}
|
|
258
|
-
// Confirm the target file has a graph row; without it there is nothing to
|
|
259
|
-
// relate. (Identity is file_path within the stash — one row per path.)
|
|
260
|
-
const row = db
|
|
261
|
-
.prepare("SELECT 1 AS present FROM graph_files WHERE stash_root = ? AND file_path = ? LIMIT 1")
|
|
262
|
-
.get(stashRoot, filePath);
|
|
263
|
-
if (row === undefined)
|
|
264
|
-
return [];
|
|
265
|
-
const effectiveLimit = Math.max(1, limit);
|
|
266
|
-
// Shared-entity count per candidate file_path. The target's entities are the
|
|
267
|
-
// rows for `filePath`; candidates are any OTHER file_path in the stash that
|
|
268
|
-
// shares a normalized entity.
|
|
269
|
-
const candidateRows = db
|
|
270
|
-
.prepare(`SELECT gf.file_path AS file_path,
|
|
271
|
-
gf.file_type AS file_type,
|
|
272
|
-
COUNT(*) AS shared
|
|
273
|
-
FROM graph_file_entities target
|
|
274
|
-
JOIN graph_file_entities e
|
|
275
|
-
ON e.stash_root = target.stash_root
|
|
276
|
-
AND e.entity_norm = target.entity_norm
|
|
277
|
-
AND e.file_path != target.file_path
|
|
278
|
-
JOIN graph_files gf
|
|
279
|
-
ON gf.stash_root = e.stash_root
|
|
280
|
-
AND gf.file_path = e.file_path
|
|
281
|
-
AND gf.body_hash = e.body_hash
|
|
282
|
-
WHERE target.file_path = ?
|
|
283
|
-
AND target.stash_root = ?
|
|
284
|
-
GROUP BY gf.file_path
|
|
285
|
-
ORDER BY shared DESC, gf.file_path ASC
|
|
286
|
-
LIMIT ?`)
|
|
287
|
-
.all(filePath, stashRoot, effectiveLimit);
|
|
288
|
-
if (candidateRows.length === 0)
|
|
289
|
-
return [];
|
|
290
|
-
const candidatePaths = candidateRows.map((r) => r.file_path);
|
|
291
|
-
const placeholders = candidatePaths.map(() => "?").join(",");
|
|
292
|
-
// Pull the shared entity names (joined by normalized casing) for display.
|
|
293
|
-
const sharedRows = db
|
|
294
|
-
.prepare(`SELECT e.file_path AS file_path, e.entity AS entity
|
|
295
|
-
FROM graph_file_entities e
|
|
296
|
-
JOIN graph_file_entities target
|
|
297
|
-
ON target.stash_root = e.stash_root
|
|
298
|
-
AND target.entity_norm = e.entity_norm
|
|
299
|
-
WHERE e.file_path IN (${placeholders})
|
|
300
|
-
AND e.stash_root = ?
|
|
301
|
-
AND target.file_path = ?
|
|
302
|
-
AND target.stash_root = ?`)
|
|
303
|
-
.all(...candidatePaths, stashRoot, filePath, stashRoot);
|
|
304
|
-
const sharedByPath = new Map();
|
|
305
|
-
for (const row of sharedRows) {
|
|
306
|
-
let bucket = sharedByPath.get(row.file_path);
|
|
307
|
-
if (!bucket) {
|
|
308
|
-
bucket = new Set();
|
|
309
|
-
sharedByPath.set(row.file_path, bucket);
|
|
310
|
-
}
|
|
311
|
-
bucket.add(row.entity);
|
|
312
|
-
}
|
|
313
|
-
// Relation count for each candidate (relations where either endpoint
|
|
314
|
-
// matches one of the shared entities).
|
|
315
|
-
const relationCountByPath = new Map();
|
|
316
|
-
const relationRows = db
|
|
317
|
-
.prepare(`SELECT file_path, from_entity, to_entity
|
|
318
|
-
FROM graph_file_relations
|
|
319
|
-
WHERE file_path IN (${placeholders})
|
|
320
|
-
AND stash_root = ?`)
|
|
321
|
-
.all(...candidatePaths, stashRoot);
|
|
322
|
-
for (const row of relationRows) {
|
|
323
|
-
const shared = sharedByPath.get(row.file_path);
|
|
324
|
-
if (!shared)
|
|
325
|
-
continue;
|
|
326
|
-
if (shared.has(row.from_entity) || shared.has(row.to_entity)) {
|
|
327
|
-
relationCountByPath.set(row.file_path, (relationCountByPath.get(row.file_path) ?? 0) + 1);
|
|
328
|
-
}
|
|
329
|
-
}
|
|
330
|
-
// This related list is scoped to one source root, so the user-facing ref is
|
|
331
|
-
// the indexed bundle-less conceptId.
|
|
332
|
-
const refByPath = new Map();
|
|
333
|
-
try {
|
|
334
|
-
const entryRows = db
|
|
335
|
-
.prepare(`SELECT file_path, concept_id FROM entries
|
|
336
|
-
WHERE file_path IN (${placeholders})`)
|
|
337
|
-
.all(...candidatePaths);
|
|
338
|
-
for (const row of entryRows) {
|
|
339
|
-
refByPath.set(row.file_path, row.concept_id);
|
|
340
|
-
}
|
|
341
|
-
}
|
|
342
|
-
catch {
|
|
343
|
-
/* ignore — refs are best-effort */
|
|
344
|
-
}
|
|
345
|
-
return candidateRows.map((row) => {
|
|
346
|
-
const sharedSet = sharedByPath.get(row.file_path) ?? new Set();
|
|
347
|
-
const sharedEntities = [...sharedSet].sort((a, b) => a.localeCompare(b));
|
|
348
|
-
const ref = refByPath.get(row.file_path);
|
|
349
|
-
return {
|
|
350
|
-
...(ref ? { ref } : {}),
|
|
351
|
-
path: row.file_path,
|
|
352
|
-
type: row.file_type,
|
|
353
|
-
sharedEntities,
|
|
354
|
-
relationCount: relationCountByPath.get(row.file_path) ?? 0,
|
|
355
|
-
};
|
|
356
|
-
});
|
|
357
|
-
}
|
|
358
|
-
/**
|
|
359
|
-
* Load and normalize graph data from SQLite once, then reuse it across all
|
|
360
|
-
* per-entry boost lookups in the current query.
|
|
361
|
-
*/
|
|
362
|
-
function readParsedGraphContext(stashRoots, db) {
|
|
363
|
-
const sortedRoots = [...new Set(stashRoots)].sort((a, b) => a.localeCompare(b));
|
|
364
|
-
if (sortedRoots.length === 0)
|
|
365
|
-
return null;
|
|
366
|
-
const metas = sortedRoots
|
|
367
|
-
.map((stashRoot) => loadStoredGraphMeta(stashRoot, db))
|
|
368
|
-
.filter((meta) => meta !== null);
|
|
369
|
-
if (metas.length === 0)
|
|
370
|
-
return null;
|
|
371
|
-
const cacheKey = metas.map((meta) => `${meta.stashPath}\u0000${meta.generatedAt}`).join("\u0001");
|
|
372
|
-
if (cachedParsedGraph && cachedParsedGraph.cacheKey === cacheKey)
|
|
373
|
-
return cachedParsedGraph.context;
|
|
374
|
-
const snapshots = metas
|
|
375
|
-
.map((meta) => loadStoredGraphSnapshot(meta.stashPath, db))
|
|
376
|
-
.filter((snapshot) => snapshot !== null);
|
|
377
|
-
if (snapshots.length === 0)
|
|
378
|
-
return null;
|
|
379
|
-
const graph = {
|
|
380
|
-
schemaVersion: Math.max(...snapshots.map((snapshot) => snapshot.schemaVersion)),
|
|
381
|
-
generatedAt: snapshots
|
|
382
|
-
.map((snapshot) => snapshot.generatedAt)
|
|
383
|
-
.sort()
|
|
384
|
-
.at(-1) ?? new Date(0).toISOString(),
|
|
385
|
-
stashRoot: snapshots[0]?.stashPath ?? "",
|
|
386
|
-
files: snapshots.flatMap((snapshot) => snapshot.files),
|
|
387
|
-
entities: [...new Set(snapshots.flatMap((snapshot) => snapshot.entities))],
|
|
388
|
-
relations: snapshots.flatMap((snapshot) => snapshot.relations),
|
|
389
|
-
};
|
|
390
|
-
const nodesByPath = new Map();
|
|
391
|
-
const entityConfidence = new Map();
|
|
392
|
-
const adjacency = new Map();
|
|
393
|
-
function setBestEntityConfidence(entity, confidence) {
|
|
394
|
-
const normalized = normalizeConfidence(confidence);
|
|
395
|
-
if (normalized === undefined)
|
|
396
|
-
return;
|
|
397
|
-
const current = entityConfidence.get(entity);
|
|
398
|
-
if (current === undefined || normalized > current)
|
|
399
|
-
entityConfidence.set(entity, normalized);
|
|
400
|
-
}
|
|
401
|
-
function setBestEdgeConfidence(from, to, confidence) {
|
|
402
|
-
const normalized = normalizeConfidence(confidence);
|
|
403
|
-
if (!adjacency.has(from))
|
|
404
|
-
adjacency.set(from, new Map());
|
|
405
|
-
const neighbors = adjacency.get(from);
|
|
406
|
-
if (!neighbors)
|
|
407
|
-
return;
|
|
408
|
-
const current = neighbors.get(to);
|
|
409
|
-
const next = normalized ?? 1;
|
|
410
|
-
if (current === undefined || next > current)
|
|
411
|
-
neighbors.set(to, next);
|
|
412
|
-
}
|
|
413
|
-
for (const node of graph.files) {
|
|
414
|
-
nodesByPath.set(node.path, node);
|
|
415
|
-
for (const entity of node.entities) {
|
|
416
|
-
setBestEntityConfidence(entity, node.confidence);
|
|
417
|
-
}
|
|
418
|
-
for (const rel of node.relations) {
|
|
419
|
-
const edgeConfidence = combineConfidence(node.confidence, rel.confidence);
|
|
420
|
-
setBestEdgeConfidence(rel.from, rel.to, edgeConfidence);
|
|
421
|
-
setBestEdgeConfidence(rel.to, rel.from, edgeConfidence);
|
|
422
|
-
}
|
|
423
|
-
}
|
|
424
|
-
const context = { graph, nodesByPath, entityConfidence, adjacency };
|
|
425
|
-
cachedParsedGraph = { cacheKey, context };
|
|
426
|
-
return context;
|
|
427
|
-
}
|
|
@@ -1,95 +0,0 @@
|
|
|
1
|
-
// This Source Code Form is subject to the terms of the Mozilla Public
|
|
2
|
-
// License, v. 2.0. If a copy of the MPL was not distributed with this
|
|
3
|
-
// file, You can obtain one at https://mozilla.org/MPL/2.0/.
|
|
4
|
-
function normalizeRelationType(raw) {
|
|
5
|
-
const normalized = raw?.trim().toLowerCase().replace(/\s+/g, " ") ?? "";
|
|
6
|
-
if (!normalized)
|
|
7
|
-
return "";
|
|
8
|
-
if (normalized === "use" || normalized === "utilizes")
|
|
9
|
-
return "uses";
|
|
10
|
-
if (normalized === "depend on" || normalized === "depends")
|
|
11
|
-
return "depends on";
|
|
12
|
-
if (normalized === "integrates" || normalized === "integration with")
|
|
13
|
-
return "integrates with";
|
|
14
|
-
return normalized;
|
|
15
|
-
}
|
|
16
|
-
function normalizeConfidence(raw) {
|
|
17
|
-
if (typeof raw !== "number" || !Number.isFinite(raw))
|
|
18
|
-
return undefined;
|
|
19
|
-
return Math.max(0, Math.min(1, raw));
|
|
20
|
-
}
|
|
21
|
-
/**
|
|
22
|
-
* Merge and deduplicate entities and relations from multiple per-asset
|
|
23
|
-
* GraphExtraction results into one canonical graph.
|
|
24
|
-
*
|
|
25
|
-
* Entities are keyed on their lowercased, trimmed form. The first-seen
|
|
26
|
-
* casing is preserved as canonical. Relations are keyed on
|
|
27
|
-
* `(from, to, type)` (all lowercased). Dangling relations — those whose
|
|
28
|
-
* `from` or `to` is absent from the deduplicated entity set — are dropped.
|
|
29
|
-
*/
|
|
30
|
-
export function deduplicateGraph(extractions, assetRefs) {
|
|
31
|
-
const entityCanonical = new Map();
|
|
32
|
-
const entitySources = new Map();
|
|
33
|
-
for (let i = 0; i < extractions.length; i++) {
|
|
34
|
-
const ref = assetRefs?.[i] ?? "unknown";
|
|
35
|
-
for (const raw of extractions[i].entities) {
|
|
36
|
-
const trimmed = raw.trim();
|
|
37
|
-
if (!trimmed)
|
|
38
|
-
continue;
|
|
39
|
-
const normalized = trimmed.toLowerCase();
|
|
40
|
-
if (!entityCanonical.has(normalized)) {
|
|
41
|
-
entityCanonical.set(normalized, trimmed);
|
|
42
|
-
entitySources.set(normalized, [ref]);
|
|
43
|
-
}
|
|
44
|
-
else {
|
|
45
|
-
const srcs = entitySources.get(normalized);
|
|
46
|
-
if (srcs && !srcs.includes(ref))
|
|
47
|
-
srcs.push(ref);
|
|
48
|
-
}
|
|
49
|
-
}
|
|
50
|
-
}
|
|
51
|
-
const entities = Array.from(entityCanonical.values());
|
|
52
|
-
const entityNormSet = new Set(entityCanonical.keys());
|
|
53
|
-
const relSeenKey = new Map();
|
|
54
|
-
const relationIndexByKey = new Map();
|
|
55
|
-
const relations = [];
|
|
56
|
-
for (let i = 0; i < extractions.length; i++) {
|
|
57
|
-
const ref = assetRefs?.[i] ?? "unknown";
|
|
58
|
-
for (const rel of extractions[i].relations) {
|
|
59
|
-
const fromNorm = rel.from.trim().toLowerCase();
|
|
60
|
-
const toNorm = rel.to.trim().toLowerCase();
|
|
61
|
-
const typeNorm = normalizeRelationType(rel.type);
|
|
62
|
-
if (!entityNormSet.has(fromNorm) || !entityNormSet.has(toNorm))
|
|
63
|
-
continue;
|
|
64
|
-
const key = `${fromNorm}\0${toNorm}\0${typeNorm}`;
|
|
65
|
-
if (!relSeenKey.has(key)) {
|
|
66
|
-
relSeenKey.set(key, [ref]);
|
|
67
|
-
const canonical = {
|
|
68
|
-
from: entityCanonical.get(fromNorm) ?? rel.from,
|
|
69
|
-
to: entityCanonical.get(toNorm) ?? rel.to,
|
|
70
|
-
};
|
|
71
|
-
if (typeNorm)
|
|
72
|
-
canonical.type = typeNorm;
|
|
73
|
-
const confidence = normalizeConfidence(rel.confidence);
|
|
74
|
-
if (confidence !== undefined)
|
|
75
|
-
canonical.confidence = confidence;
|
|
76
|
-
relationIndexByKey.set(key, relations.length);
|
|
77
|
-
relations.push(canonical);
|
|
78
|
-
}
|
|
79
|
-
else {
|
|
80
|
-
const srcs = relSeenKey.get(key);
|
|
81
|
-
if (srcs && !srcs.includes(ref))
|
|
82
|
-
srcs.push(ref);
|
|
83
|
-
const idx = relationIndexByKey.get(key);
|
|
84
|
-
const nextConfidence = normalizeConfidence(rel.confidence);
|
|
85
|
-
if (idx !== undefined && nextConfidence !== undefined) {
|
|
86
|
-
const current = normalizeConfidence(relations[idx]?.confidence) ?? 0;
|
|
87
|
-
if (nextConfidence > current && relations[idx])
|
|
88
|
-
relations[idx].confidence = nextConfidence;
|
|
89
|
-
}
|
|
90
|
-
}
|
|
91
|
-
}
|
|
92
|
-
}
|
|
93
|
-
const relationSources = new Map(relSeenKey);
|
|
94
|
-
return { entities, relations, entitySources, relationSources };
|
|
95
|
-
}
|
|
@@ -1,35 +0,0 @@
|
|
|
1
|
-
// This Source Code Form is subject to the terms of the Mozilla Public
|
|
2
|
-
// License, v. 2.0. If a copy of the MPL was not distributed with this
|
|
3
|
-
// file, You can obtain one at https://mozilla.org/MPL/2.0/.
|
|
4
|
-
import { buildLexicalQueryPlan } from "./fts-query.js";
|
|
5
|
-
/**
|
|
6
|
-
* Tokenize a display name through the same Unicode-aware lexical planner as a
|
|
7
|
-
* query. Ranking must not invent a second, punctuation-dependent name grammar.
|
|
8
|
-
*/
|
|
9
|
-
export function lexicalNameTokens(name) {
|
|
10
|
-
return buildLexicalQueryPlan(name).tokens.map((token) => token.toLowerCase());
|
|
11
|
-
}
|
|
12
|
-
/**
|
|
13
|
-
* Structural name-token evidence. Exact tokens are always meaningful; fuzzy
|
|
14
|
-
* prefix evidence requires three code points on both sides so a question's
|
|
15
|
-
* short words or digits cannot match an opaque generated storage name.
|
|
16
|
-
*/
|
|
17
|
-
export function structuralNameTokenMatch(left, right) {
|
|
18
|
-
return (left === right ||
|
|
19
|
-
(Math.min([...left].length, [...right].length) >= 3 && (left.startsWith(right) || right.startsWith(left))));
|
|
20
|
-
}
|
|
21
|
-
/**
|
|
22
|
-
* Match a query phrase only across complete, contiguous name tokens. This is
|
|
23
|
-
* deliberately not raw substring matching: `000` must not become name
|
|
24
|
-
* evidence merely because an opaque token happens to be `z9xq000`.
|
|
25
|
-
*/
|
|
26
|
-
export function structuralNamePhraseMatch(nameTokens, queryTokens) {
|
|
27
|
-
if (queryTokens.length === 0 || queryTokens.length > nameTokens.length)
|
|
28
|
-
return false;
|
|
29
|
-
for (let start = 0; start <= nameTokens.length - queryTokens.length; start += 1) {
|
|
30
|
-
if (queryTokens.every((queryToken, index) => structuralNameTokenMatch(nameTokens[start + index], queryToken))) {
|
|
31
|
-
return true;
|
|
32
|
-
}
|
|
33
|
-
}
|
|
34
|
-
return false;
|
|
35
|
-
}
|