akm-cli 0.9.17-alpha.5 → 0.9.17-alpha.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +135 -0
- package/dist/commands/improve/execution.js +5 -5
- package/dist/commands/improve/improve-strategies.js +3 -0
- package/dist/commands/improve/loop-stages.js +5 -4
- package/dist/commands/read/curate.js +31 -49
- package/dist/commands/read/show.js +2 -81
- package/dist/core/config/config.js +1 -1
- package/dist/core/config/schema/improve-processes.js +4 -3
- package/dist/core/config/schema/index-config.js +4 -23
- package/dist/indexer/db/graph-db.js +8 -81
- package/dist/indexer/graph/graph-extraction.js +74 -227
- package/dist/indexer/graph/graph-related.js +5 -4
- package/dist/indexer/search/db-search.js +10 -1
- package/dist/llm/feature-gate.js +0 -3
- package/dist/llm/graph-extract.js +81 -41
- package/dist/scripts/akm-migrate-node.js +4781 -4796
- package/dist/scripts/akm-migrate.js +4781 -4796
- package/dist/storage/repositories/index-entries-repository.js +5 -7
- package/dist/storage/repositories/index-schema.js +16 -34
- package/docs/reference/cli.md +9 -1
- package/docs/reference/configuration.md +12 -0
- package/package.json +1 -1
- package/schemas/akm-config.json +0 -6
package/CHANGELOG.md
CHANGED
|
@@ -6,6 +6,141 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/).
|
|
|
6
6
|
|
|
7
7
|
## [Unreleased]
|
|
8
8
|
|
|
9
|
+
## [0.9.17-alpha.6] - 2026-09-27
|
|
10
|
+
|
|
11
|
+
Graph extraction stops losing and wasting work. A timed-out extraction is
|
|
12
|
+
retried instead of cached as empty. Long documents are extracted once, and
|
|
13
|
+
per-file calls respect the run's concurrency. `akm improve` honors
|
|
14
|
+
`index.graph`. `akm curate` returns nothing for harness and tool envelopes,
|
|
15
|
+
and search and curate show identical content once. Lazy graph extraction,
|
|
16
|
+
which never ran under Bun, is removed.
|
|
17
|
+
|
|
18
|
+
### Changed
|
|
19
|
+
|
|
20
|
+
- **`akm curate` returns nothing, on purpose, for input that is not a task.**
|
|
21
|
+
A harness or tool envelope (input that starts with an XML-style tag and
|
|
22
|
+
contains a closing tag, such as `<task-notification>…</task-notification>`,
|
|
23
|
+
`<system-reminder>…` or `<cross-session-message …>…`) and the stash README
|
|
24
|
+
line each used to get `--limit` unrelated assets. Every caller of
|
|
25
|
+
`akm curate` (the CLI, the OpenCode plugin, other harnesses) now gets an
|
|
26
|
+
empty `items` list with a `summary` that starts with `Curate abstained` and
|
|
27
|
+
names the reason, and a `tip`. On the retrieval suite curate abstains on 57
|
|
28
|
+
of 60 recorded non-task inputs and on none of the 221 real queries (nor on
|
|
29
|
+
any of 5,725 mined task queries). Length is not a reason to abstain: the
|
|
30
|
+
other 3 are task prompts of 2,431–5,531 characters, and in a judged sample
|
|
31
|
+
of 30 inputs over 2,000 characters the top 5 held a relevant asset for 24
|
|
32
|
+
of them (P@5 0.42, against 0.46 for prompts of 400–2,000 characters).
|
|
33
|
+
(`src/commands/read/curate.ts`)
|
|
34
|
+
- **Search and curate return identical content once.** Of entries whose
|
|
35
|
+
indexed content is identical (the same body saved under another name, as
|
|
36
|
+
both a memory and a knowledge doc, or in another bundle), only the
|
|
37
|
+
highest-ranked is kept, and the next candidate takes the freed slot. On the
|
|
38
|
+
retrieval suite such copies filled 7.5% of curate's top 5. Unique
|
|
39
|
+
precision@5, where a copy of a higher-ranked result earns nothing, rises
|
|
40
|
+
from 0.467 to 0.514 (+0.046, 95% CI [+0.028, +0.067]), and the share of
|
|
41
|
+
top-5 slots that repeat a higher-ranked result falls from 0.131 to 0.055.
|
|
42
|
+
Plain P@5 (0.553 → 0.551) and nDCG@10 stay within noise: they counted each
|
|
43
|
+
copy as another relevant result. Latency is unchanged.
|
|
44
|
+
(`src/indexer/search/db-search.ts`)
|
|
45
|
+
|
|
46
|
+
### Removed
|
|
47
|
+
|
|
48
|
+
- **The unused `utility_scores_scoped` index table is gone.** It shipped in
|
|
49
|
+
0.9.17-alpha.5 for per-project scoped utility scores, but no code ever read
|
|
50
|
+
or wrote a row. An index database drops it on its next writable open, the
|
|
51
|
+
same way other retired derived tables are dropped, with no layout-version
|
|
52
|
+
change. (`src/storage/repositories/index-schema.ts`)
|
|
53
|
+
- **Lazy graph extraction in `akm show` and `akm curate`.** With
|
|
54
|
+
`index.graph.lazyGraphExtraction: true`, `show` extracted an asset's graph
|
|
55
|
+
after building its response, so only the next `show` saw it. `curate`
|
|
56
|
+
queued assets for a later pass, which drained only the working bundle's
|
|
57
|
+
queue, and extractions made this way wrote no cache entry. Under Bun neither
|
|
58
|
+
path ever ran: the "already has a graph" check read a missing row as
|
|
59
|
+
present. Graph extraction now runs only in `akm improve`. The
|
|
60
|
+
`graph_extraction_queue` table is dropped the next time the index is opened
|
|
61
|
+
for writing. A config that still sets the key loads, and the key is named
|
|
62
|
+
once as unknown. (`src/commands/read/show.ts`,
|
|
63
|
+
`src/commands/read/curate.ts`, `src/indexer/graph/graph-extraction.ts`,
|
|
64
|
+
`src/storage/repositories/index-schema.ts`)
|
|
65
|
+
|
|
66
|
+
### Fixed
|
|
67
|
+
|
|
68
|
+
- **`index.metadataEnhance`'s default is no longer contradicted by dead
|
|
69
|
+
code.** Metadata enhancement has always defaulted to off
|
|
70
|
+
(`isLlmFeatureEnabled`); a second, unreachable code path in
|
|
71
|
+
`isProcessEnabled` claimed the opposite default and had no caller. Removed,
|
|
72
|
+
so one default remains. (`src/llm/feature-gate.ts`)
|
|
73
|
+
- **Eval tooling and docs catch up to the current config and index shape.**
|
|
74
|
+
`scripts/akm-eval/src/curate-bench.ts` wrote the retired `sources` config
|
|
75
|
+
key and called a nonexistent `akm index --dir`; it now seeds its sandbox
|
|
76
|
+
the same way the other akm-eval scripts and integration tests do, and
|
|
77
|
+
drops `--dir`. The graph A/B ablation harness
|
|
78
|
+
(`scripts/akm-eval/src/graph-ablation.ts`) planted its "graph off" config
|
|
79
|
+
where the sandboxed `akm` never read it, with config keys that didn't gate
|
|
80
|
+
anything (one of them a type error); it now writes
|
|
81
|
+
`index.graph.enabled: false` to the sandbox's actual `AKM_CONFIG_DIR`.
|
|
82
|
+
Updated `scripts/akm-eval/README.md` and `docs/maintainers/eval.md` to
|
|
83
|
+
match, and corrected stale `docs/architecture/architecture.md` references
|
|
84
|
+
to `db-backup`, `staleness-detect`, and `src/commands/graph/`.
|
|
85
|
+
- **Scheduled graph extraction reads `index.graph`.** `akm improve` passed
|
|
86
|
+
graph extraction a batch size of 4 and the `memory` and `knowledge` types
|
|
87
|
+
whenever the strategy's `processes.graphExtraction` did not set them, so
|
|
88
|
+
`index.graph.graphExtractionBatchSize` and `graphExtractionIncludeTypes`
|
|
89
|
+
never applied. It did not read `index.graph`'s `engine`, `model`,
|
|
90
|
+
`timeoutMs` or `llm` either, so a setting such as
|
|
91
|
+
`index.graph.llm.enableThinking: false` had no effect on improve runs. A
|
|
92
|
+
value in the strategy's `processes.graphExtraction` still wins. A setting it
|
|
93
|
+
leaves unset now comes from `index.graph`, then from the built-in default.
|
|
94
|
+
Where `index.graph` asks for something improve did not use before, the
|
|
95
|
+
extractor changes and cached extractions stop applying, so those files are
|
|
96
|
+
extracted again. (`src/commands/improve/loop-stages.ts`,
|
|
97
|
+
`src/commands/improve/execution.ts`,
|
|
98
|
+
`src/commands/improve/improve-strategies.ts`)
|
|
99
|
+
- **A graph extraction that times out is retried, not cached as empty.** A
|
|
100
|
+
call that ran past the engine's `timeoutMs` was recorded as "no entities"
|
|
101
|
+
and cached, so the file was never extracted again. It is now recorded as
|
|
102
|
+
failed, and the next run retries it; timeouts also count toward the run's
|
|
103
|
+
failure-rate abort. A batch that times out fails its files without then
|
|
104
|
+
calling the model once per file. An empty response is likewise recorded as
|
|
105
|
+
failed. (`src/llm/graph-extract.ts`)
|
|
106
|
+
- **Long bodies are extracted once after batching turns itself off.** Two
|
|
107
|
+
non-array batch responses turn batching off for the rest of a run. From
|
|
108
|
+
then on, a body over 1,600 characters was extracted on its own and then
|
|
109
|
+
again with the rest of its batch. Each body is now extracted once.
|
|
110
|
+
(`src/llm/graph-extract.ts`)
|
|
111
|
+
- **A batch's per-file calls respect the run's concurrency.** When a batch
|
|
112
|
+
fell back to one call per file (long bodies, a non-array response, batching
|
|
113
|
+
turned off), those calls all went out at once, up to the batch size. Local
|
|
114
|
+
endpoints serve one or two requests at a time. The calls now run within the
|
|
115
|
+
limit the run applies to its batches, one at a time by default.
|
|
116
|
+
(`src/llm/graph-extract.ts`)
|
|
117
|
+
- **`related` counts a shared entity once.** `akm show`'s `related` list,
|
|
118
|
+
and curate's support refs taken from it, ranked files by the number of
|
|
119
|
+
matching entity rows. A file holding two case forms of one entity, as rows
|
|
120
|
+
from older extractors can, counted it twice and could outrank a file that
|
|
121
|
+
shared two entities. `related` now counts distinct entities. Extraction also
|
|
122
|
+
keeps one form of each entity before writing. The stored key `related`
|
|
123
|
+
matches on is now the one extraction deduplicates on, which also drops
|
|
124
|
+
surrounding quotes and backticks.
|
|
125
|
+
(`src/indexer/graph/graph-related.ts`,
|
|
126
|
+
`src/indexer/graph/graph-extraction.ts`, `src/indexer/db/graph-db.ts`)
|
|
127
|
+
- **A config change that re-extracts the graph says so.** Cached graph
|
|
128
|
+
extractions are keyed by extractor: model, batch size, included asset types
|
|
129
|
+
and prompt version. Changing any of them made every cached file extract
|
|
130
|
+
again without a word. The first run after such a change now warns once,
|
|
131
|
+
naming the change and the number of cached files it will extract again, and
|
|
132
|
+
records the warning in the run's result.
|
|
133
|
+
(`src/indexer/graph/graph-extraction.ts`)
|
|
134
|
+
- **Graph extraction reports what its parser filtered.** A run's graph
|
|
135
|
+
telemetry, part of `akm improve`'s result, now carries
|
|
136
|
+
`filteredGenericEntities`, `filteredInvalidRelations`,
|
|
137
|
+
`filteredLowConfidenceRelations` and `contextBatchRetries`. The pass
|
|
138
|
+
computed them and dropped them, and did not count batch responses at all.
|
|
139
|
+
(`src/indexer/graph/graph-extraction.ts`, `src/llm/graph-extract.ts`)
|
|
140
|
+
- **An unknown key under `index.<pass>` is kept and named once.** It was
|
|
141
|
+
dropped from the loaded config and named twice. It is now handled like an
|
|
142
|
+
unknown key anywhere else in config. (`src/core/config/schema/index-config.ts`)
|
|
143
|
+
|
|
9
144
|
## [0.9.17-alpha.5] - 2026-09-27
|
|
10
145
|
|
|
11
146
|
`akm show` works again for a memory that has a `.derived.md` child (835 of them
|
|
@@ -20,19 +20,19 @@ function mergeDefaults(farther, nearer) {
|
|
|
20
20
|
return deepMergeConfig(farther, nearer);
|
|
21
21
|
}
|
|
22
22
|
/**
|
|
23
|
-
* Resolve improve-owned model work through the canonical execution cascade
|
|
24
|
-
*
|
|
25
|
-
* defaults.llmEngine -> strategy -> process -> current invocation.
|
|
23
|
+
* Resolve improve-owned model work through the canonical execution cascade:
|
|
24
|
+
* defaults.llmEngine -> strategy -> index.<pass> -> process -> current invocation.
|
|
26
25
|
*/
|
|
27
26
|
export function resolveImproveExecution(options) {
|
|
28
27
|
const defaultEngine = options.config.defaults?.llmEngine;
|
|
29
28
|
const profileDefaults = cascadeDefaults(options.profile);
|
|
29
|
+
const indexDefaults = cascadeDefaults(options.index);
|
|
30
30
|
const processDefaults = cascadeDefaults(options.process);
|
|
31
31
|
const currentDefaults = cascadeDefaults(options.current);
|
|
32
|
-
const selectedEngine = currentDefaults.engine ?? processDefaults.engine ?? profileDefaults.engine ?? defaultEngine;
|
|
32
|
+
const selectedEngine = currentDefaults.engine ?? processDefaults.engine ?? indexDefaults.engine ?? profileDefaults.engine ?? defaultEngine;
|
|
33
33
|
if (selectedEngine === undefined || selectedEngine === null)
|
|
34
34
|
return null;
|
|
35
|
-
const invocationDefaults = mergeDefaults(defaultEngine ? { engine: defaultEngine } : {}, profileDefaults);
|
|
35
|
+
const invocationDefaults = mergeDefaults(mergeDefaults(defaultEngine ? { engine: defaultEngine } : {}, profileDefaults), indexDefaults);
|
|
36
36
|
const current = mergeDefaults(processDefaults, currentDefaults);
|
|
37
37
|
const prepared = resolveExecution({
|
|
38
38
|
content: `improve ${options.processName} execution selection`,
|
|
@@ -10,6 +10,7 @@ import quick from "../../assets/improve-strategies/quick.json" with { type: "jso
|
|
|
10
10
|
import reflectDistill from "../../assets/improve-strategies/reflect-distill.json" with { type: "json" };
|
|
11
11
|
import thorough from "../../assets/improve-strategies/thorough.json" with { type: "json" };
|
|
12
12
|
import { conceptIdFromTypeName, parseRefInput } from "../../core/asset/resolve-ref.js";
|
|
13
|
+
import { getIndexPassConfig, } from "../../core/config/config.js";
|
|
13
14
|
import { ImproveProfileConfigSchema } from "../../core/config/config-schema.js";
|
|
14
15
|
import { deepMergeConfig } from "../../core/config/deep-merge.js";
|
|
15
16
|
import { BUILTIN_IMPROVE_STRATEGY_NAMES, IMPROVE_PROCESS_ENGINE_CAPABILITIES, } from "../../core/config/engine-semantics.js";
|
|
@@ -211,6 +212,8 @@ function buildImprovePlan(strategy, config, options) {
|
|
|
211
212
|
profile: strategy.config,
|
|
212
213
|
process: sourceProcessConfig,
|
|
213
214
|
processName,
|
|
215
|
+
// Graph extraction's standing engine, model, timeout and llm settings (GR-D15).
|
|
216
|
+
...(processName === "graphExtraction" ? { index: getIndexPassConfig(config.index, "graph") } : {}),
|
|
214
217
|
});
|
|
215
218
|
runner = resolved?.runner ?? null;
|
|
216
219
|
notices = resolved?.notices ?? [];
|
|
@@ -6,14 +6,14 @@ import fs from "node:fs";
|
|
|
6
6
|
import path from "node:path";
|
|
7
7
|
import { parseRefInput } from "../../core/asset/resolve-ref.js";
|
|
8
8
|
import { daysToMs } from "../../core/common.js";
|
|
9
|
-
import {
|
|
9
|
+
import { loadConfig } from "../../core/config/config.js";
|
|
10
10
|
import { UsageError } from "../../core/errors.js";
|
|
11
11
|
import { appendEvent } from "../../core/events.js";
|
|
12
12
|
import { openLogsDatabase, purgeOldTaskLogs } from "../../core/logs-db.js";
|
|
13
13
|
import { getDbPath, getTaskLogDir } from "../../core/paths.js";
|
|
14
14
|
import { withStateDb } from "../../core/state-db.js";
|
|
15
15
|
import { info } from "../../core/warn.js";
|
|
16
|
-
import {
|
|
16
|
+
import { runGraphExtractionPass } from "../../indexer/graph/graph-extraction.js";
|
|
17
17
|
import { indexWrittenAssets } from "../../indexer/index-written-assets.js";
|
|
18
18
|
import { deriveWritableBundleIds } from "../../indexer/installations.js";
|
|
19
19
|
import { collectPendingMemories, runMemoryInferencePass, } from "../../indexer/passes/memory-inference.js";
|
|
@@ -514,8 +514,9 @@ export async function runGraphExtractionMaintenancePass(ctx, dbCell, args) {
|
|
|
514
514
|
},
|
|
515
515
|
options: {
|
|
516
516
|
candidatePaths,
|
|
517
|
-
|
|
518
|
-
|
|
517
|
+
// Only what the strategy sets: the pass falls back to index.graph, then its defaults (GR-D15).
|
|
518
|
+
...(settings?.includeTypes ? { includeTypes: settings.includeTypes } : {}),
|
|
519
|
+
...(settings?.batchSize != null ? { batchSize: settings.batchSize } : {}),
|
|
519
520
|
...(settings?.topN != null ? { topN: settings.topN } : {}),
|
|
520
521
|
...(settings?.maxChunksPerAsset != null ? { maxChunksPerAsset: settings.maxChunksPerAsset } : {}),
|
|
521
522
|
},
|
|
@@ -15,17 +15,13 @@
|
|
|
15
15
|
* The exported `akmCurate()` API is the single entry point; tests can also
|
|
16
16
|
* drive `curateSearchResults` with a fixture search response.
|
|
17
17
|
*/
|
|
18
|
-
import
|
|
19
|
-
import { parseFrontmatter } from "../../core/asset/frontmatter.js";
|
|
20
|
-
import { getIndexPassConfig, loadConfig } from "../../core/config/config.js";
|
|
18
|
+
import { loadConfig } from "../../core/config/config.js";
|
|
21
19
|
import { rethrowIfTestIsolationError, UsageError } from "../../core/errors.js";
|
|
22
20
|
import { appendEvent } from "../../core/events.js";
|
|
23
21
|
import { redactCredentialPatterns } from "../../core/redaction.js";
|
|
24
22
|
import { withStateDbTelemetry } from "../../core/state-db.js";
|
|
25
|
-
import { enqueueGraphExtraction, hasGraphData } from "../../indexer/db/graph-db.js";
|
|
26
23
|
import { searchHitContent } from "../../indexer/search/db-search.js";
|
|
27
24
|
import { copySearchHitAttribution, getSearchHitAttribution, usageEventAttributionMetadata, } from "../../indexer/search/search-attribution.js";
|
|
28
|
-
import { findSourceForPath, resolveSourceEntries } from "../../indexer/search/search-source.js";
|
|
29
25
|
import { insertUsageEvent } from "../../indexer/usage/usage-events.js";
|
|
30
26
|
import { estimateTokenCount } from "../../llm/embedders/remote.js";
|
|
31
27
|
import { isLlmFeatureEnabled, tryLlmFeature } from "../../llm/feature-gate.js";
|
|
@@ -33,11 +29,12 @@ import { rerankDocuments } from "../../llm/rerank-client.js";
|
|
|
33
29
|
import { truncateDescription } from "../../output/shapes/helpers.js";
|
|
34
30
|
import { TELEMETRY_BUSY_TIMEOUT_MS, withIndexDb } from "../../storage/repositories/index-db.js";
|
|
35
31
|
import { findEntryIdByRef, getItemRefById } from "../../storage/repositories/index-entries-repository.js";
|
|
36
|
-
import { computeBodyHash } from "../../storage/repositories/index-llm-cache-repository.js";
|
|
37
32
|
import { akmSearch, parseSearchSource } from "./search.js";
|
|
38
33
|
import { akmShowUnified } from "./show.js";
|
|
39
34
|
const DEFAULT_CURATE_LIMIT = 4;
|
|
40
35
|
const MAX_CURATE_SUPPORT_REFS = 2;
|
|
36
|
+
/** The line of `src/assets/stash-skeleton/README.md` that reaches curate verbatim as a query. */
|
|
37
|
+
const STASH_README_LINE = "This is an **AKM stash** — a structured knowledge repository that stores reusable";
|
|
41
38
|
/** Fused candidates the reranker reorders when `search.curateRerank.topN` is unset. */
|
|
42
39
|
const DEFAULT_CURATE_RERANK_TOP_N = 30;
|
|
43
40
|
/** Characters of name, description and content sent to the reranker per candidate. */
|
|
@@ -104,6 +101,19 @@ export async function akmCurate(options) {
|
|
|
104
101
|
if (!trimmedQuery) {
|
|
105
102
|
throw new UsageError('A curation query is required. Usage: akm curate "<task or prompt>" [--type <type>] [--limit <n>]', "MISSING_REQUIRED_ARGUMENT");
|
|
106
103
|
}
|
|
104
|
+
const nonTask = nonTaskInput(trimmedQuery);
|
|
105
|
+
if (nonTask) {
|
|
106
|
+
const abstained = {
|
|
107
|
+
query: options.query,
|
|
108
|
+
summary: `Curate abstained: the input is ${nonTask}, not a task.`,
|
|
109
|
+
items: [],
|
|
110
|
+
tip: 'Nothing was selected on purpose. To curate for it, pass the task itself: akm curate "<what you are trying to do>".',
|
|
111
|
+
};
|
|
112
|
+
if (!options.skipLogging) {
|
|
113
|
+
logCurateEvent(options.query, abstained, options.eventSource, options.attributionProjection);
|
|
114
|
+
}
|
|
115
|
+
return abstained;
|
|
116
|
+
}
|
|
107
117
|
const limit = options.limit && options.limit > 0 ? options.limit : DEFAULT_CURATE_LIMIT;
|
|
108
118
|
const source = options.source ?? parseSearchSource("local");
|
|
109
119
|
const searchResponse = options.searchResponse ??
|
|
@@ -121,6 +131,21 @@ export async function akmCurate(options) {
|
|
|
121
131
|
}
|
|
122
132
|
return result;
|
|
123
133
|
}
|
|
134
|
+
/**
|
|
135
|
+
* What the (trimmed) curate input is when it is not a task, else undefined.
|
|
136
|
+
* Harness and tool envelopes (`<task-notification>…`, `<system-reminder>…`,
|
|
137
|
+
* `<cross-session-message …>…`) start with a tag and close one, and the stash
|
|
138
|
+
* README line arrives verbatim; on the retrieval suite neither shape occurs in
|
|
139
|
+
* a real query. Length is not a signal: prompts over 2,000 characters found
|
|
140
|
+
* relevant assets at about the rate of shorter long prompts.
|
|
141
|
+
*/
|
|
142
|
+
function nonTaskInput(query) {
|
|
143
|
+
if (query.startsWith("<") && query.includes("</"))
|
|
144
|
+
return "a harness or tool envelope";
|
|
145
|
+
if (query === STASH_README_LINE)
|
|
146
|
+
return "the akm stash README boilerplate";
|
|
147
|
+
return undefined;
|
|
148
|
+
}
|
|
124
149
|
export async function curateSearchResults(query, result, limit, selectedType, eventSource) {
|
|
125
150
|
const allStashHits = result.hits.filter((hit) => hit.type !== "registry");
|
|
126
151
|
const registryHits = result.registryHits ?? [];
|
|
@@ -201,11 +226,6 @@ async function enrichCuratedStashHit(query, hit, selectedRefs, eventSource) {
|
|
|
201
226
|
catch {
|
|
202
227
|
shown = undefined;
|
|
203
228
|
}
|
|
204
|
-
// #624-P3: when lazy graph extraction is opted in, enqueue an ungraphed
|
|
205
|
-
// asset for a later pass to extract. Fire-and-forget, non-blocking, NO inline
|
|
206
|
-
// extraction and NO LLM call here. Default-off (flag unset) = byte-identical.
|
|
207
|
-
if (shown?.path)
|
|
208
|
-
maybeEnqueueLazyGraph(shown.path);
|
|
209
229
|
const description = shown?.description ?? hit.description;
|
|
210
230
|
const preview = buildCuratedPreview(shown, hit);
|
|
211
231
|
const supportRefs = buildCurateSupportRefs(shown?.related?.hits, selectedRefs, hit.ref);
|
|
@@ -232,44 +252,6 @@ async function enrichCuratedStashHit(query, hit, selectedRefs, eventSource) {
|
|
|
232
252
|
copySearchHitAttribution(hit, item, item.description);
|
|
233
253
|
return item;
|
|
234
254
|
}
|
|
235
|
-
/**
|
|
236
|
-
* #624-P3 — enqueue an ungraphed asset for lazy graph extraction when the
|
|
237
|
-
* `index.graph.lazyGraphExtraction` flag is on. Pure side-effect, fully
|
|
238
|
-
* best-effort: any failure (config, fs, db) is swallowed so curate never fails
|
|
239
|
-
* on it. NO LLM call and NO inline extraction — only a cheap queue insert.
|
|
240
|
-
* Default-off (flag unset) returns immediately = byte-identical behavior.
|
|
241
|
-
*/
|
|
242
|
-
function maybeEnqueueLazyGraph(assetPath) {
|
|
243
|
-
try {
|
|
244
|
-
const config = loadConfig();
|
|
245
|
-
if (getIndexPassConfig(config.index, "graph")?.lazyGraphExtraction !== true)
|
|
246
|
-
return;
|
|
247
|
-
const sources = resolveSourceEntries();
|
|
248
|
-
const source = findSourceForPath(assetPath, sources);
|
|
249
|
-
const stashRoot = source?.path;
|
|
250
|
-
if (!stashRoot)
|
|
251
|
-
return;
|
|
252
|
-
let raw;
|
|
253
|
-
try {
|
|
254
|
-
raw = fs.readFileSync(assetPath, "utf8");
|
|
255
|
-
}
|
|
256
|
-
catch {
|
|
257
|
-
return;
|
|
258
|
-
}
|
|
259
|
-
const body = parseFrontmatter(raw).content.trim();
|
|
260
|
-
if (!body)
|
|
261
|
-
return;
|
|
262
|
-
const bodyHash = computeBodyHash(body);
|
|
263
|
-
withIndexDb((db) => {
|
|
264
|
-
if (!hasGraphData(db, stashRoot, assetPath)) {
|
|
265
|
-
enqueueGraphExtraction(db, stashRoot, assetPath, bodyHash, 0);
|
|
266
|
-
}
|
|
267
|
-
}, { busyTimeoutMs: TELEMETRY_BUSY_TIMEOUT_MS });
|
|
268
|
-
}
|
|
269
|
-
catch (err) {
|
|
270
|
-
rethrowIfTestIsolationError(err);
|
|
271
|
-
}
|
|
272
|
-
}
|
|
273
255
|
function buildCuratedRegistryItem(query, hit) {
|
|
274
256
|
return {
|
|
275
257
|
source: "registry",
|
|
@@ -27,14 +27,12 @@ import { buildMarkdownLeadContext, fragmentForSelector, MARKDOWN_FRAGMENT_CONTEX
|
|
|
27
27
|
import { displayRef, typeNameFromConceptId } from "../../core/asset/resolve-ref.js";
|
|
28
28
|
import { META_DIR, parseMetaRef, readMetaFile } from "../../core/asset/stash-meta.js";
|
|
29
29
|
import { asNonEmptyString, isWithin } from "../../core/common.js";
|
|
30
|
-
import {
|
|
30
|
+
import { loadConfig } from "../../core/config/config.js";
|
|
31
31
|
import { NotFoundError, rethrowIfDataDirUnreadable, rethrowIfTestIsolationError, UsageError } from "../../core/errors.js";
|
|
32
32
|
import { appendEvent } from "../../core/events.js";
|
|
33
33
|
import { SCRIPT_EXTENSIONS } from "../../core/recognition-util.js";
|
|
34
34
|
import { presentationFor } from "../../core/type-presentation.js";
|
|
35
35
|
import { warn, warnOnce } from "../../core/warn.js";
|
|
36
|
-
import { hasGraphData } from "../../indexer/db/graph-db.js";
|
|
37
|
-
import { extractGraphForSingleFile } from "../../indexer/graph/graph-extraction.js";
|
|
38
36
|
import { listRelatedPathsForFile } from "../../indexer/graph/graph-related.js";
|
|
39
37
|
import { lookupBundleRef, lookupBundleRefWithResolution } from "../../indexer/indexer.js";
|
|
40
38
|
import { projectMarkdownFragmentContent } from "../../indexer/passes/metadata.js";
|
|
@@ -42,11 +40,8 @@ import { ensurePrimaryIndexForRead, resolveReadSources } from "../../indexer/rea
|
|
|
42
40
|
import { buildEditHint, findSourceForPath, isEditable, resolveSourceEntries, } from "../../indexer/search/search-source.js";
|
|
43
41
|
import { recentShowCount, recordShowUsage } from "../../indexer/usage/show-usage.js";
|
|
44
42
|
import { buildFileContext, buildRenderContext, getRenderer, } from "../../indexer/walk/file-context.js";
|
|
45
|
-
import { resolveIndexPassExecution } from "../../llm/index-passes.js";
|
|
46
43
|
import { resolveSourcesForOrigin } from "../../registry/origin-resolve.js";
|
|
47
|
-
import {
|
|
48
|
-
import { closeDatabase, openExistingDatabase } from "../../storage/repositories/index-connection.js";
|
|
49
|
-
import { TELEMETRY_BUSY_TIMEOUT_MS, withIndexDb } from "../../storage/repositories/index-db.js";
|
|
44
|
+
import { withIndexDb } from "../../storage/repositories/index-db.js";
|
|
50
45
|
import { getIndexedMarkdownFragment } from "../../storage/repositories/index-fts-repository.js";
|
|
51
46
|
import { getCurrentWorkflowScopeKey } from "../../workflows/authoring/scope-key.js";
|
|
52
47
|
import { buildWorkflowAction } from "../../workflows/renderer.js";
|
|
@@ -405,15 +400,6 @@ export async function showLocal(input) {
|
|
|
405
400
|
if (activeRun) {
|
|
406
401
|
fullResponse.activeRun = activeRun;
|
|
407
402
|
}
|
|
408
|
-
// #624-P3: opt-in inline graph extraction. Default OFF — when the flag is
|
|
409
|
-
// unset this whole block is skipped (no hasGraphData check, no LLM call), so
|
|
410
|
-
// behavior is byte-identical to today. When ON, it extracts graph data for an
|
|
411
|
-
// ungraphed asset, but ONLY when a model is configured (model-available
|
|
412
|
-
// guard) and ALWAYS bounded by a 30s timeout so `show` can never hang. Any
|
|
413
|
-
// timeout/model-unavailable/error path returns the response unchanged.
|
|
414
|
-
if (getIndexPassConfig(config.index, "graph")?.lazyGraphExtraction === true) {
|
|
415
|
-
await maybeExtractGraphInline(config, sourceStashDir, assetPath);
|
|
416
|
-
}
|
|
417
403
|
if (input.detail === "brief") {
|
|
418
404
|
return buildBriefResponse(fullResponse, assetPath);
|
|
419
405
|
}
|
|
@@ -470,71 +456,6 @@ function findUnrecognizedScriptSource(assetParts, sources) {
|
|
|
470
456
|
}
|
|
471
457
|
return undefined;
|
|
472
458
|
}
|
|
473
|
-
/**
|
|
474
|
-
* #624-P3 — opt-in inline graph extraction for `akm show`. Best-effort and
|
|
475
|
-
* timeout-bounded: never throws, never hangs, never mutates the response.
|
|
476
|
-
*
|
|
477
|
-
* Preconditions (caller already checked the flag): a model must be configured
|
|
478
|
-
* (model-available guard via {@link resolveIndexPassExecution}) and the asset
|
|
479
|
-
* must be ungraphed ({@link hasGraphData}). Extraction races a 30s timeout so
|
|
480
|
-
* `show` cannot block on a slow provider; any timeout/error/missing-model path
|
|
481
|
-
* is swallowed and `show` returns its already-assembled response unchanged.
|
|
482
|
-
*/
|
|
483
|
-
async function maybeExtractGraphInline(config, sourceStashDir, assetPath) {
|
|
484
|
-
try {
|
|
485
|
-
// Resolve readiness and the symbolic runner once. The inline dispatch must
|
|
486
|
-
// consume this same snapshot even if models.json changes while show runs.
|
|
487
|
-
const graphExecution = resolveIndexPassExecution("graph", config);
|
|
488
|
-
if (!graphExecution.runner)
|
|
489
|
-
return;
|
|
490
|
-
const emittedNoticeKeys = new Set();
|
|
491
|
-
const reportNotices = (notices) => {
|
|
492
|
-
for (const notice of notices) {
|
|
493
|
-
const key = JSON.stringify(notice);
|
|
494
|
-
if (emittedNoticeKeys.has(key))
|
|
495
|
-
continue;
|
|
496
|
-
emittedNoticeKeys.add(key);
|
|
497
|
-
const field = typeof notice.field === "string" ? ` field=${notice.field}` : "";
|
|
498
|
-
warn(`[akm] lazy graph extraction notice ${notice.code} adapter=${notice.adapter}${field}: ${notice.message}`);
|
|
499
|
-
}
|
|
500
|
-
};
|
|
501
|
-
reportNotices(graphExecution.notices);
|
|
502
|
-
let alreadyGraphed = false;
|
|
503
|
-
withIndexDb((db) => {
|
|
504
|
-
alreadyGraphed = hasGraphData(db, sourceStashDir, assetPath);
|
|
505
|
-
}, { busyTimeoutMs: TELEMETRY_BUSY_TIMEOUT_MS });
|
|
506
|
-
if (alreadyGraphed)
|
|
507
|
-
return;
|
|
508
|
-
// Open the db for the async extraction ourselves: `withIndexDb` is
|
|
509
|
-
// synchronous and would close the connection the instant the async fn
|
|
510
|
-
// returns its Promise (before extraction completes). Close it explicitly
|
|
511
|
-
// after the race settles instead.
|
|
512
|
-
const db = openExistingDatabase(resolveStorageLocations().indexDb);
|
|
513
|
-
let timer;
|
|
514
|
-
const timeout = new Promise((resolve) => {
|
|
515
|
-
timer = setTimeout(resolve, 30_000);
|
|
516
|
-
});
|
|
517
|
-
try {
|
|
518
|
-
await Promise.race([
|
|
519
|
-
extractGraphForSingleFile(db, sourceStashDir, assetPath, {
|
|
520
|
-
config,
|
|
521
|
-
llmRunner: graphExecution.runner,
|
|
522
|
-
onNotices: reportNotices,
|
|
523
|
-
}),
|
|
524
|
-
timeout,
|
|
525
|
-
]);
|
|
526
|
-
}
|
|
527
|
-
finally {
|
|
528
|
-
if (timer)
|
|
529
|
-
clearTimeout(timer);
|
|
530
|
-
closeDatabase(db);
|
|
531
|
-
}
|
|
532
|
-
}
|
|
533
|
-
catch (err) {
|
|
534
|
-
rethrowIfTestIsolationError(err);
|
|
535
|
-
// Any other failure: silently return the unchanged show response.
|
|
536
|
-
}
|
|
537
|
-
}
|
|
538
459
|
/**
|
|
539
460
|
* Minimal `show`: ref → indexer lookup → file contents. Used by callers that
|
|
540
461
|
* just need the raw file (e.g. clone, write-source) and don't want the full
|
|
@@ -33,7 +33,7 @@ export { FEEDBACK_FAILURE_MODES } from "./config-schema.js";
|
|
|
33
33
|
* combined prompt size well under common 8K/16K context windows (each body is
|
|
34
34
|
* sliced to ~500 chars in the graph-extract prompt builder).
|
|
35
35
|
*/
|
|
36
|
-
|
|
36
|
+
const DEFAULT_GRAPH_EXTRACTION_BATCH_SIZE = 4;
|
|
37
37
|
/**
|
|
38
38
|
* Approximate character budget per asset body inside a batched
|
|
39
39
|
* graph-extraction prompt — used by {@link resolveBatchSize} to derive a
|
|
@@ -183,9 +183,10 @@ const MEMORY_INFERENCE_PROCESS_FIELDS = {
|
|
|
183
183
|
cls: clsField,
|
|
184
184
|
};
|
|
185
185
|
/**
|
|
186
|
-
* GraphExtraction process fields:
|
|
187
|
-
* batching.
|
|
188
|
-
*
|
|
186
|
+
* GraphExtraction process fields: one strategy's graph extraction scope and
|
|
187
|
+
* batching. `includeTypes` and `batchSize` override `index.graph`'s
|
|
188
|
+
* `graphExtractionIncludeTypes` and `graphExtractionBatchSize`; unset, the
|
|
189
|
+
* pass reads those.
|
|
189
190
|
*/
|
|
190
191
|
const GRAPH_EXTRACTION_PROCESS_FIELDS = {
|
|
191
192
|
// #624 P2: when set, rank eligible files by utility_scores DESC and process
|
|
@@ -32,22 +32,11 @@ const INDEX_PASS_RETIRED_KEYS = new Set([
|
|
|
32
32
|
"maxTokens",
|
|
33
33
|
"capabilities",
|
|
34
34
|
]);
|
|
35
|
-
const INDEX_PASS_KNOWN_KEYS = new Set([
|
|
36
|
-
"engine",
|
|
37
|
-
"model",
|
|
38
|
-
"timeoutMs",
|
|
39
|
-
"enabled",
|
|
40
|
-
"llm",
|
|
41
|
-
"graphExtractionBatchSize",
|
|
42
|
-
"graphExtractionIncludeTypes",
|
|
43
|
-
"lazyGraphExtraction",
|
|
44
|
-
]);
|
|
45
35
|
/**
|
|
46
|
-
* Per-pass `index.<pass>` entry.
|
|
47
|
-
*
|
|
48
|
-
*
|
|
49
|
-
*
|
|
50
|
-
* string` strings — keeps `akm` startup errors actionable.
|
|
36
|
+
* Per-pass `index.<pass>` entry. The preprocess names and drops the retired
|
|
37
|
+
* engine settings above with a targeted message. Any other unknown key is kept
|
|
38
|
+
* and named once by the config loader's schema walk, like an unknown key
|
|
39
|
+
* anywhere else in config.
|
|
51
40
|
*/
|
|
52
41
|
export const IndexPassConfigSchema = z.preprocess((raw, ctx) => {
|
|
53
42
|
if (typeof raw !== "object" || raw === null || Array.isArray(raw)) {
|
|
@@ -62,13 +51,6 @@ export const IndexPassConfigSchema = z.preprocess((raw, ctx) => {
|
|
|
62
51
|
cleaned ??= { ...obj };
|
|
63
52
|
delete cleaned[key];
|
|
64
53
|
}
|
|
65
|
-
else if (!INDEX_PASS_KNOWN_KEYS.has(key)) {
|
|
66
|
-
warnOnce(`index-pass:unknown:${dotted}`, `Unknown key \`${dotted}\` ignored. Per-pass entries support ` +
|
|
67
|
-
"`engine`, `model`, `timeoutMs`, `enabled`, `llm`, `graphExtractionBatchSize`, " +
|
|
68
|
-
"`graphExtractionIncludeTypes`, and `lazyGraphExtraction`.");
|
|
69
|
-
cleaned ??= { ...obj };
|
|
70
|
-
delete cleaned[key];
|
|
71
|
-
}
|
|
72
54
|
}
|
|
73
55
|
return cleaned ?? raw;
|
|
74
56
|
}, z
|
|
@@ -81,7 +63,6 @@ export const IndexPassConfigSchema = z.preprocess((raw, ctx) => {
|
|
|
81
63
|
graphExtractionBatchSize: positiveInt.optional(),
|
|
82
64
|
// Accept-any until Chunk 2 (WI-9.6c) — no longer enum-restricted.
|
|
83
65
|
graphExtractionIncludeTypes: z.array(z.string().min(1)).nonempty().optional(),
|
|
84
|
-
lazyGraphExtraction: z.boolean().optional(),
|
|
85
66
|
})
|
|
86
67
|
.passthrough());
|
|
87
68
|
const MetadataEnhanceSchema = z.object({ enabled: z.boolean().optional() }).passthrough();
|