akm-cli 0.9.16-alpha.1 → 0.9.16
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +56 -132
- package/dist/assets/hints/cli-hints-full.md +13 -6
- package/dist/assets/tasks/core/index-refresh.yml +1 -1
- package/dist/assets/tasks/improve/akm-improve-catchup.yml +3 -6
- package/dist/cli/retired-commands.js +0 -4
- package/dist/cli/unknown-flags.js +3 -36
- package/dist/commands/env/env-binding.js +4 -4
- package/dist/commands/env/env-cli.js +3 -3
- package/dist/commands/improve/collapse-detector.js +2 -2
- package/dist/commands/improve/consolidate.js +4 -6
- package/dist/commands/improve/improve-cli.js +20 -15
- package/dist/commands/improve/reflect.js +23 -2
- package/dist/commands/lint/base-linter.js +9 -0
- package/dist/commands/lint/env-key-rules.js +2 -2
- package/dist/commands/proposal/propose.js +15 -1
- package/dist/commands/proposal/repository.js +3 -12
- package/dist/commands/proposal/validators/proposal-quality-validators.js +40 -3
- package/dist/commands/proposal/validators/proposal-validators.js +5 -4
- package/dist/commands/read/curate.js +44 -34
- package/dist/commands/read/search.js +35 -54
- package/dist/commands/read/show.js +21 -2
- package/dist/commands/registry-cli.js +5 -5
- package/dist/commands/sources/add-cli.js +59 -16
- package/dist/commands/sources/bundle-cli.js +35 -11
- package/dist/commands/sources/bundle-config-ops.js +30 -0
- package/dist/commands/sources/dangerous-env-audit.js +4 -4
- package/dist/commands/sources/info.js +8 -8
- package/dist/commands/sources/installed-stashes.js +55 -61
- package/dist/commands/sources/source-add.js +39 -38
- package/dist/commands/sources/source-manage.js +34 -12
- package/dist/commands/sources/stash-cli.js +111 -119
- package/dist/commands/sources/stash-skeleton.js +6 -3
- package/dist/commands/tasks/explain.js +4 -1
- package/dist/commands/tasks/tasks-cli.js +31 -9
- package/dist/commands/tasks/tasks.js +239 -194
- package/dist/commands/tasks/validate.js +20 -32
- package/dist/core/activation-policy.js +4 -4
- package/dist/core/adapter/adapters/akm-adapter.js +8 -35
- package/dist/core/adapter/adapters/akm-metadata.js +1 -11
- package/dist/core/adapter/execution-source.js +10 -29
- package/dist/core/asset/asset-placement.js +0 -35
- package/dist/core/config/config-schema.js +64 -8
- package/dist/core/config/config-sources.js +96 -2
- package/dist/core/config/config.js +190 -24
- package/dist/core/config/legacy-source-shape-shim.js +9 -0
- package/dist/core/config/schema/embedding.js +30 -7
- package/dist/core/config/schema/execution.js +23 -0
- package/dist/core/config/schema/experimental.js +1 -1
- package/dist/core/config/schema/scheduler.js +20 -0
- package/dist/core/config/schema/search.js +10 -12
- package/dist/core/config/schema/sources-bundles.js +32 -1
- package/dist/core/content-safety.js +52 -0
- package/dist/core/errors.js +2 -5
- package/dist/core/maintenance-barrier.js +11 -13
- package/dist/core/paths.js +11 -0
- package/dist/core/run-lock.js +2 -5
- package/dist/core/state/migrations.js +1 -26
- package/dist/core/state-db.js +27 -63
- package/dist/core/type-presentation.js +1 -1
- package/dist/core/write-source.js +13 -8
- package/dist/indexer/bundle-identity-guard.js +45 -8
- package/dist/indexer/ensure-index.js +0 -5
- package/dist/indexer/index-db-contention.js +56 -0
- package/dist/indexer/index-rebuild-lock.js +73 -0
- package/dist/indexer/index-written-assets.js +171 -133
- package/dist/indexer/indexer.js +1621 -458
- package/dist/indexer/lookup/adapter-concept-owner.js +5 -19
- package/dist/indexer/materialize-embeddings.js +785 -0
- package/dist/indexer/passes/dir-staleness.js +161 -0
- package/dist/indexer/passes/metadata.js +1 -18
- package/dist/indexer/scan/drain-dir.js +70 -27
- package/dist/indexer/search/db-search.js +89 -373
- package/dist/indexer/search/ranking-contributors.js +16 -21
- package/dist/indexer/search/ranking.js +57 -135
- package/dist/indexer/search/search-source.js +29 -11
- package/dist/integrations/agent/execution-lowering.js +3 -2
- package/dist/integrations/agent/execution-preparation.js +32 -1
- package/dist/integrations/agent/prompts.js +1 -1
- package/dist/integrations/agent/request-lowering.js +3 -2
- package/dist/llm/client.js +3 -11
- package/dist/llm/embedder.js +3 -10
- package/dist/llm/embedders/remote.js +104 -133
- package/dist/llm/feature-gate.js +2 -4
- package/dist/llm/rerank-client.js +3 -3
- package/dist/output/html-render.js +2 -1
- package/dist/output/shapes/passthrough.js +2 -1
- package/dist/output/stdout.js +24 -0
- package/dist/output/text/command-format.js +13 -19
- package/dist/output/text/helpers.js +1 -1
- package/dist/output/text/index.js +2 -5
- package/dist/output/text.js +4 -3
- package/dist/registry/resolve.js +37 -10
- package/dist/scripts/akm-migrate-node.js +15197 -11351
- package/dist/scripts/akm-migrate.js +15514 -11668
- package/dist/setup/semantic-assets.js +2 -2
- package/dist/setup/setup.js +3 -3
- package/dist/setup/steps/connection.js +2 -3
- package/dist/setup/steps/tasks.js +29 -36
- package/dist/sources/providers/git-install.js +17 -11
- package/dist/sources/providers/git-provider.js +12 -5
- package/dist/sources/providers/git-stash.js +38 -16
- package/dist/sources/snapshot-fetchers/website-ingest.js +3 -3
- package/dist/storage/repositories/embedding-salvage-repository.js +184 -0
- package/dist/storage/repositories/index-connection.js +3 -1
- package/dist/storage/repositories/index-entries-repository.js +68 -77
- package/dist/storage/repositories/index-entry-schema.js +25 -16
- package/dist/storage/repositories/index-fts-repository.js +263 -29
- package/dist/storage/repositories/index-meta-repository.js +29 -0
- package/dist/storage/repositories/index-schema.js +122 -115
- package/dist/storage/repositories/index-utility-repository.js +1 -1
- package/dist/storage/repositories/index-vec-repository.js +435 -22
- package/dist/tasks/activation-config.js +90 -0
- package/dist/tasks/backends/cron.js +9 -0
- package/dist/tasks/backends/launchd.js +1 -0
- package/dist/tasks/backends/schtasks.js +2 -0
- package/dist/tasks/embedded.js +4 -5
- package/dist/tasks/scheduler-binding.js +2 -2
- package/dist/tasks/scheduler-sync-preview.js +8 -1
- package/dist/tasks/scheduler-sync.js +19 -10
- package/dist/tasks/source/parse-task-source.js +10 -113
- package/dist/tasks/source/project-v4.js +2 -2
- package/dist/tasks/source/task-source-v4.js +4 -12
- package/dist/tasks/source/task-to-v3.js +4 -12
- package/dist/tasks/source/task-to-v4.js +40 -7
- package/docs/migration/README.md +1 -0
- package/docs/migration/release-notes/0.9.15.md +36 -34
- package/docs/migration/release-notes/0.9.16.md +60 -98
- package/docs/migration/release-notes/README.md +0 -5
- package/docs/migration/v0.9.1-to-v0.9.2.md +6 -9
- package/docs/reference/cli.md +124 -122
- package/docs/reference/configuration.md +137 -133
- package/docs/reference/data-and-telemetry.md +1 -2
- package/docs/reference/tasks.md +34 -29
- package/package.json +1 -1
- package/schemas/akm-config.json +170 -6
- package/schemas/akm-task.json +1 -2
- package/dist/commands/sources/index-status.js +0 -99
- package/dist/core/hash.js +0 -18
- package/dist/indexer/drain.js +0 -306
- package/dist/indexer/embedding-identity.js +0 -20
- package/dist/indexer/enrich.js +0 -260
- package/dist/indexer/reconcile.js +0 -890
- package/dist/indexer/scan/parse-file.js +0 -66
- package/dist/indexer/units/unit.js +0 -159
- package/dist/llm/embedders/provider-limits.js +0 -288
- package/dist/storage/repositories/files-repository.js +0 -181
- package/dist/storage/repositories/units-repository.js +0 -510
|
@@ -0,0 +1,161 @@
|
|
|
1
|
+
// This Source Code Form is subject to the terms of the Mozilla Public
|
|
2
|
+
// License, v. 2.0. If a copy of the MPL was not distributed with this
|
|
3
|
+
// file, You can obtain one at https://mozilla.org/MPL/2.0/.
|
|
4
|
+
/**
|
|
5
|
+
* Incremental dir-staleness engine.
|
|
6
|
+
*
|
|
7
|
+
* Decides, per stash directory, whether the directory's indexed rows are still
|
|
8
|
+
* fresh relative to what is on disk — so an incremental `akm index` run can
|
|
9
|
+
* skip unchanged directories instead of regenerating their metadata.
|
|
10
|
+
*
|
|
11
|
+
* Two persisted signals back the decision:
|
|
12
|
+
* 1. The `entries` rows already indexed for the directory (`getEntriesByDir`).
|
|
13
|
+
* 2. The `index_dir_state` row (`getIndexDirState`): the fingerprint of the
|
|
14
|
+
* directory's walked file set (basename set + max mtime, `computeDirFingerprint`)
|
|
15
|
+
* as of its last drain, plus the row count that drain persisted.
|
|
16
|
+
*
|
|
17
|
+
* `getCachedDirState` is the pre-drain gate (#900): a directory whose walked
|
|
18
|
+
* fingerprint still matches its row is skipped before any file is read.
|
|
19
|
+
*/
|
|
20
|
+
import { createHash } from "node:crypto";
|
|
21
|
+
import fs from "node:fs";
|
|
22
|
+
import path from "node:path";
|
|
23
|
+
import { compareCodePoints } from "../../core/common.js";
|
|
24
|
+
import { getEntriesByDir } from "../../storage/repositories/index-entries-repository.js";
|
|
25
|
+
import { getIndexDirState } from "../../storage/repositories/index-meta-repository.js";
|
|
26
|
+
/**
|
|
27
|
+
* Post-drain freshness verdict. `files` is the recognized file set the drain
|
|
28
|
+
* produced (compared against the persisted entries); `fingerprint` is the
|
|
29
|
+
* walked-set fingerprint compared against the persisted row and defaults to
|
|
30
|
+
* one computed over `files`.
|
|
31
|
+
*/
|
|
32
|
+
export function getDirIndexState(db, dirPath, files, builtAtMs, indexVariant = "", fingerprint = computeDirFingerprint(dirPath, files, indexVariant)) {
|
|
33
|
+
const prevEntries = getEntriesByDir(db, dirPath);
|
|
34
|
+
if (prevEntries.length > 0) {
|
|
35
|
+
const staleReason = getDirStaleReason(dirPath, files, prevEntries, builtAtMs);
|
|
36
|
+
if (staleReason)
|
|
37
|
+
return { stale: true, reason: staleReason, persistedRowCount: prevEntries.length };
|
|
38
|
+
const cachedState = getIndexDirState(db, dirPath);
|
|
39
|
+
if (!cachedState || cachedState.fileSetHash !== fingerprint.fileSetHash) {
|
|
40
|
+
return {
|
|
41
|
+
stale: true,
|
|
42
|
+
reason: { kind: "index-context-changed", detail: indexVariant },
|
|
43
|
+
persistedRowCount: prevEntries.length,
|
|
44
|
+
};
|
|
45
|
+
}
|
|
46
|
+
return { stale: false, reason: { kind: "unchanged" }, persistedRowCount: prevEntries.length };
|
|
47
|
+
}
|
|
48
|
+
const cachedState = getIndexDirState(db, dirPath);
|
|
49
|
+
if (cachedState && cachedState.fileSetHash === fingerprint.fileSetHash) {
|
|
50
|
+
return {
|
|
51
|
+
stale: false,
|
|
52
|
+
reason: { kind: "cached-zero-row-state", detail: cachedState.reason },
|
|
53
|
+
persistedRowCount: 0,
|
|
54
|
+
};
|
|
55
|
+
}
|
|
56
|
+
return {
|
|
57
|
+
stale: true,
|
|
58
|
+
reason: { kind: "no-previous-rows", detail: cachedState ? `cached=${cachedState.reason}` : undefined },
|
|
59
|
+
persistedRowCount: 0,
|
|
60
|
+
};
|
|
61
|
+
}
|
|
62
|
+
/**
|
|
63
|
+
* Pre-drain gate (#900). A directory whose walked-set fingerprint matches its
|
|
64
|
+
* persisted row cannot recognize differently than last time, so it is skipped
|
|
65
|
+
* before `drainDirDocuments` reads a single file. A row that recorded a real
|
|
66
|
+
* generation (`rowCount > 0`) is skipped outright; a zero-row or pre-#900 row
|
|
67
|
+
* goes through the entries-aware check so the dedup-order guard still applies.
|
|
68
|
+
*/
|
|
69
|
+
export function getCachedDirState(db, dirPath, files, builtAtMs, priorDirsChanged, indexVariant, fingerprint) {
|
|
70
|
+
const cached = getIndexDirState(db, dirPath);
|
|
71
|
+
if (!cached || cached.fileSetHash !== fingerprint.fileSetHash)
|
|
72
|
+
return undefined;
|
|
73
|
+
if (cached.rowCount !== undefined && cached.rowCount > 0) {
|
|
74
|
+
return { stale: false, reason: { kind: "unchanged-precheck" }, persistedRowCount: cached.rowCount };
|
|
75
|
+
}
|
|
76
|
+
const state = getDirIndexState(db, dirPath, files, builtAtMs, indexVariant, fingerprint);
|
|
77
|
+
if (state.stale || state.reason.kind !== "cached-zero-row-state")
|
|
78
|
+
return undefined;
|
|
79
|
+
if (!canUseIncrementalSkip(state, priorDirsChanged))
|
|
80
|
+
return undefined;
|
|
81
|
+
return state;
|
|
82
|
+
}
|
|
83
|
+
export function canUseIncrementalSkip(state, priorDirsChanged) {
|
|
84
|
+
return !(priorDirsChanged &&
|
|
85
|
+
state.reason.kind === "cached-zero-row-state" &&
|
|
86
|
+
state.reason.detail === "deduped-zero-row");
|
|
87
|
+
}
|
|
88
|
+
export function computeDirFingerprint(_dirPath, files, indexVariant = "") {
|
|
89
|
+
// One `statSync` per file — the same call this function has always made — but
|
|
90
|
+
// every field it returns that can witness a change is kept, per file, instead
|
|
91
|
+
// of being collapsed into a single max.
|
|
92
|
+
//
|
|
93
|
+
// `Math.max` over mtimes discarded everything except the newest file, so an
|
|
94
|
+
// edit to any other file landed below the max and was invisible; and mtime
|
|
95
|
+
// alone is writable by ordinary tooling (`touch -r`, `rsync --times`,
|
|
96
|
+
// `cp -p`, archive extraction), so a restored timestamp hid an edit outright.
|
|
97
|
+
// Size catches any length-changing edit; ctime catches the rest, because
|
|
98
|
+
// utimes(2) cannot hold the inode's change time back.
|
|
99
|
+
//
|
|
100
|
+
// This is still a heuristic: ctime also moves on metadata-only changes
|
|
101
|
+
// (chmod/chown) and after copying a tree, which costs an unnecessary rescan.
|
|
102
|
+
// That direction is safe — extra work, never stale content.
|
|
103
|
+
const entries = [];
|
|
104
|
+
let fileMtimeMaxMs = 0;
|
|
105
|
+
for (const file of [...new Set(files)].sort(compareCodePoints)) {
|
|
106
|
+
const name = path.basename(file);
|
|
107
|
+
try {
|
|
108
|
+
// `bigint: true` is the same syscall but reports nanoseconds. Millisecond
|
|
109
|
+
// floats would let an edit made inside the same millisecond as the last
|
|
110
|
+
// run's stat land on an identical digest.
|
|
111
|
+
const stat = fs.statSync(file, { bigint: true });
|
|
112
|
+
fileMtimeMaxMs = Math.max(fileMtimeMaxMs, Number(stat.mtimeMs));
|
|
113
|
+
entries.push(`${name}\0${stat.size}\0${stat.mtimeNs}\0${stat.ctimeNs}`);
|
|
114
|
+
}
|
|
115
|
+
catch {
|
|
116
|
+
// Unreadable or vanished: record it as such so the digest differs from
|
|
117
|
+
// any run where the file could be read, forcing a rescan.
|
|
118
|
+
entries.push(`${name}\0unreadable`);
|
|
119
|
+
}
|
|
120
|
+
}
|
|
121
|
+
const digest = createHash("sha256")
|
|
122
|
+
.update([indexVariant, ...entries].join("\n"), "utf8")
|
|
123
|
+
.digest("hex");
|
|
124
|
+
return { fileSetHash: digest, fileMtimeMaxMs };
|
|
125
|
+
}
|
|
126
|
+
function getDirStaleReason(_dirPath, currentFiles, previousEntries, builtAtMs) {
|
|
127
|
+
const prevFileNames = new Set(previousEntries
|
|
128
|
+
.map((ie) => {
|
|
129
|
+
const fromPath = path.basename(ie.filePath);
|
|
130
|
+
return fromPath || ie.entry.filename;
|
|
131
|
+
})
|
|
132
|
+
.filter((e) => !!e));
|
|
133
|
+
const currFileNames = new Set(currentFiles.map((f) => path.basename(f)));
|
|
134
|
+
if (prevFileNames.size !== currFileNames.size) {
|
|
135
|
+
return { kind: "file-set-changed", detail: `${prevFileNames.size} -> ${currFileNames.size} files` };
|
|
136
|
+
}
|
|
137
|
+
for (const name of currFileNames) {
|
|
138
|
+
if (!prevFileNames.has(name))
|
|
139
|
+
return { kind: "file-set-changed", detail: name };
|
|
140
|
+
}
|
|
141
|
+
for (const file of currentFiles) {
|
|
142
|
+
try {
|
|
143
|
+
if (fs.statSync(file).mtimeMs > builtAtMs)
|
|
144
|
+
return { kind: "mtime-changed", detail: path.basename(file) };
|
|
145
|
+
}
|
|
146
|
+
catch {
|
|
147
|
+
return { kind: "missing-file", detail: path.basename(file) };
|
|
148
|
+
}
|
|
149
|
+
}
|
|
150
|
+
return undefined;
|
|
151
|
+
}
|
|
152
|
+
export function inferZeroRowReason(stash, priorReason, warnings, dirPath, dedupedRows) {
|
|
153
|
+
if (dedupedRows > 0)
|
|
154
|
+
return "deduped-zero-row";
|
|
155
|
+
const workflowNoise = warnings.some((warning) => warning.startsWith("Skipped workflow ") && warning.includes(dirPath));
|
|
156
|
+
if (workflowNoise)
|
|
157
|
+
return "workflow-noise";
|
|
158
|
+
if (!stash || stash.entries.length === 0)
|
|
159
|
+
return "empty-generated-set";
|
|
160
|
+
return `zero-row:${priorReason?.kind ?? "unknown"}`;
|
|
161
|
+
}
|
|
@@ -1349,24 +1349,7 @@ export function applyPostContributorFields(entry, file, canonicalName, dirPath)
|
|
|
1349
1349
|
// filename word. `normalizeTerms` below dedupes author-restated tokens.
|
|
1350
1350
|
entry.tags = [...(entry.tags ?? []), ...extractDirTagsFromName(canonicalName)];
|
|
1351
1351
|
entry.tags = normalizeTerms(entry.tags ?? []);
|
|
1352
|
-
|
|
1353
|
-
// on a memory's OWN filename (memory-inference.ts's `derivedChildPath`/
|
|
1354
|
-
// `isDerivedByName` convention: `<parent>.derived.md`), not a topical
|
|
1355
|
-
// word — `extractTagsFromPath` above tokenizes on `.` alongside `-`/`_`,
|
|
1356
|
-
// so every derived twin's filename contributes a "derived" tag purely as
|
|
1357
|
-
// an artifact of that convention (kept in `entry.tags` itself; only the
|
|
1358
|
-
// ALIAS input below is filtered, so anything else keyed on the raw tag
|
|
1359
|
-
// set is unaffected). Left in, `buildAliases`' tags.join(" ") mints a
|
|
1360
|
-
// SYNTHETIC "<base> derived" alias purely because tags.length > 1, which
|
|
1361
|
-
// then wins `alias-ranking` credit any time the base name is a query
|
|
1362
|
-
// token — see `exactNameRankingContributor`, which already strips this
|
|
1363
|
-
// exact suffix before treating a derived twin's name as content, for the
|
|
1364
|
-
// identical reason. Scoped to memory so a coincidentally
|
|
1365
|
-
// ".derived"-suffixed asset of another type is untouched.
|
|
1366
|
-
const aliasTagInput = entry.type === "memory" && canonicalName.toLowerCase().endsWith(".derived")
|
|
1367
|
-
? entry.tags.filter((tag) => tag !== "derived")
|
|
1368
|
-
: entry.tags;
|
|
1369
|
-
entry.aliases = mergeAliases(entry.aliases, buildAliases(canonicalName, aliasTagInput));
|
|
1352
|
+
entry.aliases = mergeAliases(entry.aliases, buildAliases(canonicalName, entry.tags));
|
|
1370
1353
|
// Search hints are only generated when LLM is configured (via enhanceStashWithLlm)
|
|
1371
1354
|
// Heuristic search hints are too noisy to be useful for search quality
|
|
1372
1355
|
entry.filename = path.basename(file);
|
|
@@ -5,18 +5,19 @@
|
|
|
5
5
|
* Per-directory document drain — akm 0.9.0 Chunk 5, milestone F4a M-core-2 (the
|
|
6
6
|
* engine swap). Replaces the live indexer's per-dir flat-walk matcher-pass
|
|
7
7
|
* `IndexDocument` stream with the `akm` adapter's `recognize` `IndexDocument`
|
|
8
|
-
* stream, reconstructing the durable `IndexDocument` via {@link
|
|
9
|
-
* (proven lossless by the shadow-parity gate).
|
|
8
|
+
* stream, reconstructing the durable `IndexDocument` via {@link
|
|
9
|
+
* indexDocumentToStashEntry} (proven lossless by the shadow-parity gate).
|
|
10
10
|
*
|
|
11
|
-
*
|
|
12
|
-
*
|
|
11
|
+
* Two behaviors the adapter fold does NOT carry, restored here at the drain
|
|
12
|
+
* layer (spec §14.2 "drain the full document stream"):
|
|
13
13
|
*
|
|
14
|
-
* - **
|
|
15
|
-
*
|
|
16
|
-
*
|
|
17
|
-
*
|
|
18
|
-
*
|
|
19
|
-
*
|
|
14
|
+
* - **Broken-workflow drop (item 3).** The live path dropped a broken workflow
|
|
15
|
+
* via the renderer contributor's throw → metadata-pass skip-with-warning; the
|
|
16
|
+
* `akm` adapter's synchronous `foldRecognizedMetadata` SWALLOWS the parse
|
|
17
|
+
* error, so a broken workflow would otherwise silently index. We run the
|
|
18
|
+
* shared source-IR compiler on drained workflow docs and DROP
|
|
19
|
+
* the entry with the same `Skipped workflow …` warning
|
|
20
|
+
* ({@link buildMetadataSkipWarning}), so the workflow-skip summary counts it.
|
|
20
21
|
*
|
|
21
22
|
* `doc.hash` (= sha256 of the file content) is surfaced per recognized file so
|
|
22
23
|
* the persist layer can populate the `content_hash` column (item 2). It is keyed
|
|
@@ -30,8 +31,12 @@ import { akmAdapter } from "../../core/adapter/adapters/akm-adapter.js";
|
|
|
30
31
|
import { compareCodePoints } from "../../core/common.js";
|
|
31
32
|
import { canonicalizeWorkflowName } from "../../core/recognition-util.js";
|
|
32
33
|
import { resolveWorkflowSourceDomains, workflowNameForSourcePath } from "../../workflows/source-files.js";
|
|
34
|
+
import { compileWorkflowSource } from "../../workflows/source-ir/compile.js";
|
|
35
|
+
import { buildMetadataSkipWarning } from "../passes/metadata.js";
|
|
33
36
|
import { buildFileContext } from "../walk/file-context.js";
|
|
34
|
-
import {
|
|
37
|
+
import { indexDocumentToStashEntry } from "./doc-to-entry.js";
|
|
38
|
+
/** The markdown-workflow renderer name the `akm` adapter carries on `documentJson.renderer`. */
|
|
39
|
+
const WORKFLOW_MD_RENDERER = "workflow-md";
|
|
35
40
|
/**
|
|
36
41
|
* Drain one directory's recognized documents into durable entries.
|
|
37
42
|
*
|
|
@@ -83,24 +88,26 @@ export function drainDirDocuments(adapter, component, fileContexts) {
|
|
|
83
88
|
continue;
|
|
84
89
|
}
|
|
85
90
|
}
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
91
|
+
const doc = adapter.recognize(component, file);
|
|
92
|
+
if (doc === null)
|
|
93
|
+
continue;
|
|
94
|
+
if (!doc.conceptId) {
|
|
95
|
+
warnings.push(`Skipped ${file.absPath}: adapter "${adapter.id}" returned no conceptId.`);
|
|
96
|
+
continue;
|
|
97
|
+
}
|
|
98
|
+
const entry = indexDocumentToStashEntry(doc);
|
|
99
|
+
// Workflow docs: drop-with-warning if broken; otherwise cache a lossless
|
|
100
|
+
// runtime projection when the current executor can represent the source.
|
|
101
|
+
const dropWarning = handleWorkflowDoc(doc, file, component.root);
|
|
102
|
+
if (dropWarning !== null) {
|
|
103
|
+
warnings.push(dropWarning);
|
|
104
|
+
if (workflowName !== undefined)
|
|
105
|
+
invalidWorkflowOwnerNames.add(canonicalizeWorkflowName(workflowName));
|
|
98
106
|
continue;
|
|
99
107
|
}
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
conceptIdByFile.set(file.absPath, conceptId);
|
|
108
|
+
if (doc.hash !== undefined)
|
|
109
|
+
hashByFile.set(file.absPath, doc.hash);
|
|
110
|
+
conceptIdByFile.set(file.absPath, doc.conceptId);
|
|
104
111
|
entries.push(entry);
|
|
105
112
|
}
|
|
106
113
|
return { entries, warnings, hashByFile, conceptIdByFile, rejectedPaths, rejectedConceptIds };
|
|
@@ -126,3 +133,39 @@ export function recognizeStashEntries(stashRoot, files) {
|
|
|
126
133
|
? { entries: drained.entries, warnings: drained.warnings }
|
|
127
134
|
: { entries: drained.entries };
|
|
128
135
|
}
|
|
136
|
+
/**
|
|
137
|
+
* If `doc` is a workflow, compile it through source IR: return a
|
|
138
|
+
* `Skipped workflow …` drop warning when it is broken, or return `null` when
|
|
139
|
+
* it compiles. Non-workflow docs return `null` immediately.
|
|
140
|
+
*/
|
|
141
|
+
function handleWorkflowDoc(doc, file, workspaceRoot) {
|
|
142
|
+
if (doc.type !== "workflow" ||
|
|
143
|
+
(doc.adapterId !== "akm" && doc.adapterId !== "akm-workflow") ||
|
|
144
|
+
(docRenderer(doc) !== WORKFLOW_MD_RENDERER && doc.adapterId !== "akm-workflow")) {
|
|
145
|
+
return null;
|
|
146
|
+
}
|
|
147
|
+
const result = compileWorkflowSource(file.content(), { path: file.relPath, workspaceRoot });
|
|
148
|
+
if (!result.ok)
|
|
149
|
+
return workflowDropWarning(file, result.errors);
|
|
150
|
+
return null;
|
|
151
|
+
}
|
|
152
|
+
/** The winning renderer name the `akm` adapter carries on `documentJson.renderer`, or `undefined`. */
|
|
153
|
+
function docRenderer(doc) {
|
|
154
|
+
const dj = doc.documentJson;
|
|
155
|
+
if (dj !== null && typeof dj === "object" && "renderer" in dj) {
|
|
156
|
+
const renderer = dj.renderer;
|
|
157
|
+
return typeof renderer === "string" ? renderer : undefined;
|
|
158
|
+
}
|
|
159
|
+
return undefined;
|
|
160
|
+
}
|
|
161
|
+
/**
|
|
162
|
+
* Build the `Skipped workflow <path>:\n…` warning byte-for-byte the way the live
|
|
163
|
+
* pipeline did: the workflow parser's `path:line — message` summary wrapped in
|
|
164
|
+
* the `Workflow has errors:` prefix (the string `loadDocument`/`loadProgram`
|
|
165
|
+
* threw), then {@link buildMetadataSkipWarning}'s workflow branch. `startsWith
|
|
166
|
+
* "Skipped workflow "` so `isWorkflowSkipWarning` counts it for the summary.
|
|
167
|
+
*/
|
|
168
|
+
function workflowDropWarning(file, errors) {
|
|
169
|
+
const summary = errors.map((e) => `${file.relPath}:${e.line} — ${e.message}`).join("\n");
|
|
170
|
+
return buildMetadataSkipWarning(file.absPath, "workflow", `Workflow has errors:\n${summary}`);
|
|
171
|
+
}
|