akm-cli 0.9.16-alpha.1 → 0.9.16

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (147) hide show
  1. package/CHANGELOG.md +56 -132
  2. package/dist/assets/hints/cli-hints-full.md +13 -6
  3. package/dist/assets/tasks/core/index-refresh.yml +1 -1
  4. package/dist/assets/tasks/improve/akm-improve-catchup.yml +3 -6
  5. package/dist/cli/retired-commands.js +0 -4
  6. package/dist/cli/unknown-flags.js +3 -36
  7. package/dist/commands/env/env-binding.js +4 -4
  8. package/dist/commands/env/env-cli.js +3 -3
  9. package/dist/commands/improve/collapse-detector.js +2 -2
  10. package/dist/commands/improve/consolidate.js +4 -6
  11. package/dist/commands/improve/improve-cli.js +20 -15
  12. package/dist/commands/improve/reflect.js +23 -2
  13. package/dist/commands/lint/base-linter.js +9 -0
  14. package/dist/commands/lint/env-key-rules.js +2 -2
  15. package/dist/commands/proposal/propose.js +15 -1
  16. package/dist/commands/proposal/repository.js +3 -12
  17. package/dist/commands/proposal/validators/proposal-quality-validators.js +40 -3
  18. package/dist/commands/proposal/validators/proposal-validators.js +5 -4
  19. package/dist/commands/read/curate.js +44 -34
  20. package/dist/commands/read/search.js +35 -54
  21. package/dist/commands/read/show.js +21 -2
  22. package/dist/commands/registry-cli.js +5 -5
  23. package/dist/commands/sources/add-cli.js +59 -16
  24. package/dist/commands/sources/bundle-cli.js +35 -11
  25. package/dist/commands/sources/bundle-config-ops.js +30 -0
  26. package/dist/commands/sources/dangerous-env-audit.js +4 -4
  27. package/dist/commands/sources/info.js +8 -8
  28. package/dist/commands/sources/installed-stashes.js +55 -61
  29. package/dist/commands/sources/source-add.js +39 -38
  30. package/dist/commands/sources/source-manage.js +34 -12
  31. package/dist/commands/sources/stash-cli.js +111 -119
  32. package/dist/commands/sources/stash-skeleton.js +6 -3
  33. package/dist/commands/tasks/explain.js +4 -1
  34. package/dist/commands/tasks/tasks-cli.js +31 -9
  35. package/dist/commands/tasks/tasks.js +239 -194
  36. package/dist/commands/tasks/validate.js +20 -32
  37. package/dist/core/activation-policy.js +4 -4
  38. package/dist/core/adapter/adapters/akm-adapter.js +8 -35
  39. package/dist/core/adapter/adapters/akm-metadata.js +1 -11
  40. package/dist/core/adapter/execution-source.js +10 -29
  41. package/dist/core/asset/asset-placement.js +0 -35
  42. package/dist/core/config/config-schema.js +64 -8
  43. package/dist/core/config/config-sources.js +96 -2
  44. package/dist/core/config/config.js +190 -24
  45. package/dist/core/config/legacy-source-shape-shim.js +9 -0
  46. package/dist/core/config/schema/embedding.js +30 -7
  47. package/dist/core/config/schema/execution.js +23 -0
  48. package/dist/core/config/schema/experimental.js +1 -1
  49. package/dist/core/config/schema/scheduler.js +20 -0
  50. package/dist/core/config/schema/search.js +10 -12
  51. package/dist/core/config/schema/sources-bundles.js +32 -1
  52. package/dist/core/content-safety.js +52 -0
  53. package/dist/core/errors.js +2 -5
  54. package/dist/core/maintenance-barrier.js +11 -13
  55. package/dist/core/paths.js +11 -0
  56. package/dist/core/run-lock.js +2 -5
  57. package/dist/core/state/migrations.js +1 -26
  58. package/dist/core/state-db.js +27 -63
  59. package/dist/core/type-presentation.js +1 -1
  60. package/dist/core/write-source.js +13 -8
  61. package/dist/indexer/bundle-identity-guard.js +45 -8
  62. package/dist/indexer/ensure-index.js +0 -5
  63. package/dist/indexer/index-db-contention.js +56 -0
  64. package/dist/indexer/index-rebuild-lock.js +73 -0
  65. package/dist/indexer/index-written-assets.js +171 -133
  66. package/dist/indexer/indexer.js +1621 -458
  67. package/dist/indexer/lookup/adapter-concept-owner.js +5 -19
  68. package/dist/indexer/materialize-embeddings.js +785 -0
  69. package/dist/indexer/passes/dir-staleness.js +161 -0
  70. package/dist/indexer/passes/metadata.js +1 -18
  71. package/dist/indexer/scan/drain-dir.js +70 -27
  72. package/dist/indexer/search/db-search.js +89 -373
  73. package/dist/indexer/search/ranking-contributors.js +16 -21
  74. package/dist/indexer/search/ranking.js +57 -135
  75. package/dist/indexer/search/search-source.js +29 -11
  76. package/dist/integrations/agent/execution-lowering.js +3 -2
  77. package/dist/integrations/agent/execution-preparation.js +32 -1
  78. package/dist/integrations/agent/prompts.js +1 -1
  79. package/dist/integrations/agent/request-lowering.js +3 -2
  80. package/dist/llm/client.js +3 -11
  81. package/dist/llm/embedder.js +3 -10
  82. package/dist/llm/embedders/remote.js +104 -133
  83. package/dist/llm/feature-gate.js +2 -4
  84. package/dist/llm/rerank-client.js +3 -3
  85. package/dist/output/html-render.js +2 -1
  86. package/dist/output/shapes/passthrough.js +2 -1
  87. package/dist/output/stdout.js +24 -0
  88. package/dist/output/text/command-format.js +13 -19
  89. package/dist/output/text/helpers.js +1 -1
  90. package/dist/output/text/index.js +2 -5
  91. package/dist/output/text.js +4 -3
  92. package/dist/registry/resolve.js +37 -10
  93. package/dist/scripts/akm-migrate-node.js +15197 -11351
  94. package/dist/scripts/akm-migrate.js +15514 -11668
  95. package/dist/setup/semantic-assets.js +2 -2
  96. package/dist/setup/setup.js +3 -3
  97. package/dist/setup/steps/connection.js +2 -3
  98. package/dist/setup/steps/tasks.js +29 -36
  99. package/dist/sources/providers/git-install.js +17 -11
  100. package/dist/sources/providers/git-provider.js +12 -5
  101. package/dist/sources/providers/git-stash.js +38 -16
  102. package/dist/sources/snapshot-fetchers/website-ingest.js +3 -3
  103. package/dist/storage/repositories/embedding-salvage-repository.js +184 -0
  104. package/dist/storage/repositories/index-connection.js +3 -1
  105. package/dist/storage/repositories/index-entries-repository.js +68 -77
  106. package/dist/storage/repositories/index-entry-schema.js +25 -16
  107. package/dist/storage/repositories/index-fts-repository.js +263 -29
  108. package/dist/storage/repositories/index-meta-repository.js +29 -0
  109. package/dist/storage/repositories/index-schema.js +122 -115
  110. package/dist/storage/repositories/index-utility-repository.js +1 -1
  111. package/dist/storage/repositories/index-vec-repository.js +435 -22
  112. package/dist/tasks/activation-config.js +90 -0
  113. package/dist/tasks/backends/cron.js +9 -0
  114. package/dist/tasks/backends/launchd.js +1 -0
  115. package/dist/tasks/backends/schtasks.js +2 -0
  116. package/dist/tasks/embedded.js +4 -5
  117. package/dist/tasks/scheduler-binding.js +2 -2
  118. package/dist/tasks/scheduler-sync-preview.js +8 -1
  119. package/dist/tasks/scheduler-sync.js +19 -10
  120. package/dist/tasks/source/parse-task-source.js +10 -113
  121. package/dist/tasks/source/project-v4.js +2 -2
  122. package/dist/tasks/source/task-source-v4.js +4 -12
  123. package/dist/tasks/source/task-to-v3.js +4 -12
  124. package/dist/tasks/source/task-to-v4.js +40 -7
  125. package/docs/migration/README.md +1 -0
  126. package/docs/migration/release-notes/0.9.15.md +36 -34
  127. package/docs/migration/release-notes/0.9.16.md +60 -98
  128. package/docs/migration/release-notes/README.md +0 -5
  129. package/docs/migration/v0.9.1-to-v0.9.2.md +6 -9
  130. package/docs/reference/cli.md +124 -122
  131. package/docs/reference/configuration.md +137 -133
  132. package/docs/reference/data-and-telemetry.md +1 -2
  133. package/docs/reference/tasks.md +34 -29
  134. package/package.json +1 -1
  135. package/schemas/akm-config.json +170 -6
  136. package/schemas/akm-task.json +1 -2
  137. package/dist/commands/sources/index-status.js +0 -99
  138. package/dist/core/hash.js +0 -18
  139. package/dist/indexer/drain.js +0 -306
  140. package/dist/indexer/embedding-identity.js +0 -20
  141. package/dist/indexer/enrich.js +0 -260
  142. package/dist/indexer/reconcile.js +0 -890
  143. package/dist/indexer/scan/parse-file.js +0 -66
  144. package/dist/indexer/units/unit.js +0 -159
  145. package/dist/llm/embedders/provider-limits.js +0 -288
  146. package/dist/storage/repositories/files-repository.js +0 -181
  147. package/dist/storage/repositories/units-repository.js +0 -510
@@ -1,66 +0,0 @@
1
- // This Source Code Form is subject to the terms of the Mozilla Public
2
- // License, v. 2.0. If a copy of the MPL was not distributed with this
3
- // file, You can obtain one at https://mozilla.org/MPL/2.0/.
4
- import { compileWorkflowSource } from "../../workflows/source-ir/compile.js";
5
- import { buildMetadataSkipWarning } from "../passes/metadata.js";
6
- import { indexDocumentToStashEntry } from "./doc-to-entry.js";
7
- /** The markdown-workflow renderer name the `akm` adapter carries on `documentJson.renderer`. */
8
- const WORKFLOW_MD_RENDERER = "workflow-md";
9
- /**
10
- * Parse one file through `adapter.recognize`, reconstruct its durable entry,
11
- * and drop it with a warning when it names no conceptId or is a workflow
12
- * document that fails to compile. Pure of DB/global state.
13
- */
14
- export function parseFileDocument(adapter, component, file) {
15
- const doc = adapter.recognize(component, file);
16
- if (doc === null)
17
- return { parsed: null, warning: null, isWorkflowDrop: false };
18
- if (!doc.conceptId) {
19
- return {
20
- parsed: null,
21
- warning: `Skipped ${file.absPath}: adapter "${adapter.id}" returned no conceptId.`,
22
- isWorkflowDrop: false,
23
- };
24
- }
25
- const entry = indexDocumentToStashEntry(doc);
26
- const dropWarning = handleWorkflowDoc(doc, file, component.root);
27
- if (dropWarning !== null) {
28
- return { parsed: null, warning: dropWarning, isWorkflowDrop: true };
29
- }
30
- return { parsed: { entry, hash: doc.hash, conceptId: doc.conceptId }, warning: null, isWorkflowDrop: false };
31
- }
32
- /**
33
- * If `doc` is a workflow, compile it through source IR: return a
34
- * `Skipped workflow …` drop warning when it is broken, or return `null` when
35
- * it compiles. Non-workflow docs return `null` immediately.
36
- */
37
- function handleWorkflowDoc(doc, file, workspaceRoot) {
38
- if (doc.type !== "workflow" ||
39
- (doc.adapterId !== "akm" && doc.adapterId !== "akm-workflow") ||
40
- (docRenderer(doc) !== WORKFLOW_MD_RENDERER && doc.adapterId !== "akm-workflow")) {
41
- return null;
42
- }
43
- const result = compileWorkflowSource(file.content(), { path: file.relPath, workspaceRoot });
44
- if (!result.ok)
45
- return workflowDropWarning(file, result.errors);
46
- return null;
47
- }
48
- /** The winning renderer name the `akm` adapter carries on `documentJson.renderer`, or `undefined`. */
49
- function docRenderer(doc) {
50
- const dj = doc.documentJson;
51
- if (dj !== null && typeof dj === "object" && "renderer" in dj) {
52
- const renderer = dj.renderer;
53
- return typeof renderer === "string" ? renderer : undefined;
54
- }
55
- return undefined;
56
- }
57
- /**
58
- * Build the `Skipped workflow <path>:\n…` warning byte-for-byte the way the live
59
- * pipeline did: the workflow parser's `path:line — message` summary wrapped in
60
- * the `Workflow has errors:` prefix (the string `loadDocument`/`loadProgram`
61
- * threw), then {@link buildMetadataSkipWarning}'s workflow branch.
62
- */
63
- function workflowDropWarning(file, errors) {
64
- const summary = errors.map((e) => `${file.relPath}:${e.line} — ${e.message}`).join("\n");
65
- return buildMetadataSkipWarning(file.absPath, "workflow", `Workflow has errors:\n${summary}`);
66
- }
@@ -1,159 +0,0 @@
1
- // This Source Code Form is subject to the terms of the Mozilla Public
2
- // License, v. 2.0. If a copy of the MPL was not distributed with this
3
- // file, You can obtain one at https://mozilla.org/MPL/2.0/.
4
- import { parseMarkdownToc } from "../../core/asset/markdown.js";
5
- import { splitMarkdownFragments } from "../../core/asset/markdown-fragments.js";
6
- import { hashEmbeddableText } from "../../core/hash.js";
7
- import { getMarkdownFragmentContent, hasMarkdownFragmentContent } from "../passes/metadata.js";
8
- import { buildSearchFields } from "../search/search-fields.js";
9
- /**
10
- * Separator between the entry name and a fragment's section title in a unit
11
- * header — the exact format the contract specifies, kept as one named
12
- * constant so every header is built the same way.
13
- */
14
- const UNIT_HEADER_SECTION_SEPARATOR = " › ";
15
- /** Unit 0's body: description, tags, hints, parameters — the non-empty ones, one per line, in that order. */
16
- function structuredFieldsText(source) {
17
- const body = [source.description, source.tags, source.hints, source.parameters]
18
- .filter((field) => field.length > 0)
19
- .join("\n");
20
- return `${source.name}\n${body}`;
21
- }
22
- /**
23
- * One line per parameter — `name`, or `name: description` when the
24
- * parameter has a description — lowercased to match `buildSearchFields`'s
25
- * other structured fields (`units_fts` is case-insensitive either way; this
26
- * keeps the card unit's casing uniform). `""` when `parameters` is absent or
27
- * empty, so `structuredFieldsText`'s filter drops it cleanly.
28
- */
29
- function parametersText(parameters) {
30
- if (!parameters || parameters.length === 0)
31
- return "";
32
- return parameters
33
- .map((param) => (param.description ? `${param.name}: ${param.description}` : param.name))
34
- .join("\n")
35
- .toLowerCase();
36
- }
37
- function fragmentHeaderText(name, sectionTitle) {
38
- return sectionTitle ? `${name}${UNIT_HEADER_SECTION_SEPARATOR}${sectionTitle}` : name;
39
- }
40
- /**
41
- * The section title standing over each fragment, in fragment order: a
42
- * fragment's own first heading line when it starts with one, else the
43
- * nearest heading at or before it (tracked while scanning fragments in
44
- * document order), else `null`. Reuses `parseMarkdownToc`, the same heading
45
- * list `splitMarkdownFragments` derives its section boundaries from, so a
46
- * "#" inside a fenced code block is never mistaken for a real heading.
47
- */
48
- function fragmentSectionTitles(safeMarkdown, fragments) {
49
- const headings = parseMarkdownToc(safeMarkdown).headings;
50
- let headingIndex = 0;
51
- let current = null;
52
- return fragments.map((fragment) => {
53
- while (headingIndex < headings.length && headings[headingIndex].line <= fragment.startLine) {
54
- current = headings[headingIndex].text;
55
- headingIndex++;
56
- }
57
- return current;
58
- });
59
- }
60
- /**
61
- * Split `text` into pieces no longer than `maxChars`: cut at the last "\n"
62
- * before the bound, else the last space, else hard-split a single unbroken
63
- * run (e.g. a URL) so every piece still respects the bound and the loop
64
- * always terminates. The cut character itself is dropped, not carried by
65
- * either side.
66
- */
67
- function splitOverflowingText(text, maxChars) {
68
- const pieces = [];
69
- let rest = text;
70
- while (rest.length > maxChars) {
71
- const window = rest.slice(0, maxChars);
72
- const newlineCut = window.lastIndexOf("\n");
73
- if (newlineCut > 0) {
74
- pieces.push(rest.slice(0, newlineCut));
75
- rest = rest.slice(newlineCut + 1);
76
- continue;
77
- }
78
- const spaceCut = window.lastIndexOf(" ");
79
- if (spaceCut > 0) {
80
- pieces.push(rest.slice(0, spaceCut));
81
- rest = rest.slice(spaceCut + 1);
82
- continue;
83
- }
84
- pieces.push(rest.slice(0, maxChars));
85
- rest = rest.slice(maxChars);
86
- }
87
- if (rest.length > 0)
88
- pieces.push(rest);
89
- return pieces;
90
- }
91
- /**
92
- * maxChars bounds every unit's text; a unit over it is split at the last
93
- * "\n" (else the last space) before the bound into sub-units that keep the
94
- * source fragmentId and take the next ordinals.
95
- */
96
- export function deriveUnits(source, maxChars) {
97
- if (!Number.isFinite(maxChars) || maxChars <= 0) {
98
- throw new RangeError("deriveUnits: maxChars must be a positive finite number");
99
- }
100
- const units = [];
101
- let ordinal = 0;
102
- const pushUnit = (fragmentId, text) => {
103
- units.push({ entryId: source.entryId, ordinal: ordinal++, fragmentId, hash: hashEmbeddableText(text), text });
104
- };
105
- for (const text of splitOverflowingText(structuredFieldsText(source), maxChars))
106
- pushUnit(null, text);
107
- if (source.safeMarkdown != null) {
108
- const fragments = splitMarkdownFragments(source.safeMarkdown);
109
- const sectionTitles = fragmentSectionTitles(source.safeMarkdown, fragments);
110
- fragments.forEach((fragment, index) => {
111
- const header = fragmentHeaderText(source.name, sectionTitles[index] ?? null);
112
- const text = `${header}\n${fragment.text}`;
113
- for (const piece of splitOverflowingText(text, maxChars))
114
- pushUnit(fragment.fragmentId, piece);
115
- });
116
- }
117
- return units;
118
- }
119
- /**
120
- * `UnitSource` from an already-parsed `IndexDocument` — `buildSearchFields`
121
- * for the structured fields, the entry's own carried markdown for fragments.
122
- * Shared by `reconcile.ts` (a freshly-parsed entry) and `enrich.ts` (the same
123
- * entry merged with LLM-enriched description/tags/searchHints, index-redesign
124
- * B5e) — a leaf in `units/`, not either caller, so importing it never creates
125
- * a reconcile.ts ↔ enrich.ts cycle.
126
- *
127
- * `hasMarkdownFragmentContent`/`getMarkdownFragmentContent` is the `akm`
128
- * adapter's own line-structure-preserving fragment projection
129
- * (`applyPreContributorFields`, gated `.md`-only and excluding sensitive
130
- * types), set during `recognize` and present ONLY for that adapter (or a
131
- * caller that re-tags a derived copy via `setMarkdownFragmentContent`, as
132
- * `enrich.ts` does). Every other adapter (`okf`, ...) never calls it, so
133
- * `hasMarkdownFragmentContent` is always false for their entries — falling
134
- * straight to `null` here would leave their body content in `entries_fts`'s
135
- * single per-entry `content` column (`buildSearchFields`, unconditional) but
136
- * in NO unit at all, an asymmetry that would silently blank a whole
137
- * adapter's fragment search once unit coverage is complete
138
- * (index-redesign-contract.md B3). `entry.content` — the same field already
139
- * surfaced through search hits and `show`, so already that adapter's own
140
- * public-safe projection — is the fallback fragment source for exactly this
141
- * case.
142
- */
143
- export function toUnitSource(entryId, entry) {
144
- const fields = buildSearchFields(entry);
145
- const safeMarkdown = hasMarkdownFragmentContent(entry)
146
- ? (getMarkdownFragmentContent(entry) ?? null)
147
- : typeof entry.content === "string" && entry.content.trim()
148
- ? entry.content
149
- : null;
150
- return {
151
- entryId,
152
- name: fields.name,
153
- description: fields.description,
154
- tags: fields.tags,
155
- hints: fields.hints,
156
- parameters: parametersText(entry.parameters),
157
- safeMarkdown,
158
- };
159
- }
@@ -1,288 +0,0 @@
1
- // This Source Code Form is subject to the terms of the Mozilla Public
2
- // License, v. 2.0. If a copy of the MPL was not distributed with this
3
- // file, You can obtain one at https://mozilla.org/MPL/2.0/.
4
- import { HEALTH_PROBE_TIMEOUT_MS } from "../client.js";
5
- /**
6
- * Window assumed for a provider that reports nothing about its own context
7
- * size (an OpenAI-compatible server, a gateway such as Bifrost): the most
8
- * common embedding window; the same-run adaptive shrink already in
9
- * `src/llm/embedders/remote.ts` corrects an overestimate.
10
- */
11
- export const DEFAULT_WINDOW_TOKENS = 8_192;
12
- /**
13
- * Chars-per-token ratio assumed when the provider exposes no exact
14
- * tokenizer to calibrate against: field-measured p99 on dense markdown
15
- * (#954, the same field evidence `DEFAULT_TOKEN_BUDGET` in
16
- * `src/llm/embedders/remote.ts` was tuned against).
17
- */
18
- export const CHARS_PER_TOKEN_TAIL = 2.6;
19
- /**
20
- * Reserves room, inside the provider's own token window, for the one-line
21
- * header every unit's text is prefixed with (entry name, or
22
- * "entry name › section title" — see A1's `deriveUnits`), so a unit's
23
- * header plus body never together exceed the real window. 64 tokens
24
- * comfortably covers a realistic header without materially shrinking the
25
- * usable window on a small-context provider.
26
- */
27
- export const UNIT_HEADER_MARGIN_TOKENS = 64;
28
- /**
29
- * Percentile (of chars-per-token ratios, sorted ascending) calibration
30
- * reports: the single densest sampled text, i.e. the smallest
31
- * chars-per-token ratio — the same conservative, worst-case-density intent
32
- * the {@link CHARS_PER_TOKEN_TAIL} fallback encodes as a fixed p99. Over
33
- * {@link CALIBRATION_SHAPES}' eight samples this percentile always resolves
34
- * to index 0 — it is simply the minimum ratio observed.
35
- */
36
- const CALIBRATION_PERCENTILE = 0.01;
37
- /** Content posted to `/tokenize` purely to check the route exists, before spending the full calibration corpus on it. */
38
- const TOKENIZE_PRESENCE_PROBE_TEXT = "ping";
39
- /**
40
- * Representative text shapes to calibrate a provider's chars-per-token
41
- * ratio against, tokenized once each (eight requests total): mirrors the
42
- * mix real markdown fragments produce — prose, code, a table, a list, a
43
- * link, non-Latin text (which tokenizes at a very different ratio than
44
- * English prose), SQL, and dense technical prose.
45
- */
46
- const CALIBRATION_SHAPES = [
47
- "This is a short sentence describing typical prose content used to calibrate the tokenizer.",
48
- "```ts\nfunction add(a: number, b: number): number {\n return a + b;\n}\n```",
49
- "| Column A | Column B | Column C |\n| --- | --- | --- |\n| 1 | 2 | 3 |\n| 4 | 5 | 6 |",
50
- "- first item in a list\n- second item in a list\n- third item, a little longer than the rest",
51
- "See [the reference documentation](https://example.com/docs/reference) for more detail on this API.",
52
- "日本語のテキストは英語と比べてトークンあたりの文字数が大きく異なることがあります。",
53
- "SELECT id, name, description FROM entries WHERE tags LIKE '%embedding%' ORDER BY updated_at DESC LIMIT 50;",
54
- "A longer paragraph mixing punctuation, numbers (like 42 and 3.14), and technical terms such as `tokenizer`, `embedding`, and `context window`.",
55
- ];
56
- /** Index into a `length`-element array, sorted ascending, at `percentile` (0-1). Clamped so a tiny array still yields a valid index. */
57
- function percentileIndex(length, percentile) {
58
- if (length <= 1)
59
- return 0;
60
- return Math.max(0, Math.min(length - 1, Math.floor((length - 1) * percentile)));
61
- }
62
- /**
63
- * Calibrate `charsPerToken` against a real provider/model by tokenizing
64
- * {@link CALIBRATION_SHAPES} once each and taking the
65
- * {@link CALIBRATION_PERCENTILE} (densest-text) chars-per-token ratio across
66
- * them — with eight samples that percentile is simply the minimum ratio
67
- * observed. A sample whose tokenize call fails is skipped rather than
68
- * aborting the whole calibration; only a total wipeout (every sample
69
- * failed) falls back to {@link CHARS_PER_TOKEN_TAIL}.
70
- */
71
- async function calibrateCharsPerToken(countTokens) {
72
- const ratios = [];
73
- for (const text of CALIBRATION_SHAPES) {
74
- try {
75
- const tokens = await countTokens(text);
76
- if (tokens > 0)
77
- ratios.push(text.length / tokens);
78
- }
79
- catch {
80
- // One bad calibration sample must not abort the whole probe.
81
- }
82
- }
83
- if (ratios.length === 0)
84
- return CHARS_PER_TOKEN_TAIL;
85
- ratios.sort((a, b) => a - b);
86
- return ratios[percentileIndex(ratios.length, CALIBRATION_PERCENTILE)];
87
- }
88
- /** GET/POST `url` bounded by `timeoutMs`, additionally aborting if `externalSignal` fires. Never throws on timeout/abort itself — that surfaces as a normal fetch rejection to the caller, which every call site here already catches. */
89
- function timedFetch(fetchImpl, url, init, timeoutMs, externalSignal) {
90
- const timeoutSignal = AbortSignal.timeout(timeoutMs);
91
- const signal = externalSignal ? AbortSignal.any([externalSignal, timeoutSignal]) : timeoutSignal;
92
- return fetchImpl(url, { ...init, signal });
93
- }
94
- /** Parse a response body as JSON, or `undefined` on anything that is not valid JSON — a malformed response is a failed probe, not a thrown error. */
95
- async function readJson(res) {
96
- try {
97
- return await res.json();
98
- }
99
- catch {
100
- return undefined;
101
- }
102
- }
103
- /**
104
- * Probe llama.cpp's `/tokenize` route: one presence check with a fixed
105
- * short string, then (only if that succeeds) a reusable `countTokens`
106
- * closure. The closure deliberately does NOT carry the probe's own
107
- * `probeSignal` forward — that signal belongs to this one
108
- * `probeProviderLimits` call's lifecycle, while the returned function is
109
- * held onto and invoked much later (real indexing), so it is bound only to
110
- * its own fresh per-call timeout.
111
- */
112
- async function probeLlamaCppCountTokens(origin, fetchImpl, timeoutMs, probeSignal) {
113
- const tokenizeOnce = (text, signal) => timedFetch(fetchImpl, `${origin}/tokenize`, { method: "POST", headers: { "Content-Type": "application/json" }, body: JSON.stringify({ content: text }) }, timeoutMs, signal);
114
- const presence = await tokenizeOnce(TOKENIZE_PRESENCE_PROBE_TEXT, probeSignal);
115
- if (!presence.ok)
116
- return undefined;
117
- const presenceBody = (await readJson(presence));
118
- if (!Array.isArray(presenceBody?.tokens))
119
- return undefined;
120
- return async (text) => {
121
- const res = await tokenizeOnce(text, undefined);
122
- if (!res.ok)
123
- throw new Error(`llama.cpp /tokenize request failed (${res.status})`);
124
- const json = (await readJson(res));
125
- if (!Array.isArray(json?.tokens))
126
- throw new Error("Unexpected /tokenize response: missing tokens array");
127
- return json.tokens.length;
128
- };
129
- }
130
- async function probeLlamaCpp(origin, config, fetchImpl, timeoutMs, signal) {
131
- const res = await timedFetch(fetchImpl, `${origin}/props`, { method: "GET" }, timeoutMs, signal);
132
- if (!res.ok)
133
- return undefined;
134
- const body = (await readJson(res));
135
- const windowTokens = body?.default_generation_settings?.n_ctx;
136
- if (typeof windowTokens !== "number" || !Number.isFinite(windowTokens) || windowTokens <= 0)
137
- return undefined;
138
- const probedSlots = typeof body?.total_slots === "number" && body.total_slots > 0 ? body.total_slots : 1;
139
- const slots = config.concurrency ?? probedSlots;
140
- const countTokens = await probeLlamaCppCountTokens(origin, fetchImpl, timeoutMs, signal).catch(() => undefined);
141
- const charsPerToken = countTokens ? await calibrateCharsPerToken(countTokens) : CHARS_PER_TOKEN_TAIL;
142
- return { windowTokens, slots, source: "llama.cpp", charsPerToken };
143
- }
144
- /** Ollama does not expose `num_parallel` (its in-flight slot count) via any API route, so the probe always reports 1 slot unless `config.concurrency` overrides it. */
145
- const OLLAMA_DEFAULT_SLOTS = 1;
146
- /** Find `"<arch>.context_length"` in Ollama's `model_info` — the arch prefix varies per model family, so every key is checked rather than assuming one name. */
147
- function findOllamaContextLength(modelInfo) {
148
- for (const [key, value] of Object.entries(modelInfo)) {
149
- if (key.endsWith(".context_length") && typeof value === "number" && Number.isFinite(value) && value > 0) {
150
- return value;
151
- }
152
- }
153
- return undefined;
154
- }
155
- async function probeOllama(origin, config, fetchImpl, timeoutMs, signal) {
156
- if (!config.model)
157
- return undefined;
158
- const res = await timedFetch(fetchImpl, `${origin}/api/show`, { method: "POST", headers: { "Content-Type": "application/json" }, body: JSON.stringify({ model: config.model }) }, timeoutMs, signal);
159
- if (!res.ok)
160
- return undefined;
161
- const body = (await readJson(res));
162
- if (!body?.model_info)
163
- return undefined;
164
- const windowTokens = findOllamaContextLength(body.model_info);
165
- if (windowTokens === undefined)
166
- return undefined;
167
- return {
168
- windowTokens,
169
- slots: config.concurrency ?? OLLAMA_DEFAULT_SLOTS,
170
- source: "ollama",
171
- charsPerToken: CHARS_PER_TOKEN_TAIL,
172
- };
173
- }
174
- /** The origin (`scheme://host[:port]`) to probe against, or `undefined` when `endpoint` is absent, unparseable, or not http(s) — a local-only embedder (no remote `endpoint`) never touches the network. */
175
- function resolveOrigin(endpoint) {
176
- if (!endpoint)
177
- return undefined;
178
- try {
179
- const parsed = new URL(endpoint);
180
- if (parsed.protocol !== "http:" && parsed.protocol !== "https:")
181
- return undefined;
182
- return parsed.origin;
183
- }
184
- catch {
185
- return undefined;
186
- }
187
- }
188
- /** The probe's own per-request timeout: `config.timeoutMs` when set, else the shared health-probe default (`src/llm/client.ts`'s `HEALTH_PROBE_TIMEOUT_MS`) reused rather than redefined. */
189
- function resolveProbeTimeoutMs(config) {
190
- return config.timeoutMs ?? HEALTH_PROBE_TIMEOUT_MS;
191
- }
192
- function defaultLimits(config) {
193
- return {
194
- windowTokens: DEFAULT_WINDOW_TOKENS,
195
- slots: config.concurrency ?? 1,
196
- source: "default",
197
- charsPerToken: CHARS_PER_TOKEN_TAIL,
198
- };
199
- }
200
- /**
201
- * An implausible probed window that cannot even fit its own header margin
202
- * plus one real character of unit text — `unitMaxChars` would floor to 0,
203
- * and `deriveUnits` (`src/indexer/units/unit.ts`) throws on a non-positive
204
- * `maxChars`. Reusing `unitMaxChars` itself as the usability test, rather
205
- * than a second hand-picked threshold, keeps the two in lockstep by
206
- * construction: a window this function accepts is, by definition, one
207
- * `deriveUnits` can never crash on.
208
- */
209
- function isUsableWindow(limits) {
210
- return unitMaxChars(limits) > 0;
211
- }
212
- async function probeProviderLimitsUncached(config, opts) {
213
- const origin = resolveOrigin(config.endpoint);
214
- if (!origin)
215
- return defaultLimits(config);
216
- const fetchImpl = opts?.fetch ?? fetch;
217
- const timeoutMs = resolveProbeTimeoutMs(config);
218
- const signal = opts?.signal;
219
- // An endpoint that answers with a window too small to be usable (at or
220
- // below UNIT_HEADER_MARGIN_TOKENS) is treated exactly like one that
221
- // reported nothing recognisable: falling through here means EVERY window
222
- // this module ever hands out is safe to feed straight into
223
- // `unitMaxChars`/`deriveUnits`, so the crash guard lives in exactly one
224
- // place instead of being re-defended at every downstream call site.
225
- const llamaCpp = await probeLlamaCpp(origin, config, fetchImpl, timeoutMs, signal).catch(() => undefined);
226
- if (llamaCpp && isUsableWindow(llamaCpp))
227
- return llamaCpp;
228
- const ollama = await probeOllama(origin, config, fetchImpl, timeoutMs, signal).catch(() => undefined);
229
- if (ollama && isUsableWindow(ollama))
230
- return ollama;
231
- return defaultLimits(config);
232
- }
233
- /**
234
- * Per-process memoisation of {@link probeProviderLimits}, keyed by the parts
235
- * of `config` that change what gets probed (`endpoint`, `model`,
236
- * `concurrency`, `timeoutMs`) — `reconcileRoots`, `reconcilePaths` and
237
- * `drainEmbeddingQueue` each probe once per call, so a single `akm index` or
238
- * `akm remember` otherwise repeated the same handful of HTTP requests two or
239
- * three times over. The cached PROMISE is stored (not just its resolved
240
- * value), so concurrent callers before the first probe settles share the one
241
- * in-flight request set rather than each starting their own. A probe that
242
- * falls back to `source: "default"` (network error, malformed response) is
243
- * cached too: a process is one CLI run, and a flapping endpoint is the
244
- * drain's own retry/back-off's problem, not this cache's.
245
- */
246
- const providerLimitsCache = new Map();
247
- /** TEST-ONLY: clear the per-process probe cache so each test starts from a clean slate. */
248
- export function _resetProviderLimitsCacheForTests() {
249
- providerLimitsCache.clear();
250
- }
251
- /**
252
- * Probe the configured embedding endpoint for its OWN window/slot limits.
253
- * Tries llama.cpp's `GET /props` first, then Ollama's `POST /api/show`; an
254
- * endpoint that answers neither (an OpenAI-compatible server, a gateway),
255
- * one that answers with an implausibly small window (see
256
- * {@link isUsableWindow}) — or a config with no remote `endpoint` at all (a
257
- * local-only embedder) — gets the conservative default. Never throws: any
258
- * probe failure (a network error, a malformed response, an unparseable
259
- * endpoint) resolves to the same default shape rather than rejecting.
260
- *
261
- * Memoised per process — see {@link providerLimitsCache} — so every caller
262
- * with the same effective config (`endpoint`/`model`/`concurrency`/
263
- * `timeoutMs`) shares one probe's HTTP requests instead of repeating them.
264
- */
265
- export async function probeProviderLimits(config, opts) {
266
- const cacheKey = JSON.stringify({
267
- endpoint: config.endpoint,
268
- model: config.model,
269
- concurrency: config.concurrency,
270
- timeoutMs: config.timeoutMs,
271
- });
272
- const cached = providerLimitsCache.get(cacheKey);
273
- if (cached)
274
- return cached;
275
- const probe = probeProviderLimitsUncached(config, opts);
276
- providerLimitsCache.set(cacheKey, probe);
277
- return probe;
278
- }
279
- /**
280
- * Character bound for one embedding unit's text (A1's `deriveUnits`
281
- * `maxChars` parameter): `windowTokens` minus the header margin, converted
282
- * to characters via `charsPerToken`. Never negative — a pathologically
283
- * small window still yields a usable (if tiny) bound rather than a
284
- * negative `maxChars` that would make every split trivially fail.
285
- */
286
- export function unitMaxChars(limits) {
287
- return Math.max(0, Math.floor((limits.windowTokens - UNIT_HEADER_MARGIN_TOKENS) * limits.charsPerToken));
288
- }