akm-cli 0.9.16-alpha.1 → 0.9.16-alpha.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +40 -132
- package/dist/assets/hints/cli-hints-full.md +13 -6
- package/dist/assets/tasks/core/index-refresh.yml +1 -1
- package/dist/assets/tasks/improve/akm-improve-catchup.yml +3 -6
- package/dist/cli/retired-commands.js +0 -4
- package/dist/cli/unknown-flags.js +3 -36
- package/dist/commands/env/env-binding.js +4 -4
- package/dist/commands/env/env-cli.js +3 -3
- package/dist/commands/improve/collapse-detector.js +2 -2
- package/dist/commands/improve/consolidate.js +4 -6
- package/dist/commands/improve/improve-cli.js +20 -15
- package/dist/commands/improve/reflect.js +23 -2
- package/dist/commands/lint/base-linter.js +9 -0
- package/dist/commands/lint/env-key-rules.js +2 -2
- package/dist/commands/proposal/propose.js +15 -1
- package/dist/commands/proposal/repository.js +3 -12
- package/dist/commands/proposal/validators/proposal-quality-validators.js +40 -3
- package/dist/commands/proposal/validators/proposal-validators.js +5 -4
- package/dist/commands/read/curate.js +44 -34
- package/dist/commands/read/search.js +35 -54
- package/dist/commands/read/show.js +21 -2
- package/dist/commands/registry-cli.js +5 -5
- package/dist/commands/sources/add-cli.js +59 -16
- package/dist/commands/sources/bundle-cli.js +35 -11
- package/dist/commands/sources/bundle-config-ops.js +30 -0
- package/dist/commands/sources/dangerous-env-audit.js +4 -4
- package/dist/commands/sources/info.js +8 -8
- package/dist/commands/sources/installed-stashes.js +55 -61
- package/dist/commands/sources/source-add.js +39 -38
- package/dist/commands/sources/source-manage.js +34 -12
- package/dist/commands/sources/stash-cli.js +111 -119
- package/dist/commands/sources/stash-skeleton.js +6 -3
- package/dist/commands/tasks/explain.js +4 -1
- package/dist/commands/tasks/tasks-cli.js +31 -9
- package/dist/commands/tasks/tasks.js +239 -194
- package/dist/commands/tasks/validate.js +20 -32
- package/dist/core/activation-policy.js +4 -4
- package/dist/core/adapter/adapters/akm-adapter.js +8 -35
- package/dist/core/adapter/adapters/akm-metadata.js +1 -11
- package/dist/core/adapter/execution-source.js +10 -29
- package/dist/core/asset/asset-placement.js +0 -35
- package/dist/core/config/config-schema.js +64 -8
- package/dist/core/config/config-sources.js +96 -2
- package/dist/core/config/config.js +190 -24
- package/dist/core/config/legacy-source-shape-shim.js +9 -0
- package/dist/core/config/schema/embedding.js +30 -7
- package/dist/core/config/schema/execution.js +23 -0
- package/dist/core/config/schema/experimental.js +1 -1
- package/dist/core/config/schema/scheduler.js +20 -0
- package/dist/core/config/schema/search.js +10 -12
- package/dist/core/config/schema/sources-bundles.js +32 -1
- package/dist/core/content-safety.js +52 -0
- package/dist/core/errors.js +2 -5
- package/dist/core/maintenance-barrier.js +11 -13
- package/dist/core/paths.js +11 -0
- package/dist/core/run-lock.js +2 -5
- package/dist/core/state/migrations.js +1 -26
- package/dist/core/state-db.js +27 -63
- package/dist/core/type-presentation.js +1 -1
- package/dist/core/write-source.js +13 -8
- package/dist/indexer/bundle-identity-guard.js +45 -8
- package/dist/indexer/ensure-index.js +0 -5
- package/dist/indexer/index-db-contention.js +56 -0
- package/dist/indexer/index-rebuild-lock.js +73 -0
- package/dist/indexer/index-written-assets.js +171 -133
- package/dist/indexer/indexer.js +1621 -458
- package/dist/indexer/lookup/adapter-concept-owner.js +5 -19
- package/dist/indexer/materialize-embeddings.js +785 -0
- package/dist/indexer/passes/dir-staleness.js +161 -0
- package/dist/indexer/passes/metadata.js +1 -18
- package/dist/indexer/scan/drain-dir.js +70 -27
- package/dist/indexer/search/db-search.js +89 -373
- package/dist/indexer/search/ranking-contributors.js +16 -21
- package/dist/indexer/search/ranking.js +57 -135
- package/dist/indexer/search/search-source.js +29 -11
- package/dist/integrations/agent/execution-lowering.js +3 -2
- package/dist/integrations/agent/execution-preparation.js +32 -1
- package/dist/integrations/agent/prompts.js +1 -1
- package/dist/integrations/agent/request-lowering.js +3 -2
- package/dist/llm/client.js +3 -11
- package/dist/llm/embedder.js +3 -10
- package/dist/llm/embedders/remote.js +104 -133
- package/dist/llm/feature-gate.js +2 -4
- package/dist/llm/rerank-client.js +3 -3
- package/dist/output/shapes/passthrough.js +2 -1
- package/dist/output/text/command-format.js +13 -19
- package/dist/output/text/helpers.js +1 -1
- package/dist/output/text/index.js +2 -5
- package/dist/registry/resolve.js +37 -10
- package/dist/scripts/akm-migrate-node.js +15197 -11351
- package/dist/scripts/akm-migrate.js +15514 -11668
- package/dist/setup/semantic-assets.js +2 -2
- package/dist/setup/setup.js +3 -3
- package/dist/setup/steps/connection.js +2 -3
- package/dist/setup/steps/tasks.js +29 -36
- package/dist/sources/providers/git-install.js +17 -11
- package/dist/sources/providers/git-provider.js +12 -5
- package/dist/sources/providers/git-stash.js +38 -16
- package/dist/sources/snapshot-fetchers/website-ingest.js +3 -3
- package/dist/storage/repositories/embedding-salvage-repository.js +184 -0
- package/dist/storage/repositories/index-connection.js +3 -1
- package/dist/storage/repositories/index-entries-repository.js +68 -77
- package/dist/storage/repositories/index-entry-schema.js +25 -16
- package/dist/storage/repositories/index-fts-repository.js +263 -29
- package/dist/storage/repositories/index-meta-repository.js +29 -0
- package/dist/storage/repositories/index-schema.js +122 -115
- package/dist/storage/repositories/index-utility-repository.js +1 -1
- package/dist/storage/repositories/index-vec-repository.js +435 -22
- package/dist/tasks/activation-config.js +90 -0
- package/dist/tasks/backends/cron.js +9 -0
- package/dist/tasks/backends/launchd.js +1 -0
- package/dist/tasks/backends/schtasks.js +2 -0
- package/dist/tasks/embedded.js +4 -5
- package/dist/tasks/scheduler-binding.js +2 -2
- package/dist/tasks/scheduler-sync-preview.js +8 -1
- package/dist/tasks/scheduler-sync.js +19 -10
- package/dist/tasks/source/parse-task-source.js +10 -113
- package/dist/tasks/source/project-v4.js +2 -2
- package/dist/tasks/source/task-source-v4.js +4 -12
- package/dist/tasks/source/task-to-v3.js +4 -12
- package/dist/tasks/source/task-to-v4.js +40 -7
- package/docs/migration/README.md +1 -0
- package/docs/migration/release-notes/0.9.15.md +36 -34
- package/docs/migration/release-notes/0.9.16.md +60 -98
- package/docs/migration/release-notes/README.md +0 -5
- package/docs/migration/v0.9.1-to-v0.9.2.md +6 -9
- package/docs/reference/cli.md +124 -122
- package/docs/reference/configuration.md +137 -133
- package/docs/reference/data-and-telemetry.md +1 -2
- package/docs/reference/tasks.md +34 -29
- package/package.json +1 -1
- package/schemas/akm-config.json +170 -6
- package/schemas/akm-task.json +1 -2
- package/dist/commands/sources/index-status.js +0 -99
- package/dist/core/hash.js +0 -18
- package/dist/indexer/drain.js +0 -306
- package/dist/indexer/embedding-identity.js +0 -20
- package/dist/indexer/enrich.js +0 -260
- package/dist/indexer/reconcile.js +0 -890
- package/dist/indexer/scan/parse-file.js +0 -66
- package/dist/indexer/units/unit.js +0 -159
- package/dist/llm/embedders/provider-limits.js +0 -288
- package/dist/storage/repositories/files-repository.js +0 -181
- package/dist/storage/repositories/units-repository.js +0 -510
|
@@ -17,18 +17,16 @@ import { warnVerbose } from "../../core/warn.js";
|
|
|
17
17
|
import { resolveSecretFromStore } from "../../sources/snapshot-fetchers/secret-seam.js";
|
|
18
18
|
/**
|
|
19
19
|
* Upper bound on the number of documents in one HTTP request, independent of
|
|
20
|
-
* the token budget below. Overridable via `
|
|
21
|
-
*
|
|
22
|
-
*
|
|
23
|
-
*
|
|
24
|
-
* pack one request) — the token budget is what actually keeps a request
|
|
25
|
-
* inside the endpoint's context window and inside the timeout (#874).
|
|
20
|
+
* the token budget below. Overridable via `config.batchSize`. Purely a
|
|
21
|
+
* safety cap (very many tiny documents could otherwise pack one request) —
|
|
22
|
+
* the token budget is what actually keeps a request inside the endpoint's
|
|
23
|
+
* context window and inside the timeout (#874).
|
|
26
24
|
*/
|
|
27
25
|
export const DEFAULT_REMOTE_BATCH_SIZE = 100;
|
|
28
26
|
/**
|
|
29
|
-
* Conservative default token budget per HTTP request when
|
|
30
|
-
*
|
|
31
|
-
*
|
|
27
|
+
* Conservative default token budget per HTTP request when the config gives
|
|
28
|
+
* no better number (`maxTokens` — see #956 for why `contextLength`
|
|
29
|
+
* no longer feeds this). #874's measurements:
|
|
32
30
|
* a batch of 100 small docs (~400 KB, ~100K tokens) took 14.8s against a
|
|
33
31
|
* healthy local endpoint — half the 30s request timeout — and a single
|
|
34
32
|
* 128 KB (~24K token) document alone was rejected by the endpoint as
|
|
@@ -47,6 +45,38 @@ export const DEFAULT_TOKEN_BUDGET = 6000;
|
|
|
47
45
|
export function estimateTokenCount(text) {
|
|
48
46
|
return Math.round(text.length / 4);
|
|
49
47
|
}
|
|
48
|
+
/**
|
|
49
|
+
* Default per-document embedding cap (`embedding.maxInputTokens`, #956)
|
|
50
|
+
* — the materializer truncates a document's embedded text to
|
|
51
|
+
* this cap (head only) instead of skipping it outright, so one oversized
|
|
52
|
+
* entry can no longer fail a whole batch. Fragments are not embedded at all
|
|
53
|
+
* (only the entry's own search text is), so this is the only lever on how
|
|
54
|
+
* much of a large document contributes to its vector.
|
|
55
|
+
*/
|
|
56
|
+
export const DEFAULT_MAX_INPUT_TOKENS = 512;
|
|
57
|
+
/**
|
|
58
|
+
* Truncate `text` to at most `maxTokens` (estimated via
|
|
59
|
+
* {@link estimateTokenCount}, the same 4-chars≈1-token rule the batching
|
|
60
|
+
* budget uses), keeping only its head. The cut never splits a UTF-16
|
|
61
|
+
* surrogate pair. Text already at or under the cap is returned unchanged
|
|
62
|
+
* (`truncated: false`) — including empty text, which is never itself
|
|
63
|
+
* "truncated".
|
|
64
|
+
*/
|
|
65
|
+
export function capEmbeddingText(text, maxTokens) {
|
|
66
|
+
if (estimateTokenCount(text) <= maxTokens)
|
|
67
|
+
return { text, truncated: false };
|
|
68
|
+
const charBudget = Math.max(0, maxTokens * 4);
|
|
69
|
+
let cut = Math.min(charBudget, text.length);
|
|
70
|
+
if (cut > 0 && cut < text.length) {
|
|
71
|
+
const code = text.charCodeAt(cut);
|
|
72
|
+
// A low surrogate (0xDC00-0xDFFF) at the cut point means its high
|
|
73
|
+
// surrogate is the character just before it — back off one position so
|
|
74
|
+
// the pair stays together rather than yielding a lone surrogate.
|
|
75
|
+
if (code >= 0xdc00 && code <= 0xdfff)
|
|
76
|
+
cut -= 1;
|
|
77
|
+
}
|
|
78
|
+
return { text: text.slice(0, cut), truncated: true };
|
|
79
|
+
}
|
|
50
80
|
/**
|
|
51
81
|
* Default per-request timeout when `embedding.timeoutMs` is unset (#954).
|
|
52
82
|
* The prior fixed 30s cut off exactly the field-report case: a
|
|
@@ -164,10 +194,9 @@ export function isContextExceededResponse(status, body) {
|
|
|
164
194
|
* either direction, bounded 1-16 at the config schema — added after field
|
|
165
195
|
* evidence that a multi-slot local server (llama.cpp `--parallel N`, vLLM)
|
|
166
196
|
* genuinely serves parallel requests and was left idle by the fixed default.
|
|
167
|
-
* Request SIZE remains the first throughput lever regardless:
|
|
168
|
-
*
|
|
169
|
-
*
|
|
170
|
-
* `maxTokens`/`contextLength` config keys' replacement, index redesign B5)
|
|
197
|
+
* Request SIZE remains the first throughput lever regardless:
|
|
198
|
+
* `embedding.batchSize` (document cap) and `embedding.maxTokens` (request
|
|
199
|
+
* token budget — see #956; `contextLength` no longer feeds it)
|
|
171
200
|
* reach a larger batch per request, which is where most of the win is for a
|
|
172
201
|
* single-slot server — a 32-input batch takes about the same wall time as
|
|
173
202
|
* one input against a healthy endpoint.
|
|
@@ -178,22 +207,15 @@ export function resolveEmbeddingConcurrency(config) {
|
|
|
178
207
|
return defaultConcurrencyForEndpoint(config.endpoint);
|
|
179
208
|
}
|
|
180
209
|
/**
|
|
181
|
-
* Group `texts` into request-sized batches bounded by BOTH
|
|
182
|
-
* a document-count cap, so one large document does not
|
|
183
|
-
* batch past the endpoint's context window (#874).
|
|
184
|
-
*
|
|
185
|
-
* `tokenCounts[i]`, when given, is the count to use for `texts[i]` instead of
|
|
186
|
-
* {@link estimateTokenCount}'s fixed 4-chars≈1-token guess — `embedBatch`
|
|
187
|
-
* passes the calibrated `charsPerToken` estimate for every text
|
|
188
|
-
* (`EmbeddingRequestPacking.charsPerToken`, sourced from the provider's own
|
|
189
|
-
* probed limits). Omitted (or shorter than `texts`, e.g. a caller with no
|
|
190
|
-
* packing at all) falls back to the estimate for the texts it does not cover.
|
|
210
|
+
* Group `texts` into request-sized batches bounded by BOTH an estimated
|
|
211
|
+
* token budget and a document-count cap, so one large document does not
|
|
212
|
+
* silently blow the batch past the endpoint's context window (#874).
|
|
191
213
|
*
|
|
192
|
-
* A single document whose own
|
|
193
|
-
* batch — it is reported as its own oversized "batch" so the caller can
|
|
194
|
-
* it without ever making an HTTP request for it.
|
|
214
|
+
* A single document whose own estimate exceeds `tokenBudget` can never fit
|
|
215
|
+
* any batch — it is reported as its own oversized "batch" so the caller can
|
|
216
|
+
* skip it without ever making an HTTP request for it.
|
|
195
217
|
*/
|
|
196
|
-
export function buildTokenBoundedBatches(texts, tokenBudget, maxCount
|
|
218
|
+
export function buildTokenBoundedBatches(texts, tokenBudget, maxCount) {
|
|
197
219
|
const batches = [];
|
|
198
220
|
let current = [];
|
|
199
221
|
let currentTokens = 0;
|
|
@@ -205,7 +227,7 @@ export function buildTokenBoundedBatches(texts, tokenBudget, maxCount, tokenCoun
|
|
|
205
227
|
}
|
|
206
228
|
};
|
|
207
229
|
for (let i = 0; i < texts.length; i++) {
|
|
208
|
-
const tokens =
|
|
230
|
+
const tokens = estimateTokenCount(texts[i]);
|
|
209
231
|
if (tokens > tokenBudget) {
|
|
210
232
|
flush();
|
|
211
233
|
batches.push({ indices: [i], oversized: true });
|
|
@@ -225,20 +247,16 @@ export function buildTokenBoundedBatches(texts, tokenBudget, maxCount, tokenCoun
|
|
|
225
247
|
* context-size rejection of an `embedBatch` run (#954, field report on
|
|
226
248
|
* beta.1): one 25% cut absorbs the estimator's measured undercount without
|
|
227
249
|
* repeatedly re-shrinking mid-run — see the "shrink at most once" rule on
|
|
228
|
-
* {@link RemoteEmbedder.embedBatch}.
|
|
229
|
-
* is not already authoritative (`packing.windowIsKnown` false) — see
|
|
230
|
-
* {@link EmbeddingRequestPacking}.
|
|
250
|
+
* {@link RemoteEmbedder.embedBatch}.
|
|
231
251
|
*/
|
|
232
252
|
const ADAPTIVE_BUDGET_SHRINK_FACTOR = 0.75;
|
|
233
253
|
/**
|
|
234
|
-
* Floor on the adaptive-budget shrink above
|
|
235
|
-
*
|
|
236
|
-
* longer
|
|
237
|
-
*
|
|
238
|
-
* could no longer batch more than a couple of average-sized units per
|
|
239
|
-
* request, defeating the point of batching at all.
|
|
254
|
+
* Floor on the adaptive-budget shrink above, as a multiple of
|
|
255
|
+
* `maxInputTokens` (#954): a request budget below twice the per-document cap
|
|
256
|
+
* could no longer batch more than one document per request, defeating the
|
|
257
|
+
* point of batching at all.
|
|
240
258
|
*/
|
|
241
|
-
const
|
|
259
|
+
const ADAPTIVE_BUDGET_FLOOR_MULTIPLIER = 2;
|
|
242
260
|
export class RemoteEmbedder {
|
|
243
261
|
config;
|
|
244
262
|
endpoint;
|
|
@@ -264,9 +282,6 @@ export class RemoteEmbedder {
|
|
|
264
282
|
if (ollamaOpts) {
|
|
265
283
|
body.options = ollamaOpts;
|
|
266
284
|
}
|
|
267
|
-
if (isOllamaNativeEmbedEndpoint(this.endpoint)) {
|
|
268
|
-
body.truncate = false;
|
|
269
|
-
}
|
|
270
285
|
const timeoutMs = resolveEmbeddingTimeoutMs(this.config);
|
|
271
286
|
// `signal` MUST go through fetchWithTimeout's dedicated 4th parameter, not
|
|
272
287
|
// the RequestInit: fetchWithTimeout replaces `opts.signal` with its own
|
|
@@ -347,47 +362,40 @@ export class RemoteEmbedder {
|
|
|
347
362
|
* on a small batch rather than always waiting out the full configured
|
|
348
363
|
* `embedding.timeoutMs`.
|
|
349
364
|
*
|
|
350
|
-
* Run-scoped adaptive budget (#954, field report on beta.1
|
|
351
|
-
*
|
|
352
|
-
*
|
|
353
|
-
*
|
|
354
|
-
* — the
|
|
355
|
-
*
|
|
356
|
-
* budget
|
|
357
|
-
*
|
|
358
|
-
*
|
|
359
|
-
*
|
|
360
|
-
*
|
|
361
|
-
* (llama.cpp/Ollama, probed via `probeProviderLimits`) is already
|
|
362
|
-
* authoritative, so a rejection against it is unexpected — split-and-retry
|
|
363
|
-
* (above) still recovers that one batch, but the run-wide budget is left
|
|
364
|
-
* alone rather than second-guessing a real number. This never touches the
|
|
365
|
-
* split-and-retry of the rejected batch itself, and never fires a second
|
|
366
|
-
* time in the same run even if a later batch is also rejected — a budget
|
|
367
|
-
* that is simply too big for the endpoint should self-correct once, not
|
|
368
|
-
* ratchet down forever.
|
|
365
|
+
* Run-scoped adaptive budget (#954, field report on beta.1): the FIRST
|
|
366
|
+
* context-size rejection of the run shrinks the effective request budget
|
|
367
|
+
* by {@link ADAPTIVE_BUDGET_SHRINK_FACTOR} (floored at
|
|
368
|
+
* {@link ADAPTIVE_BUDGET_FLOOR_MULTIPLIER} times `maxInputTokens`) for
|
|
369
|
+
* every batch not yet dispatched — the still-planned tail of `texts` is
|
|
370
|
+
* re-batched with `buildTokenBoundedBatches` at the smaller budget, and a
|
|
371
|
+
* `budget-lowered` `onBatch` event reports it once. This never touches the
|
|
372
|
+
* split-and-retry of the rejected batch itself (above), and never fires a
|
|
373
|
+
* second time in the same run even if a later batch is also rejected — a
|
|
374
|
+
* static configured budget that is simply too big for the endpoint should
|
|
375
|
+
* self-correct once, not ratchet down forever.
|
|
369
376
|
*/
|
|
370
|
-
async embedBatch(texts, signal, onSkip, onBatch
|
|
377
|
+
async embedBatch(texts, signal, onSkip, onBatch) {
|
|
371
378
|
if (texts.length === 0)
|
|
372
379
|
return [];
|
|
373
380
|
const results = new Array(texts.length).fill(undefined);
|
|
374
381
|
const headers = this.buildHeaders();
|
|
375
|
-
const ollamaOpts = resolveOllamaOptions(this.config
|
|
376
|
-
//
|
|
377
|
-
//
|
|
378
|
-
//
|
|
379
|
-
//
|
|
380
|
-
|
|
381
|
-
//
|
|
382
|
-
//
|
|
383
|
-
//
|
|
384
|
-
//
|
|
382
|
+
const ollamaOpts = resolveOllamaOptions(this.config);
|
|
383
|
+
// #956: `contextLength` is Ollama's `num_ctx` ONLY (see
|
|
384
|
+
// resolveOllamaOptions below) — it used to double as this client-side
|
|
385
|
+
// request budget too, so a config author setting it for one purpose
|
|
386
|
+
// silently changed the other. `maxTokens` is the sole knob for the
|
|
387
|
+
// request budget now; unset falls back to DEFAULT_TOKEN_BUDGET.
|
|
388
|
+
//
|
|
389
|
+
// `effectiveTokenBudget` (#954) starts at the configured/default value
|
|
390
|
+
// and MAY shrink once, on the run's first context-size rejection — see
|
|
391
|
+
// `maybeShrinkBudget` below. `textBatches` is mutated in place (spliced)
|
|
392
|
+
// by that shrink rather than reassigned, so the in-flight
|
|
385
393
|
// `concurrentMap` pool below (which reads this same array by reference)
|
|
386
394
|
// picks up the re-planned tail without restarting.
|
|
387
|
-
let effectiveTokenBudget =
|
|
388
|
-
const maxCount =
|
|
389
|
-
const
|
|
390
|
-
const textBatches = buildTokenBoundedBatches(texts, effectiveTokenBudget, maxCount
|
|
395
|
+
let effectiveTokenBudget = this.config.maxTokens ?? DEFAULT_TOKEN_BUDGET;
|
|
396
|
+
const maxCount = this.config.batchSize ?? DEFAULT_REMOTE_BATCH_SIZE;
|
|
397
|
+
const maxInputTokens = this.config.maxInputTokens ?? DEFAULT_MAX_INPUT_TOKENS;
|
|
398
|
+
const textBatches = buildTokenBoundedBatches(texts, effectiveTokenBudget, maxCount);
|
|
391
399
|
const configuredTimeoutMs = resolveEmbeddingTimeoutMs(this.config);
|
|
392
400
|
// How many of `textBatches` concurrentMap has already claimed (its own
|
|
393
401
|
// `nextIndex`, mirrored here so a budget shrink knows where the
|
|
@@ -396,28 +404,24 @@ export class RemoteEmbedder {
|
|
|
396
404
|
// order, so the highest `batchIndex` seen so far IS the claimed count.
|
|
397
405
|
let dispatchedBatchCount = 0;
|
|
398
406
|
// Set once the run's first context-size rejection has shrunk the budget
|
|
399
|
-
// (#954) — guards `maybeShrinkBudget` so it never fires twice.
|
|
400
|
-
|
|
401
|
-
// see the method's doc comment.
|
|
402
|
-
let budgetShrunk = windowIsKnown;
|
|
407
|
+
// (#954) — guards `maybeShrinkBudget` so it never fires twice.
|
|
408
|
+
let budgetShrunk = false;
|
|
403
409
|
// On the FIRST context-size rejection of this `embedBatch` call, shrink
|
|
404
410
|
// `effectiveTokenBudget` and re-plan every batch `concurrentMap` has not
|
|
405
411
|
// yet claimed from the smaller budget. Never touches `rejectedIndices`
|
|
406
412
|
// itself — the caller's own split-and-retry handles that batch — and is
|
|
407
|
-
// a no-op after the first call (`budgetShrunk`)
|
|
408
|
-
// already known (`windowIsKnown`, folded into `budgetShrunk`'s initial
|
|
409
|
-
// value above).
|
|
413
|
+
// a no-op after the first call (`budgetShrunk`).
|
|
410
414
|
const maybeShrinkBudget = (rejectedIndices, rejectedBatchIndex, rejectedRequestTokens) => {
|
|
411
415
|
if (budgetShrunk)
|
|
412
416
|
return;
|
|
413
417
|
budgetShrunk = true;
|
|
414
|
-
|
|
418
|
+
const floor = ADAPTIVE_BUDGET_FLOOR_MULTIPLIER * maxInputTokens;
|
|
419
|
+
effectiveTokenBudget = Math.max(Math.round(effectiveTokenBudget * ADAPTIVE_BUDGET_SHRINK_FACTOR), floor);
|
|
415
420
|
const notYetDispatched = textBatches.slice(dispatchedBatchCount);
|
|
416
421
|
const remainingIndices = notYetDispatched.flatMap((batch) => batch.indices);
|
|
417
422
|
if (remainingIndices.length > 0) {
|
|
418
423
|
const remainingTexts = remainingIndices.map((i) => texts[i]);
|
|
419
|
-
const
|
|
420
|
-
const replanned = buildTokenBoundedBatches(remainingTexts, effectiveTokenBudget, maxCount, remainingCounts).map((batch) => ({
|
|
424
|
+
const replanned = buildTokenBoundedBatches(remainingTexts, effectiveTokenBudget, maxCount).map((batch) => ({
|
|
421
425
|
indices: batch.indices.map((localIndex) => remainingIndices[localIndex]),
|
|
422
426
|
oversized: batch.oversized,
|
|
423
427
|
}));
|
|
@@ -435,7 +439,7 @@ export class RemoteEmbedder {
|
|
|
435
439
|
});
|
|
436
440
|
};
|
|
437
441
|
// Stops the pool from claiming any FURTHER provider batch once the
|
|
438
|
-
// caller's onBatch has failed once (the
|
|
442
|
+
// caller's onBatch has failed once (the materializer's transaction
|
|
439
443
|
// failed, so a subsequent commit would just fail again) — dispatching
|
|
440
444
|
// real HTTP requests whose results can never be persisted is pure waste.
|
|
441
445
|
// Deliberately a SEPARATE controller from the caller's own `signal`,
|
|
@@ -459,7 +463,7 @@ export class RemoteEmbedder {
|
|
|
459
463
|
}
|
|
460
464
|
}
|
|
461
465
|
// First error thrown BY the caller's onBatch callback (e.g. a real
|
|
462
|
-
// competing-process SQLITE_BUSY from the
|
|
466
|
+
// competing-process SQLITE_BUSY from the materializer's db.transaction())
|
|
463
467
|
// rather than by requestBatch itself. Captured here instead of being left
|
|
464
468
|
// to reach requestAndCommit's try/catch below, which exists solely to
|
|
465
469
|
// classify requestBatch's own provider/network failures — a persistence
|
|
@@ -515,7 +519,7 @@ export class RemoteEmbedder {
|
|
|
515
519
|
if (dispatchAbort.signal.aborted)
|
|
516
520
|
return;
|
|
517
521
|
const batch = indices.map((i) => texts[i]);
|
|
518
|
-
const requestTokens =
|
|
522
|
+
const requestTokens = batch.reduce((sum, text) => sum + estimateTokenCount(text), 0);
|
|
519
523
|
const requestTimeoutMs = scaleEmbeddingTimeoutMs(configuredTimeoutMs, requestTokens, effectiveTokenBudget);
|
|
520
524
|
const requestStart = Date.now();
|
|
521
525
|
let batchEmbeddings;
|
|
@@ -565,8 +569,8 @@ export class RemoteEmbedder {
|
|
|
565
569
|
// verbose line above — a run silently waiting out a multi-minute
|
|
566
570
|
// back-off looked identical to a hang otherwise. Nothing has
|
|
567
571
|
// failed or succeeded yet, so there is nothing to persist:
|
|
568
|
-
// `embeddings` are all `undefined` and the
|
|
569
|
-
// not touch storage for this event.
|
|
572
|
+
// `embeddings` are all `undefined` and the materializer's onBatch
|
|
573
|
+
// must not touch storage for this event.
|
|
570
574
|
commitBatch(indices, indices.map(() => undefined), undefined, {
|
|
571
575
|
batchIndex,
|
|
572
576
|
batchCount: textBatches.length,
|
|
@@ -601,12 +605,13 @@ export class RemoteEmbedder {
|
|
|
601
605
|
// Default-level visibility for a failed batch (not verbose-only) is
|
|
602
606
|
// still guaranteed here — just not via warn(). The `commitBatch` call
|
|
603
607
|
// below carries `outcome: "failed"` and this `message` as `reason`
|
|
604
|
-
// through `onBatch`, and
|
|
608
|
+
// through `onBatch`, and materialize-embeddings.ts's per-batch line
|
|
605
609
|
// (also default-level) prints it from there. A warn() call here used
|
|
606
610
|
// to print the identical event a second time on stderr — the same
|
|
607
|
-
// class of double-print bug
|
|
608
|
-
// materialize-embeddings.ts (#954, field-report follow-up).
|
|
609
|
-
// Per-entry batch-mapping detail stays verbose-only
|
|
611
|
+
// class of double-print bug fixed for the truncation/re-embed-reason
|
|
612
|
+
// lines in materialize-embeddings.ts (#954, field-report follow-up).
|
|
613
|
+
// Per-entry batch-mapping detail stays verbose-only
|
|
614
|
+
// (materialize-embeddings.ts).
|
|
610
615
|
let stopRequested = false;
|
|
611
616
|
for (const [k, idx] of indices.entries()) {
|
|
612
617
|
if (onSkip?.({
|
|
@@ -649,7 +654,7 @@ export class RemoteEmbedder {
|
|
|
649
654
|
dispatchedBatchCount = batchIndex;
|
|
650
655
|
if (textBatch.oversized) {
|
|
651
656
|
const idx = textBatch.indices[0];
|
|
652
|
-
const estTokens =
|
|
657
|
+
const estTokens = estimateTokenCount(texts[idx]);
|
|
653
658
|
onSkip?.({
|
|
654
659
|
index: idx,
|
|
655
660
|
reason: "context-window-exceeded",
|
|
@@ -719,9 +724,6 @@ export class RemoteEmbedder {
|
|
|
719
724
|
if (ollamaOpts) {
|
|
720
725
|
body.options = ollamaOpts;
|
|
721
726
|
}
|
|
722
|
-
if (isOllamaNativeEmbedEndpoint(this.endpoint)) {
|
|
723
|
-
body.truncate = false;
|
|
724
|
-
}
|
|
725
727
|
// See embed(): `signal` goes through the 4th parameter, not the
|
|
726
728
|
// RequestInit, or fetchWithTimeout drops it.
|
|
727
729
|
const response = await fetchWithTimeout(normalizeEmbeddingEndpoint(this.endpoint), {
|
|
@@ -814,34 +816,6 @@ export function normalizeEmbeddingEndpoint(endpoint) {
|
|
|
814
816
|
parsed.pathname = normalizedPath ? `${normalizedPath}/embeddings` : "/embeddings";
|
|
815
817
|
return parsed.toString();
|
|
816
818
|
}
|
|
817
|
-
/**
|
|
818
|
-
* True when `endpoint`'s normalized path is Ollama's native `/api/embed`
|
|
819
|
-
* route (see {@link normalizeEmbeddingEndpoint}) rather than an
|
|
820
|
-
* OpenAI-compatible `/embeddings` route. Gates `truncate: false` on the
|
|
821
|
-
* request body (round-2 field finding): akm never sent `truncate` at all, so
|
|
822
|
-
* Ollama's default — silently truncate an over-budget input and still return
|
|
823
|
-
* 200 — meant a unit denser than the calibrated chars-per-token ratio was
|
|
824
|
-
* embedded from a truncated prefix and stored as a complete, correct-looking
|
|
825
|
-
* vector: never counted `failed` or `skipped`, coverage reporting it done,
|
|
826
|
-
* that content's search quality silently degraded forever. `truncate: false`
|
|
827
|
-
* makes an over-budget request fail loudly instead, so it flows into the
|
|
828
|
-
* existing context-window handling (split-and-retry, ultimately a genuine
|
|
829
|
-
* `skipped` unit) rather than a silent truncation. Scoped to the native
|
|
830
|
-
* route specifically because that is the one shape this field evidence is
|
|
831
|
-
* about — an OpenAI-compatible endpoint ignores the unknown field either
|
|
832
|
-
* way, so this is not a safety boundary, just not sending an option that
|
|
833
|
-
* does nothing elsewhere.
|
|
834
|
-
*/
|
|
835
|
-
function isOllamaNativeEmbedEndpoint(endpoint) {
|
|
836
|
-
let parsed;
|
|
837
|
-
try {
|
|
838
|
-
parsed = new URL(normalizeEmbeddingEndpoint(endpoint));
|
|
839
|
-
}
|
|
840
|
-
catch {
|
|
841
|
-
return false;
|
|
842
|
-
}
|
|
843
|
-
return parsed.pathname.replace(/\/+$/, "").endsWith("/embed");
|
|
844
|
-
}
|
|
845
819
|
function embeddingEndpointPathHint(endpoint) {
|
|
846
820
|
const normalizedEndpoint = normalizeEmbeddingEndpoint(endpoint);
|
|
847
821
|
if (normalizedEndpoint !== endpoint) {
|
|
@@ -854,22 +828,19 @@ function embeddingEndpointPathHint(endpoint) {
|
|
|
854
828
|
*
|
|
855
829
|
* Resolution order:
|
|
856
830
|
* 1. `ollamaOptions` — forwarded verbatim (explicit opt-in, takes precedence).
|
|
857
|
-
* 2. `
|
|
858
|
-
* window (`ProviderLimits.ollamaNumCtx`, sourced from
|
|
859
|
-
* `probeProviderLimits`, NOT the retired `embedding.contextLength`
|
|
860
|
-
* config key), wrapped as `{ num_ctx: ollamaNumCtx }`.
|
|
831
|
+
* 2. `contextLength` — wrapped as `{ num_ctx: contextLength }`.
|
|
861
832
|
* 3. Neither set → returns `undefined` (no `options` field in the request body).
|
|
862
833
|
*
|
|
863
834
|
* These options are only meaningful for Ollama's native `/api/embed` endpoint.
|
|
864
835
|
* OpenAI-compatible endpoints ignore unknown request fields, so passing them to
|
|
865
836
|
* other providers is harmless but has no effect.
|
|
866
837
|
*/
|
|
867
|
-
function resolveOllamaOptions(config
|
|
838
|
+
function resolveOllamaOptions(config) {
|
|
868
839
|
if (config.ollamaOptions && Object.keys(config.ollamaOptions).length > 0) {
|
|
869
840
|
return config.ollamaOptions;
|
|
870
841
|
}
|
|
871
|
-
if (
|
|
872
|
-
return { num_ctx:
|
|
842
|
+
if (config.contextLength) {
|
|
843
|
+
return { num_ctx: config.contextLength };
|
|
873
844
|
}
|
|
874
845
|
return undefined;
|
|
875
846
|
}
|
package/dist/llm/feature-gate.js
CHANGED
|
@@ -10,11 +10,9 @@ const FEATURE_LOCATION = {
|
|
|
10
10
|
graph_extraction: (cfg) => cfg.index?.graph?.enabled ?? true,
|
|
11
11
|
metadata_enhance: (cfg) => cfg.index?.metadataEnhance?.enabled ?? false,
|
|
12
12
|
// #951: a real implementation of the dead `curate_rerank` key removed in
|
|
13
|
-
// 0.8.0
|
|
14
|
-
// (renamed `curate_rerank` → `search_rerank`; the pass was always meant for
|
|
15
|
-
// search). Off by default — it requires a `search.rerank.endpoint` a
|
|
13
|
+
// 0.8.0. Off by default — it requires a `search.curateRerank.endpoint` a
|
|
16
14
|
// caller must explicitly configure.
|
|
17
|
-
|
|
15
|
+
curate_rerank: (cfg) => Boolean(cfg.search?.curateRerank?.enabled),
|
|
18
16
|
// Always on at the LLM-wrapper level. Enablement is decided ONCE at the
|
|
19
17
|
// extract entry point (`akmExtract`): the `extract.enabled` process toggle
|
|
20
18
|
// gates extract as a STAGE of `akm improve` (the active improve strategy, per
|
|
@@ -21,7 +21,7 @@
|
|
|
21
21
|
* `relevance_score` descending and callers treat a missing index as
|
|
22
22
|
* "unscored" (kept in its original relative position, after every scored
|
|
23
23
|
* document). Deliberately independent of `EngineConfigSchema`'s "llm"/"agent"
|
|
24
|
-
* kinds — see the comment on `
|
|
24
|
+
* kinds — see the comment on `CurateRerankConfigSchema` in
|
|
25
25
|
* `core/config/schema/search.ts` for why.
|
|
26
26
|
*/
|
|
27
27
|
import { fetchWithTimeout, readBodyWithByteCap } from "../core/common.js";
|
|
@@ -45,13 +45,13 @@ export class RerankCallError extends Error {
|
|
|
45
45
|
* after every scored document (never dropped).
|
|
46
46
|
*
|
|
47
47
|
* Throws {@link RerankCallError} on any transport/parse failure — callers
|
|
48
|
-
* that want a graceful fallback should use `tryLlmFeature("
|
|
48
|
+
* that want a graceful fallback should use `tryLlmFeature("curate_rerank", ...)`
|
|
49
49
|
* (`llm/feature-gate.ts`), matching every other bounded in-tree LLM/rerank
|
|
50
50
|
* call site.
|
|
51
51
|
*/
|
|
52
52
|
export async function rerankDocuments(config, query, documents) {
|
|
53
53
|
if (!config.endpoint) {
|
|
54
|
-
throw new RerankCallError("search.
|
|
54
|
+
throw new RerankCallError("search.curateRerank.endpoint is not configured.", "provider_error");
|
|
55
55
|
}
|
|
56
56
|
if (documents.length === 0)
|
|
57
57
|
return [];
|
|
@@ -54,7 +54,6 @@ const PASSTHROUGH_COMMANDS = [
|
|
|
54
54
|
"improve-report",
|
|
55
55
|
"import",
|
|
56
56
|
"index",
|
|
57
|
-
"index-status",
|
|
58
57
|
"info",
|
|
59
58
|
"lint",
|
|
60
59
|
"list",
|
|
@@ -72,7 +71,9 @@ const PASSTHROUGH_COMMANDS = [
|
|
|
72
71
|
"setup",
|
|
73
72
|
"sync",
|
|
74
73
|
"task-add",
|
|
74
|
+
"task-disable",
|
|
75
75
|
"task-doctor",
|
|
76
|
+
"task-enable",
|
|
76
77
|
"task-explain",
|
|
77
78
|
"task-history",
|
|
78
79
|
"task-prune",
|
|
@@ -422,13 +422,21 @@ export function formatInitPlain(r) {
|
|
|
422
422
|
}
|
|
423
423
|
export function formatIndexPlain(r) {
|
|
424
424
|
const indexResult = r;
|
|
425
|
-
let out = `Indexed ${indexResult.totalEntries ?? 0} entries from ${indexResult.
|
|
425
|
+
let out = `Indexed ${indexResult.totalEntries ?? 0} entries from ${indexResult.directoriesScanned ?? 0} directories (mode: ${indexResult.mode ?? "unknown"})`;
|
|
426
426
|
const warnings = indexResult.warnings;
|
|
427
427
|
if (Array.isArray(warnings) && warnings.length > 0) {
|
|
428
428
|
out += `\nWarnings (${warnings.length}):`;
|
|
429
429
|
for (const message of warnings)
|
|
430
430
|
out += `\n - ${String(message)}`;
|
|
431
431
|
}
|
|
432
|
+
const notices = Array.isArray(indexResult.notices) ? indexResult.notices : [];
|
|
433
|
+
for (const notice of notices) {
|
|
434
|
+
const severity = notice.severity === "info" ? "info" : "warning";
|
|
435
|
+
const field = typeof notice.field === "string" ? ` field=${notice.field}` : "";
|
|
436
|
+
out +=
|
|
437
|
+
`\n notice[${severity}] ${notice.code} adapter=${notice.adapter}${field}` +
|
|
438
|
+
(notice.message ? `: ${notice.message}` : "");
|
|
439
|
+
}
|
|
432
440
|
const verification = indexResult.verification;
|
|
433
441
|
if (verification?.ok === false && verification.message) {
|
|
434
442
|
out += `\nVerification: ${String(verification.message)}`;
|
|
@@ -438,30 +446,16 @@ export function formatIndexPlain(r) {
|
|
|
438
446
|
out +=
|
|
439
447
|
`\nTiming: total ${timing.totalMs}ms` +
|
|
440
448
|
`, preflight ${timing.preflightMs}ms` +
|
|
441
|
-
`,
|
|
442
|
-
`,
|
|
449
|
+
`, walk ${timing.walkMs}ms` +
|
|
450
|
+
`, llm ${timing.llmMs}ms` +
|
|
443
451
|
`, embeddings ${timing.embedMs}ms` +
|
|
452
|
+
`, fts ${timing.ftsMs}ms` +
|
|
444
453
|
`, finalize ${timing.finalizeMs}ms` +
|
|
454
|
+
`, clean ${timing.cleanMs}ms` +
|
|
445
455
|
`, end-to-end ${timing.endToEndMs}ms`;
|
|
446
456
|
}
|
|
447
457
|
return out;
|
|
448
458
|
}
|
|
449
|
-
/** Render `akm index status`'s `IndexStatusResponse` (src/commands/sources/index-status.ts). */
|
|
450
|
-
export function formatIndexStatusPlain(r) {
|
|
451
|
-
const units = (r.units ?? {});
|
|
452
|
-
const lines = [
|
|
453
|
-
`Index: ${String(r.indexPath ?? "unknown")}`,
|
|
454
|
-
`Files: ${Number(r.files ?? 0)}`,
|
|
455
|
-
`Entries: ${Number(r.entries ?? 0)}`,
|
|
456
|
-
`Units: ${Number(units.total ?? 0)} total, ${Number(units.withVector ?? 0)} with a vector, ${Number(units.pending ?? 0)} pending`,
|
|
457
|
-
`Active identity: ${typeof r.activeIdentity === "string" ? r.activeIdentity : "none yet"}`,
|
|
458
|
-
`Last reconcile: ${typeof r.lastReconcileAt === "string" ? r.lastReconcileAt : "never"}`,
|
|
459
|
-
`Built at: ${typeof r.builtAt === "string" ? r.builtAt : "never"}`,
|
|
460
|
-
];
|
|
461
|
-
if (typeof r.unreadable === "string")
|
|
462
|
-
lines.push(`Unreadable: ${r.unreadable}`);
|
|
463
|
-
return lines.join("\n");
|
|
464
|
-
}
|
|
465
459
|
export function formatListPlain(r) {
|
|
466
460
|
const sources = Array.isArray(r.sources) ? r.sources : [];
|
|
467
461
|
if (sources.length === 0)
|
|
@@ -15,7 +15,7 @@
|
|
|
15
15
|
*
|
|
16
16
|
* No registry imports — no circular dependencies.
|
|
17
17
|
*/
|
|
18
|
-
export { formatAddPlain, formatBundleShowPlain, formatClonePlain, formatConfigPlain, formatCuratePlain, formatEnvCreatePlain, formatEnvExportPlain, formatEnvListPlain, formatEnvRemovePlain, formatEventLine, formatEventsPlain, formatFeedbackPlain, formatImportPlain, formatIndexPlain,
|
|
18
|
+
export { formatAddPlain, formatBundleShowPlain, formatClonePlain, formatConfigPlain, formatCuratePlain, formatEnvCreatePlain, formatEnvExportPlain, formatEnvListPlain, formatEnvRemovePlain, formatEventLine, formatEventsPlain, formatFeedbackPlain, formatImportPlain, formatIndexPlain, formatInfoPlain, formatInitPlain, formatListPlain, formatModelsListPlain, formatRegistryAddPlain, formatRegistryListPlain, formatRegistryRemovePlain, formatRegistrySearchPlain, formatRememberPlain, formatRemovePlain, formatSearchPlain, formatSyncPlain, formatUpdatePlain, formatUpgradePlain, } from "./command-format.js";
|
|
19
19
|
export { formatHealthPlain } from "./health-format.js";
|
|
20
20
|
export { formatLintPlain } from "./lint-format.js";
|
|
21
21
|
export { formatGateDecisionSummary, formatProposalAcceptPlain, formatProposalDiffPlain, formatProposalDrainPlain, formatProposalListPlain, formatProposalProducerPlain, formatProposalRejectPlain, formatProposalShowPlain, } from "./proposal-format.js";
|
|
@@ -1,8 +1,5 @@
|
|
|
1
1
|
// This Source Code Form is subject to the terms of the Mozilla Public
|
|
2
2
|
// License, v. 2.0. If a copy of the MPL was not distributed with this
|
|
3
3
|
// file, You can obtain one at https://mozilla.org/MPL/2.0/.
|
|
4
|
-
import { formatIndexPlain
|
|
5
|
-
export const indexFormatters = [
|
|
6
|
-
{ command: "index", handler: (r) => formatIndexPlain(r) },
|
|
7
|
-
{ command: "index-status", handler: (r) => formatIndexStatusPlain(r) },
|
|
8
|
-
];
|
|
4
|
+
import { formatIndexPlain } from "./helpers.js";
|
|
5
|
+
export const indexFormatters = [{ command: "index", handler: (r) => formatIndexPlain(r) }];
|