akm-cli 0.9.15-beta.3 → 0.9.15-beta.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -29,7 +29,7 @@ import { parseFrontmatter } from "../../core/asset/frontmatter.js";
29
29
  import { conceptIdFromTypeName, parseRefInput } from "../../core/asset/resolve-ref.js";
30
30
  import { DESCRIPTION_MAX_CHARS, requiresDescription } from "../../core/authoring-rules.js";
31
31
  import { loadConfig } from "../../core/config/config.js";
32
- import { ConfigError } from "../../core/errors.js";
32
+ import { ConfigError, UsageError } from "../../core/errors.js";
33
33
  import { appendEvent, readEvents } from "../../core/events.js";
34
34
  import { lintLessonContent } from "../../core/lesson-lint.js";
35
35
  import { parseEmbeddedJsonResponse } from "../../core/parse.js";
@@ -1371,6 +1371,92 @@ async function resolveReflectSource(options, stash, emitReflectFailed) {
1371
1371
  }
1372
1372
  return { assetContent, parsedRef };
1373
1373
  }
1374
+ /**
1375
+ * #952 — the flat REFLECT_CONTENT_CAP (12 000 chars) exists only to avoid
1376
+ * E2BIG when the prompt travels through CLI argv (agent/SDK runners). The
1377
+ * direct-LLM HTTP path never touches argv, so it can use the resolved
1378
+ * engine's own context window instead. The reserve for "the rest of the
1379
+ * prompt" is measured directly (not guessed): build the same prompt with
1380
+ * the content cap forced to zero and use its length as the overhead, so
1381
+ * feedback/standards/schema-hints/prior-draft size is accounted for
1382
+ * exactly, per this call. A reflect rewrite returns a body roughly the
1383
+ * size of the input, so the budget only spends HALF of the usable window
1384
+ * on input content and reserves the other half for the model's own
1385
+ * output — otherwise a full-context request leaves no room for a
1386
+ * response. Never drops below the flat floor.
1387
+ *
1388
+ * Shared by the real dispatch path ({@link runReflectRefineIterations}) and
1389
+ * `renderReflectPromptPreview`'s `--show-prompt` preview, so the preview
1390
+ * renders the exact prompt reflect would actually send for LLM runners
1391
+ * instead of always the flat-cap prompt.
1392
+ */
1393
+ function computeReflectContentBudgetChars(promptInput, runnerSpec) {
1394
+ return runnerIsLlm(runnerSpec) && promptInput.assetContent?.trim()
1395
+ ? Math.max(REFLECT_CONTENT_CAP, Math.floor(((runnerSpec.connection.contextLength ?? DEFAULT_CONTEXT_LENGTH_TOKENS) * CHARS_PER_TOKEN -
1396
+ buildReflectPrompt({ ...promptInput, contentBudgetChars: 0 }).prompt.length) /
1397
+ 2))
1398
+ : undefined;
1399
+ }
1400
+ /**
1401
+ * #952 — gather every read-only prompt-input source {@link buildReflectPromptInput}
1402
+ * folds into a `ReflectPromptInput`: recent feedback, schema/lint hints, related
1403
+ * lessons, previously-rejected proposals, and stash standards context.
1404
+ *
1405
+ * Shared by the real dispatch path (`akmReflect`'s step 4, via
1406
+ * {@link runReflectRefineIterations}) and `renderReflectPromptPreview`'s
1407
+ * `--show-prompt` preview, so both gather from exactly one definition instead
1408
+ * of two copies that can drift out of agreement.
1409
+ */
1410
+ async function gatherReflectPromptSources(options, stash, parsedRef, assetContent, assetCtx) {
1411
+ const feedback = readRecentFeedback(options.ref ? (options.itemRef ?? durableImproveRef(options.ref)) : undefined, options.eventsCtx);
1412
+ const schemaHints = buildSchemaHints(parsedRef?.type ?? "", assetContent);
1413
+ const relatedLessons = options.ref && parsedRef ? await readRelatedLessons(assetCtx, stash, options.ref, parsedRef, options.itemRef) : [];
1414
+ // Reflexion-style verbal-RL: inject rejected proposals so the agent avoids
1415
+ // reproducing proposals that have already been reviewed and refused.
1416
+ const rejectedProposals = readRejectedProposals(stash, options.ref, options.ctx);
1417
+ // Standards "rulebook" for this target — stash convention/meta facts; empty
1418
+ // when none fire.
1419
+ const standardsContext = resolveStandardsContext(options.ref, stash);
1420
+ return { feedback, schemaHints, relatedLessons, rejectedProposals, standardsContext };
1421
+ }
1422
+ /**
1423
+ * #952 — assemble the `ReflectPromptInput` object literal reflect actually
1424
+ * sends, from gathered sources plus the per-call values (draft path, prior
1425
+ * draft). Shared by the real dispatch path ({@link runReflectRefineIterations})
1426
+ * and `renderReflectPromptPreview`'s `--show-prompt` preview — including
1427
+ * `avoidPatterns`, which the preview previously omitted even though a live
1428
+ * improve loop passes it (recent-error context, O-5 / #378).
1429
+ */
1430
+ function buildReflectPromptInput(args) {
1431
+ const { options, parsedRef, assetContent, sources, runnerSpec, draftFilePath, priorDraft } = args;
1432
+ const { feedback, schemaHints, relatedLessons, rejectedProposals, standardsContext } = sources;
1433
+ const outputMode = runnerIsLlm(runnerSpec)
1434
+ ? wantsJsonSchemaOutput(runnerSpec.connection)
1435
+ ? "json_schema"
1436
+ : "framed_markdown"
1437
+ : undefined;
1438
+ return {
1439
+ ...(options.ref ? { ref: options.ref } : {}),
1440
+ ...(parsedRef?.type ? { type: parsedRef.type } : {}),
1441
+ ...(parsedRef?.name ? { name: parsedRef.name } : {}),
1442
+ ...(assetContent !== undefined ? { assetContent } : {}),
1443
+ ...(feedback.length > 0 ? { feedback } : {}),
1444
+ ...(schemaHints.length > 0 ? { schemaHints } : {}),
1445
+ ...(relatedLessons.length > 0 ? { relatedLessons } : {}),
1446
+ ...(options.task ? { task: options.task } : {}),
1447
+ ...(standardsContext.trim() ? { standardsContext } : {}),
1448
+ ...(options.avoidPatterns && options.avoidPatterns.length > 0 ? { avoidPatterns: options.avoidPatterns } : {}),
1449
+ ...(rejectedProposals.length > 0 ? { rejectedProposals } : {}),
1450
+ // R-1: inject prior draft as self-critique target on iterations > 0
1451
+ ...(priorDraft !== undefined ? { priorDraft } : {}),
1452
+ // Issue A (#reflect-pipeline file-write contract): when the runner can
1453
+ // touch the filesystem, instruct the agent to write the proposal body
1454
+ // to a tmp file instead of inlining it in JSON. Avoids parse failures
1455
+ // on long bodies (e.g. knowledge/systems/KOKORO_USAGE_GUIDE 8.4KB).
1456
+ ...(draftFilePath ? { draftFilePath } : {}),
1457
+ ...(outputMode ? { outputMode } : {}),
1458
+ };
1459
+ }
1374
1460
  /**
1375
1461
  * Run the agent with the optional Self-Refine loop (R-1 / #372): up to
1376
1462
  * `maxRefineIters` invocations, each injecting the prior draft as self-critique
@@ -1379,17 +1465,12 @@ async function resolveReflectSource(options, stash, emitReflectFailed) {
1379
1465
  * result + last draft path. Extracted verbatim from `akmReflect`.
1380
1466
  */
1381
1467
  async function runReflectRefineIterations(args) {
1382
- const { options, parsedRef, assetContent, feedback, schemaHints, relatedLessons, rejectedProposals, standardsContext, runnerSpec, lease, agentEnv, draftPathsToCleanup, onNotices, } = args;
1468
+ const { options, parsedRef, assetContent, sources, runnerSpec, lease, agentEnv, draftPathsToCleanup, onNotices } = args;
1383
1469
  const maxRefineIters = Math.max(1, options.maxRefineIters ?? 1);
1384
1470
  // Determine whether this dispatch can honour the file-write contract.
1385
1471
  // Agent CLI + OpenCode SDK runners both have filesystem access; the direct
1386
1472
  // LLM HTTP runner does NOT.
1387
1473
  const canRunnerWriteFile = runnerSupportsFileWrite(runnerSpec);
1388
- const outputMode = runnerIsLlm(runnerSpec)
1389
- ? wantsJsonSchemaOutput(runnerSpec.connection)
1390
- ? "json_schema"
1391
- : "framed_markdown"
1392
- : undefined;
1393
1474
  // Initialized to a sentinel; always overwritten in the first loop iteration
1394
1475
  // (maxRefineIters is clamped to >= 1 above).
1395
1476
  let result = {};
@@ -1404,44 +1485,16 @@ async function runReflectRefineIterations(args) {
1404
1485
  draftPathsToCleanup.push(iterDraftPath);
1405
1486
  lastDraftPath = iterDraftPath;
1406
1487
  }
1407
- const promptInput = {
1408
- ...(options.ref ? { ref: options.ref } : {}),
1409
- ...(parsedRef?.type ? { type: parsedRef.type } : {}),
1410
- ...(parsedRef?.name ? { name: parsedRef.name } : {}),
1411
- ...(assetContent !== undefined ? { assetContent } : {}),
1412
- ...(feedback.length > 0 ? { feedback } : {}),
1413
- ...(schemaHints.length > 0 ? { schemaHints } : {}),
1414
- ...(relatedLessons.length > 0 ? { relatedLessons } : {}),
1415
- ...(options.task ? { task: options.task } : {}),
1416
- ...(standardsContext.trim() ? { standardsContext } : {}),
1417
- ...(options.avoidPatterns && options.avoidPatterns.length > 0 ? { avoidPatterns: options.avoidPatterns } : {}),
1418
- ...(rejectedProposals.length > 0 ? { rejectedProposals } : {}),
1419
- // R-1: inject prior draft as self-critique target on iterations > 0
1420
- ...(priorDraft !== undefined ? { priorDraft } : {}),
1421
- // Issue A (#reflect-pipeline file-write contract): when the runner can
1422
- // touch the filesystem, instruct the agent to write the proposal body
1423
- // to a tmp file instead of inlining it in JSON. Avoids parse failures
1424
- // on long bodies (e.g. knowledge/systems/KOKORO_USAGE_GUIDE 8.4KB).
1425
- ...(iterDraftPath ? { draftFilePath: iterDraftPath } : {}),
1426
- ...(outputMode ? { outputMode } : {}),
1427
- };
1428
- // #952 — the flat REFLECT_CONTENT_CAP (12 000 chars) exists only to avoid
1429
- // E2BIG when the prompt travels through CLI argv (agent/SDK runners). The
1430
- // direct-LLM HTTP path never touches argv, so it can use the resolved
1431
- // engine's own context window instead. The reserve for "the rest of the
1432
- // prompt" is measured directly (not guessed): build the same prompt with
1433
- // the content cap forced to zero and use its length as the overhead, so
1434
- // feedback/standards/schema-hints/prior-draft size is accounted for
1435
- // exactly, per this call. A reflect rewrite returns a body roughly the
1436
- // size of the input, so the budget only spends HALF of the usable window
1437
- // on input content and reserves the other half for the model's own
1438
- // output — otherwise a full-context request leaves no room for a
1439
- // response. Never drops below the flat floor.
1440
- const contentBudgetChars = runnerIsLlm(runnerSpec) && assetContent?.trim()
1441
- ? Math.max(REFLECT_CONTENT_CAP, Math.floor(((runnerSpec.connection.contextLength ?? DEFAULT_CONTEXT_LENGTH_TOKENS) * CHARS_PER_TOKEN -
1442
- buildReflectPrompt({ ...promptInput, contentBudgetChars: 0 }).prompt.length) /
1443
- 2))
1444
- : undefined;
1488
+ const promptInput = buildReflectPromptInput({
1489
+ options,
1490
+ parsedRef,
1491
+ assetContent,
1492
+ sources,
1493
+ runnerSpec,
1494
+ draftFilePath: iterDraftPath,
1495
+ priorDraft,
1496
+ });
1497
+ const contentBudgetChars = computeReflectContentBudgetChars(promptInput, runnerSpec);
1445
1498
  const { prompt } = buildReflectPrompt({
1446
1499
  ...promptInput,
1447
1500
  ...(contentBudgetChars !== undefined ? { contentBudgetChars } : {}),
@@ -1459,10 +1512,10 @@ async function runReflectRefineIterations(args) {
1459
1512
  ...(options.signal ? { signal: options.signal } : {}),
1460
1513
  priorDraft,
1461
1514
  iteration: iter,
1462
- ...(outputMode === "json_schema"
1515
+ ...(promptInput.outputMode === "json_schema"
1463
1516
  ? { responseSchema: options.ref ? REFLECT_JSON_SCHEMA : REFLECT_UNSCOPED_JSON_SCHEMA }
1464
1517
  : {}),
1465
- outputMode: outputMode ?? "framed_markdown",
1518
+ outputMode: promptInput.outputMode ?? "framed_markdown",
1466
1519
  ...(options.ref ? { targetRef: options.ref } : {}),
1467
1520
  allowRepair: repairAttempts === 0,
1468
1521
  ...(options.chat ? { chat: options.chat } : {}),
@@ -1650,6 +1703,62 @@ function validateReflectPayloadRef(args) {
1650
1703
  return undefined;
1651
1704
  }
1652
1705
  }
1706
+ /**
1707
+ * #952 — render the composed reflect prompt for exactly one asset with no
1708
+ * engine dispatch. Reuses every read-only step `akmReflect` performs before
1709
+ * {@link buildReflectPrompt} (source resolution, runner resolution, feedback /
1710
+ * schema-hint / related-lesson / rejected-proposal gathering) and stops right
1711
+ * there: no dispatch lease is acquired, no request is sent, and — because the
1712
+ * `emitReflectFailed` callback passed to {@link resolveReflectSource} here is
1713
+ * a no-op — no `reflect_invoked`/`reflect_completed` event is appended either.
1714
+ *
1715
+ * `akm improve <ref> --show-prompt` (`improve-cli.ts`) is the CLI surface: a
1716
+ * field operator uses it to see the exact prompt reflect would send, in
1717
+ * seconds, without running a full improve cycle or needing a reachable
1718
+ * engine.
1719
+ */
1720
+ export async function renderReflectPromptPreview(options) {
1721
+ if (!options.ref) {
1722
+ throw new UsageError("renderReflectPromptPreview requires options.ref.", "INVALID_FLAG_VALUE");
1723
+ }
1724
+ const ref = options.ref;
1725
+ const stash = resolveRunStashDir(options.stashDir);
1726
+ const sourceResolved = await resolveReflectSource(options, stash, () => {
1727
+ // No event emitted: this is a read-only preview, not a real invocation.
1728
+ });
1729
+ if ("failure" in sourceResolved) {
1730
+ const { failure } = sourceResolved;
1731
+ throw new UsageError((!failure.ok && failure.error) || `Reflect cannot preview ref "${ref}".`, "INVALID_FLAG_VALUE");
1732
+ }
1733
+ const { assetContent, parsedRef } = sourceResolved;
1734
+ const { runnerSpec, engineName } = resolveReflectRunner(options);
1735
+ const ctx = buildReflectRunContext({ options, stash, config: options.config ?? loadConfig(), runnerSpec });
1736
+ const assetCtx = ctx.withFreshAssetMemo();
1737
+ const sources = await gatherReflectPromptSources(options, stash, parsedRef, assetContent, assetCtx);
1738
+ const canRunnerWriteFile = runnerSupportsFileWrite(runnerSpec);
1739
+ // Same tmp-path synthesis a real dispatch would use (Issue A) — never
1740
+ // written to, since this preview never runs the agent.
1741
+ const draftFilePath = canRunnerWriteFile ? synthesizeReflectDraftPath(ref) : undefined;
1742
+ const previewPromptInput = buildReflectPromptInput({
1743
+ options,
1744
+ parsedRef,
1745
+ assetContent,
1746
+ sources,
1747
+ runnerSpec,
1748
+ draftFilePath,
1749
+ priorDraft: undefined,
1750
+ });
1751
+ // #952 — mirror the real dispatch path's context-aware content budget (see
1752
+ // computeReflectContentBudgetChars) so the preview shows the exact prompt
1753
+ // reflect would send: an LLM engine with a large context window gets the
1754
+ // full asset with no truncation marker, not the flat 12 000-char cap.
1755
+ const contentBudgetChars = computeReflectContentBudgetChars(previewPromptInput, runnerSpec);
1756
+ const { prompt } = buildReflectPrompt({
1757
+ ...previewPromptInput,
1758
+ ...(contentBudgetChars !== undefined ? { contentBudgetChars } : {}),
1759
+ });
1760
+ return { ref, prompt, engine: engineName, engineKind: runnerSpec.kind };
1761
+ }
1653
1762
  export async function akmReflect(options = {}) {
1654
1763
  const stash = resolveRunStashDir(options.stashDir);
1655
1764
  // Build lazy event emitters. The invocation row is committed only after the
@@ -1695,17 +1804,7 @@ export async function akmReflect(options = {}) {
1695
1804
  // 4. Build the shared prompt inputs — feedback, hints, lessons, rejected
1696
1805
  // proposals. These are stable across refinement iterations; only the
1697
1806
  // `priorDraft` field changes per-iteration (R-1 / #372).
1698
- const feedback = readRecentFeedback(options.ref ? (options.itemRef ?? durableImproveRef(options.ref)) : undefined, options.eventsCtx);
1699
- const schemaHints = buildSchemaHints(parsedRef?.type ?? "", assetContent);
1700
- const relatedLessons = options.ref && parsedRef
1701
- ? await readRelatedLessons(assetCtx, stash, options.ref, parsedRef, options.itemRef)
1702
- : [];
1703
- // Reflexion-style verbal-RL: inject rejected proposals so the agent avoids
1704
- // reproducing proposals that have already been reviewed and refused.
1705
- const rejectedProposals = readRejectedProposals(stash, options.ref, options.ctx);
1706
- // Standards "rulebook" for this target — stash convention/meta facts; empty
1707
- // when none fire.
1708
- const standardsContext = resolveStandardsContext(options.ref, stash);
1807
+ const sources = await gatherReflectPromptSources(options, stash, parsedRef, assetContent, assetCtx);
1709
1808
  // 5. Spawn the agent — with the optional Self-Refine loop (R-1 / #372),
1710
1809
  // extracted to {@link runReflectRefineIterations}.
1711
1810
  const agentEnv = options.eventSource === "improve" ? { AKM_EVENT_SOURCE: "improve" } : {};
@@ -1725,11 +1824,7 @@ export async function akmReflect(options = {}) {
1725
1824
  options,
1726
1825
  parsedRef,
1727
1826
  assetContent,
1728
- feedback,
1729
- schemaHints,
1730
- relatedLessons,
1731
- rejectedProposals,
1732
- standardsContext,
1827
+ sources,
1733
1828
  runnerSpec,
1734
1829
  lease: generationLease,
1735
1830
  agentEnv,
@@ -1808,7 +1903,7 @@ export async function akmReflect(options = {}) {
1808
1903
  qualityGateSkippedNoJudge,
1809
1904
  qualityJudgeRunner,
1810
1905
  qualityJudgeLease,
1811
- feedback,
1906
+ feedback: sources.feedback,
1812
1907
  stash,
1813
1908
  emitReflectFailed,
1814
1909
  onNotices: collectExecutionNotices,
@@ -4,7 +4,7 @@
4
4
  import { isDeepStrictEqual } from "node:util";
5
5
  import { defineGroupCommand, defineJsonCommand, output } from "../cli/shared.js";
6
6
  import { loadConfig } from "../core/config/config.js";
7
- import { copyDefaultModelMap, loadModelMapLayers, mergedModelMapProfiles, mergeModelMapLayers, } from "../integrations/agent/model-map.js";
7
+ import { copyDefaultModelMap, loadModelMapLayers, mergedModelMapProfiles, mergeModelMapLayers, WILDCARD_ENGINE_KEY, } from "../integrations/agent/model-map.js";
8
8
  /**
9
9
  * Compute the effective alias table (#946): every (alias, column) pair from
10
10
  * the fully resolved map, labeled with where its value came from.
@@ -25,6 +25,11 @@ function modelsListRows() {
25
25
  const rows = [];
26
26
  for (const [alias, columns] of Object.entries(resolved.aliases)) {
27
27
  for (const [column, profile] of Object.entries(columns)) {
28
+ // The wildcard default (#946) is an alias-wide fallback, not a real
29
+ // platform column an operator would dispatch to; report it folded into
30
+ // the real per-platform rows above instead of as its own row.
31
+ if (column === WILDCARD_ENGINE_KEY)
32
+ continue;
28
33
  const raw = rawProfiles[alias]?.[column];
29
34
  const via = raw?.engine !== undefined ? "engine" : "literal";
30
35
  const defaultProfile = defaultsOnly.aliases[alias]?.[column];
@@ -25,6 +25,8 @@ import { copySearchHitAttribution, getSearchHitAttribution, usageEventAttributio
25
25
  import { findSourceForPath, resolveSourceEntries } from "../../indexer/search/search-source.js";
26
26
  import { insertUsageEvent } from "../../indexer/usage/usage-events.js";
27
27
  import { estimateTokenCount } from "../../llm/embedders/remote.js";
28
+ import { tryLlmFeature } from "../../llm/feature-gate.js";
29
+ import { rerankDocuments } from "../../llm/rerank-client.js";
28
30
  import { truncateDescription } from "../../output/shapes/helpers.js";
29
31
  import { TELEMETRY_BUSY_TIMEOUT_MS, withIndexDb } from "../../storage/repositories/index-db.js";
30
32
  import { findEntryIdByRef, getItemRefById } from "../../storage/repositories/index-entries-repository.js";
@@ -145,7 +147,7 @@ export async function curateSearchResults(query, result, limit, selectedType, ev
145
147
  // fixtures) with a `SearchResponse` that was never type-filtered.
146
148
  const stashHits = selectedType && selectedType !== "any" ? allStashHits.filter((hit) => hit.type === selectedType) : allStashHits;
147
149
  const selected = selectCuratedStashHits(query, stashHits, limit);
148
- const selectedStashHits = selected.selected;
150
+ const selectedStashHits = await maybeRerankCuratedStashHits(query, selected.selected);
149
151
  const supportRefsByRef = selected.supportRefsByRef;
150
152
  // F4/R-019: respect `--limit` for registry fill instead of hard-capping it
151
153
  // at a bare literal 2 — the remaining slots after stash hits ARE the cap.
@@ -594,6 +596,37 @@ function appendCurateSupportRef(supportRefsByRef, ownerRef, supportRef) {
594
596
  return;
595
597
  supportRefsByRef.set(ownerRef, [...existing, supportRef]);
596
598
  }
599
+ /** Default number of `selectCuratedStashHits` candidates sent to the reranker when `search.curateRerank.topN` isn't set. */
600
+ const DEFAULT_CURATE_RERANK_TOP_N = 8;
601
+ /**
602
+ * Optional cross-encoder rerank pass over curate's already-selected, already-
603
+ * ranked candidates (#951). Disabled by default (`search.curateRerank.enabled`
604
+ * is falsy) and, when enabled, best-effort: any failure (misconfigured
605
+ * endpoint, network error, timeout, malformed response) falls back to
606
+ * `selectCuratedStashHits`'s own ranking unchanged — a reranker outage must
607
+ * never turn into a curate failure.
608
+ *
609
+ * Only the top `topN` (default {@link DEFAULT_CURATE_RERANK_TOP_N}) already-
610
+ * selected hits are sent (bounded request size); anything past that keeps its
611
+ * original position appended after the reranked prefix.
612
+ */
613
+ async function maybeRerankCuratedStashHits(query, hits) {
614
+ if (hits.length <= 1)
615
+ return hits;
616
+ const config = loadConfig();
617
+ const rerankConfig = config.search?.curateRerank;
618
+ return tryLlmFeature("curate_rerank", config, async () => {
619
+ const topN = rerankConfig?.topN ?? DEFAULT_CURATE_RERANK_TOP_N;
620
+ const head = hits.slice(0, topN);
621
+ const tail = hits.slice(topN);
622
+ const documents = head.map((hit) => [hit.name, hit.description].filter(Boolean).join(" — "));
623
+ const ranked = await rerankDocuments(rerankConfig ?? {}, query, documents);
624
+ const rerankedHead = ranked
625
+ .map(({ index }) => head[index])
626
+ .filter((hit) => hit !== undefined);
627
+ return [...rerankedHead, ...tail];
628
+ }, hits, { timeoutMs: rerankConfig?.timeoutMs ?? null });
629
+ }
597
630
  function selectCuratedStashHits(query, hits, limit) {
598
631
  const intent = parseCurateIntent(query);
599
632
  const collapsed = collapseCurateFamilies(query, hits);
@@ -6,7 +6,7 @@
6
6
  * former `config-schema.ts` monolith — no behavior change.
7
7
  */
8
8
  import { z } from "zod";
9
- import { nonEmptyString, nonNegativeNumber, positiveInt } from "./primitives.js";
9
+ import { httpUrl, nonEmptyString, nonNegativeNumber, positiveInt, symbolicOrWarnApiKey } from "./primitives.js";
10
10
  // ── Search ──────────────────────────────────────────────────────────────────
11
11
  const SearchGraphBoostSchema = z
12
12
  .object({
@@ -22,10 +22,41 @@ const SearchGraphBoostSchema = z
22
22
  confidenceWeight: z.number().finite().min(0).max(1).default(0.2).optional(),
23
23
  })
24
24
  .passthrough();
25
+ /**
26
+ * `search.curateRerank` (#951) — an optional cross-encoder rerank pass over
27
+ * `akm curate`'s already-selected candidates.
28
+ *
29
+ * Deliberately its own small config arm rather than a third member of the
30
+ * `engines` map (`EngineConfigSchema` in ./engines.ts): that union's "llm" /
31
+ * "agent" kinds are load-bearing all the way through execution-lowering,
32
+ * runner dispatch, and the harness model map (100+ call sites narrow on
33
+ * `engine.kind`). A reranker is neither — it never dispatches an agent or
34
+ * lowers to a chat-completions call — so folding it into that union would
35
+ * force every one of those call sites to account for a kind they can't do
36
+ * anything with. `endpoint` + `model` (+ optional `apiKey`) is the same
37
+ * connection shape as an LLM engine without inheriting that machinery.
38
+ *
39
+ * `curate_rerank` was removed as a dead `llm.features.*` key in 0.8.0 (no
40
+ * implementation ever sent a request); this is a new, real implementation,
41
+ * disabled by default.
42
+ */
43
+ export const CurateRerankConfigSchema = z
44
+ .object({
45
+ enabled: z.boolean().optional(),
46
+ /** Full URL of the reranker's rerank endpoint, e.g. `http://host:port/rerank`. */
47
+ endpoint: httpUrl.optional(),
48
+ model: nonEmptyString.optional(),
49
+ apiKey: symbolicOrWarnApiKey("search.curateRerank.apiKey").optional(),
50
+ timeoutMs: positiveInt.optional(),
51
+ /** How many of curate's already-ranked candidates to send to the reranker. Default 8. */
52
+ topN: positiveInt.max(50).optional(),
53
+ })
54
+ .passthrough();
25
55
  export const SearchConfigSchema = z
26
56
  .object({
27
57
  minScore: nonNegativeNumber.optional(),
28
58
  defaultExcludeTypes: z.array(nonEmptyString).optional(),
29
59
  graphBoost: SearchGraphBoostSchema.optional(),
60
+ curateRerank: CurateRerankConfigSchema.optional(),
30
61
  })
31
62
  .passthrough();
@@ -83,6 +83,8 @@ const TRANSIENT_HINTS = {
83
83
  RUN_LEASE_HELD: "Wait for the named engine invocation to finish or for the lease to expire, then retry. `akm workflow status <id>` shows the current lease.",
84
84
  STATE_DB_CONTENDED: "Another akm process is writing state.db right now. Wait a few seconds and retry; commands that support --skip-if-locked can skip instead of failing.",
85
85
  INDEX_DB_CONTENDED: "Another akm process is writing index.db; retry shortly, or pass --skip-if-locked on scheduled runs.",
86
+ MAINTENANCE_BARRIER_BUSY: "Another akm process is registering a lock or lease right now. Retry shortly, or pass --skip-if-locked on scheduled index/improve/workflow runs.",
87
+ IMPROVE_LOCK_HELD: "Another akm improve run holds the whole-run lock right now. Wait for it to finish and retry, or pass --skip-if-locked on scheduled runs.",
86
88
  };
87
89
  /** Default hint for each NotFoundError code. */
88
90
  const NOT_FOUND_HINTS = {
@@ -6,7 +6,8 @@ import { randomUUID } from "node:crypto";
6
6
  import fs from "node:fs";
7
7
  import path from "node:path";
8
8
  import { sleepSync } from "../runtime.js";
9
- import { ConfigError } from "./errors.js";
9
+ import { backoffDelay } from "./common.js";
10
+ import { ConfigError, TransientError } from "./errors.js";
10
11
  import { createLockPayload, probeLock, reclaimStaleLock, releaseLock, tryAcquireLockSync } from "./file-lock.js";
11
12
  import { getMaintenanceBarrierPath } from "./paths.js";
12
13
  const heldBarrierContext = new AsyncLocalStorage();
@@ -24,6 +25,26 @@ const heldBarrierContext = new AsyncLocalStorage();
24
25
  * (`commands/improve/extract.ts`).
25
26
  */
26
27
  const MAINTENANCE_BARRIER_STALE_AFTER_MS = 5 * 60 * 1000;
28
+ /**
29
+ * The barrier normally holds for one lock-file write — sub-millisecond on
30
+ * any real filesystem. Two akm processes racing to register a lock in the
31
+ * very same instant (e.g. two `akm index` runs a scheduler launched back to
32
+ * back) can still collide on it; retrying briefly resolves that ordinary
33
+ * case instead of failing a legitimate concurrent invocation outright
34
+ * (field follow-up to #956, G1). Bounded short so a genuinely wedged holder
35
+ * still surfaces the busy error promptly rather than making a losing
36
+ * process hang — comfortably above the barrier's normal hold time, well
37
+ * below a length that would make this feel like the blocking lock #872
38
+ * removed. Never applies to the rebuild lock itself, which stays
39
+ * non-blocking (#872).
40
+ */
41
+ const MAINTENANCE_BARRIER_BUSY_RETRY_BOUND_MS = 1_500;
42
+ let busyRetryBoundMsForTests;
43
+ /** Test-only override for {@link MAINTENANCE_BARRIER_BUSY_RETRY_BOUND_MS}, so a unit test can exercise the
44
+ * exhausted-retry throw without a real ~1.5s wait. Restored via tests/_helpers/seams.ts's resetAllSeams(). */
45
+ export function _setMaintenanceBarrierBusyRetryBoundMsForTests(ms) {
46
+ busyRetryBoundMsForTests = ms;
47
+ }
27
48
  /**
28
49
  * Serialize the short critical section that creates each long-lived AKM lock,
29
50
  * lease, or state activity. The operation keeps its own ownership record; this
@@ -44,11 +65,19 @@ export function tryAcquireMaintenanceBarrier() {
44
65
  return undefined;
45
66
  }
46
67
  export function acquireMaintenanceBarrier() {
47
- const release = tryAcquireMaintenanceBarrier();
48
- if (release)
49
- return release;
50
- throw new ConfigError(`AKM maintenance is in progress (barrier ${getMaintenanceBarrierPath()}); retry after it completes. ` +
51
- `A sentinel older than ${MAINTENANCE_BARRIER_STALE_AFTER_MS / 60_000} minute(s) is reclaimed automatically on the next attempt.`, "INVALID_CONFIG_FILE");
68
+ const boundMs = busyRetryBoundMsForTests ?? MAINTENANCE_BARRIER_BUSY_RETRY_BOUND_MS;
69
+ const deadline = Date.now() + boundMs;
70
+ for (let attempt = 0;; attempt += 1) {
71
+ const release = tryAcquireMaintenanceBarrier();
72
+ if (release)
73
+ return release;
74
+ const remainingMs = deadline - Date.now();
75
+ if (remainingMs <= 0)
76
+ break;
77
+ sleepSync(Math.min(backoffDelay(attempt), remainingMs));
78
+ }
79
+ throw new TransientError(`AKM maintenance is in progress (barrier ${getMaintenanceBarrierPath()}); retry shortly. ` +
80
+ `A sentinel older than ${MAINTENANCE_BARRIER_STALE_AFTER_MS / 60_000} minute(s) is reclaimed automatically on the next attempt.`, "MAINTENANCE_BARRIER_BUSY");
52
81
  }
53
82
  export function withMaintenanceStartBarrier(run) {
54
83
  if (heldBarrierContext.getStore()?.active)
@@ -0,0 +1,56 @@
1
+ // This Source Code Form is subject to the terms of the Mozilla Public
2
+ // License, v. 2.0. If a copy of the MPL was not distributed with this
3
+ // file, You can obtain one at https://mozilla.org/MPL/2.0/.
4
+ /**
5
+ * Shared index.db contention reclassification (field follow-up to #956).
6
+ *
7
+ * Extracted out of `indexer.ts` so both `akmIndex`'s outer catch AND
8
+ * `generateEmbeddingsForDb`'s own catch (`materialize-embeddings.ts`) can
9
+ * reuse the ONE classifier instead of each building a raw
10
+ * `Semantic search verification failed: <driver message>` string. Living in
11
+ * its own module (rather than one importing the other) avoids the import
12
+ * cycle `indexer.ts` <-> `materialize-embeddings.ts` would otherwise form.
13
+ */
14
+ import { AkmError, TransientError } from "../core/errors.js";
15
+ import { probeLock } from "../core/file-lock.js";
16
+ import { formatLockHolderPid } from "../core/run-lock.js";
17
+ import { isSqliteContentionError } from "../core/state-db.js";
18
+ import { indexRebuildLockPath } from "./index-rebuild-lock.js";
19
+ /**
20
+ * Read-only description of the rebuild lock's current holder, appended to a
21
+ * reclassified index.db contention message when known (field follow-up to
22
+ * #956). `probeLock` only inspects the sentinel — it never acquires or
23
+ * mutates it — so this is safe to call from inside an error path.
24
+ */
25
+ function describeIndexRebuildLockHolder() {
26
+ const probe = probeLock(indexRebuildLockPath());
27
+ if (probe.state !== "held")
28
+ return "";
29
+ return ` The rebuild lock is currently held by pid ${formatLockHolderPid({
30
+ pid: probe.holderPid,
31
+ launcherPid: probe.launcherPid ?? null,
32
+ })}.`;
33
+ }
34
+ /**
35
+ * Reclassify a contention-shaped error escaping the walk, index, or
36
+ * embedding phase into a retryable-shortly `TransientError` (field
37
+ * follow-up to #956, dev-team field review 2026-09-10): a concurrent writer
38
+ * (another `akm index`, a source-update embedding pass, the per-command
39
+ * background reindex) can make index.db busy, and the raw SQLite driver
40
+ * error ("database is locked") used to escape as exit 70
41
+ * (internal/unclassified) instead of the "retry shortly" contract exit 75
42
+ * gives a scheduler to branch on — mirroring `STATE_DB_CONTENDED`'s
43
+ * precedent for state.db (`core/state-db.ts`). Reuses the ONE shared
44
+ * classifier, `isSqliteContentionError`, rather than a second one. An error
45
+ * that is already a classified akm error (e.g. a `STATE_DB_CONTENDED`
46
+ * TransientError from an inner state.db write) is never re-wrapped — only a
47
+ * raw, unclassified error matching the shared contention shape is
48
+ * reclassified. Every other error is rethrown unchanged.
49
+ */
50
+ export function reclassifyIndexDbContention(error) {
51
+ if (error instanceof AkmError || !isSqliteContentionError(error))
52
+ return error;
53
+ const contended = new TransientError(`akm's index database is busy (another akm process is writing it); retry shortly.${describeIndexRebuildLockHolder()}`, "INDEX_DB_CONTENDED");
54
+ contended.cause = error;
55
+ return contended;
56
+ }
@@ -7,14 +7,12 @@ import { detectAdapterId } from "../core/adapter/detect-adapter.js";
7
7
  import { adapterForId } from "../core/adapter/registry.js";
8
8
  import { isHttpUrl, toErrorMessage } from "../core/common.js";
9
9
  import { concurrentMap } from "../core/concurrent.js";
10
- import { AkmError, ConfigError, TransientError } from "../core/errors.js";
11
- import { probeLock } from "../core/file-lock.js";
10
+ import { ConfigError } from "../core/errors.js";
12
11
  import { defaultConcurrencyForEndpoint } from "../core/loopback.js";
13
12
  import { classifyPathAccess, describeInaccessiblePath } from "../core/path-access.js";
14
13
  import { getDbPath } from "../core/paths.js";
15
14
  import { SCRIPT_EXTENSIONS } from "../core/recognition-util.js";
16
- import { formatLockHolderPid } from "../core/run-lock.js";
17
- import { isSqliteContentionError, withStateDb } from "../core/state-db.js";
15
+ import { withStateDb } from "../core/state-db.js";
18
16
  import { isVerbose, warn, warnOnce, warnVerbose } from "../core/warn.js";
19
17
  import { disposeLoweredExecutionDispatchLease, } from "../integrations/agent/execution-lowering.js";
20
18
  import { isLlmFeatureEnabled } from "../llm/feature-gate.js";
@@ -30,7 +28,7 @@ import { upsertUtilityScore } from "../storage/repositories/index-utility-reposi
30
28
  import { getEmbeddingCount, isVecAvailable, isVecFastPathReady, warnIfVecMissing, } from "../storage/repositories/index-vec-repository.js";
31
29
  import { assertIndexedWorkflowSourceIdentity, WorkflowSourceIdentityError } from "../workflows/source-files.js";
32
30
  import { deleteStoredGraph } from "./db/graph-db.js";
33
- import { indexRebuildLockPath } from "./index-rebuild-lock.js";
31
+ import { reclassifyIndexDbContention } from "./index-db-contention.js";
34
32
  import { deriveEntryProvenance, deriveInstallations } from "./installations.js";
35
33
  import { indexedPathMatchesOwner, resolveAdapterConceptOwner, } from "./lookup/adapter-concept-owner.js";
36
34
  import { generateEmbeddingsForDb } from "./materialize-embeddings.js";
@@ -374,44 +372,13 @@ let akmIndexOverride;
374
372
  export function _setAkmIndexForTests(fake) {
375
373
  akmIndexOverride = fake;
376
374
  }
377
- /**
378
- * Read-only description of the rebuild lock's current holder, appended to a
379
- * reclassified index.db contention message when known (field follow-up to
380
- * #956). `probeLock` only inspects the sentinel — it never acquires or
381
- * mutates it — so this is safe to call from inside an error path.
382
- */
383
- function describeIndexRebuildLockHolder() {
384
- const probe = probeLock(indexRebuildLockPath());
385
- if (probe.state !== "held")
386
- return "";
387
- return ` The rebuild lock is currently held by pid ${formatLockHolderPid({
388
- pid: probe.holderPid,
389
- launcherPid: probe.launcherPid ?? null,
390
- })}.`;
391
- }
392
- /**
393
- * Reclassify a contention-shaped error escaping the walk, index, or
394
- * embedding phase into a retryable-shortly `TransientError` (field
395
- * follow-up to #956, dev-team field review 2026-09-10): a concurrent writer
396
- * (another `akm index`, a source-update embedding pass, the per-command
397
- * background reindex) can make index.db busy, and the raw SQLite driver
398
- * error ("database is locked") used to escape as exit 70
399
- * (internal/unclassified) instead of the "retry shortly" contract exit 75
400
- * gives a scheduler to branch on — mirroring `STATE_DB_CONTENDED`'s
401
- * precedent for state.db (`core/state-db.ts`). Reuses the ONE shared
402
- * classifier, `isSqliteContentionError`, rather than a second one. An error
403
- * that is already a classified akm error (e.g. a `STATE_DB_CONTENDED`
404
- * TransientError from an inner state.db write) is never re-wrapped — only a
405
- * raw, unclassified error matching the shared contention shape is
406
- * reclassified. Every other error is rethrown unchanged.
407
- */
408
- export function reclassifyIndexDbContention(error) {
409
- if (error instanceof AkmError || !isSqliteContentionError(error))
410
- return error;
411
- const contended = new TransientError(`akm's index database is busy (another akm process is writing it); retry shortly.${describeIndexRebuildLockHolder()}`, "INDEX_DB_CONTENDED");
412
- contended.cause = error;
413
- return contended;
414
- }
375
+ // Moved to its own module (field follow-up to #956) so
376
+ // `generateEmbeddingsForDb` (materialize-embeddings.ts) can reuse the same
377
+ // classifier without an indexer.ts <-> materialize-embeddings.ts import
378
+ // cycle. Re-exported here for back-compat with existing call sites/tests
379
+ // that import it from `./indexer`. See index-db-contention.ts for the full
380
+ // rationale.
381
+ export { reclassifyIndexDbContention };
415
382
  export async function akmIndex(options) {
416
383
  try {
417
384
  const override = akmIndexOverride;