akm-cli 0.9.16-alpha.1 → 0.9.16
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +56 -132
- package/dist/assets/hints/cli-hints-full.md +13 -6
- package/dist/assets/tasks/core/index-refresh.yml +1 -1
- package/dist/assets/tasks/improve/akm-improve-catchup.yml +3 -6
- package/dist/cli/retired-commands.js +0 -4
- package/dist/cli/unknown-flags.js +3 -36
- package/dist/commands/env/env-binding.js +4 -4
- package/dist/commands/env/env-cli.js +3 -3
- package/dist/commands/improve/collapse-detector.js +2 -2
- package/dist/commands/improve/consolidate.js +4 -6
- package/dist/commands/improve/improve-cli.js +20 -15
- package/dist/commands/improve/reflect.js +23 -2
- package/dist/commands/lint/base-linter.js +9 -0
- package/dist/commands/lint/env-key-rules.js +2 -2
- package/dist/commands/proposal/propose.js +15 -1
- package/dist/commands/proposal/repository.js +3 -12
- package/dist/commands/proposal/validators/proposal-quality-validators.js +40 -3
- package/dist/commands/proposal/validators/proposal-validators.js +5 -4
- package/dist/commands/read/curate.js +44 -34
- package/dist/commands/read/search.js +35 -54
- package/dist/commands/read/show.js +21 -2
- package/dist/commands/registry-cli.js +5 -5
- package/dist/commands/sources/add-cli.js +59 -16
- package/dist/commands/sources/bundle-cli.js +35 -11
- package/dist/commands/sources/bundle-config-ops.js +30 -0
- package/dist/commands/sources/dangerous-env-audit.js +4 -4
- package/dist/commands/sources/info.js +8 -8
- package/dist/commands/sources/installed-stashes.js +55 -61
- package/dist/commands/sources/source-add.js +39 -38
- package/dist/commands/sources/source-manage.js +34 -12
- package/dist/commands/sources/stash-cli.js +111 -119
- package/dist/commands/sources/stash-skeleton.js +6 -3
- package/dist/commands/tasks/explain.js +4 -1
- package/dist/commands/tasks/tasks-cli.js +31 -9
- package/dist/commands/tasks/tasks.js +239 -194
- package/dist/commands/tasks/validate.js +20 -32
- package/dist/core/activation-policy.js +4 -4
- package/dist/core/adapter/adapters/akm-adapter.js +8 -35
- package/dist/core/adapter/adapters/akm-metadata.js +1 -11
- package/dist/core/adapter/execution-source.js +10 -29
- package/dist/core/asset/asset-placement.js +0 -35
- package/dist/core/config/config-schema.js +64 -8
- package/dist/core/config/config-sources.js +96 -2
- package/dist/core/config/config.js +190 -24
- package/dist/core/config/legacy-source-shape-shim.js +9 -0
- package/dist/core/config/schema/embedding.js +30 -7
- package/dist/core/config/schema/execution.js +23 -0
- package/dist/core/config/schema/experimental.js +1 -1
- package/dist/core/config/schema/scheduler.js +20 -0
- package/dist/core/config/schema/search.js +10 -12
- package/dist/core/config/schema/sources-bundles.js +32 -1
- package/dist/core/content-safety.js +52 -0
- package/dist/core/errors.js +2 -5
- package/dist/core/maintenance-barrier.js +11 -13
- package/dist/core/paths.js +11 -0
- package/dist/core/run-lock.js +2 -5
- package/dist/core/state/migrations.js +1 -26
- package/dist/core/state-db.js +27 -63
- package/dist/core/type-presentation.js +1 -1
- package/dist/core/write-source.js +13 -8
- package/dist/indexer/bundle-identity-guard.js +45 -8
- package/dist/indexer/ensure-index.js +0 -5
- package/dist/indexer/index-db-contention.js +56 -0
- package/dist/indexer/index-rebuild-lock.js +73 -0
- package/dist/indexer/index-written-assets.js +171 -133
- package/dist/indexer/indexer.js +1621 -458
- package/dist/indexer/lookup/adapter-concept-owner.js +5 -19
- package/dist/indexer/materialize-embeddings.js +785 -0
- package/dist/indexer/passes/dir-staleness.js +161 -0
- package/dist/indexer/passes/metadata.js +1 -18
- package/dist/indexer/scan/drain-dir.js +70 -27
- package/dist/indexer/search/db-search.js +89 -373
- package/dist/indexer/search/ranking-contributors.js +16 -21
- package/dist/indexer/search/ranking.js +57 -135
- package/dist/indexer/search/search-source.js +29 -11
- package/dist/integrations/agent/execution-lowering.js +3 -2
- package/dist/integrations/agent/execution-preparation.js +32 -1
- package/dist/integrations/agent/prompts.js +1 -1
- package/dist/integrations/agent/request-lowering.js +3 -2
- package/dist/llm/client.js +3 -11
- package/dist/llm/embedder.js +3 -10
- package/dist/llm/embedders/remote.js +104 -133
- package/dist/llm/feature-gate.js +2 -4
- package/dist/llm/rerank-client.js +3 -3
- package/dist/output/html-render.js +2 -1
- package/dist/output/shapes/passthrough.js +2 -1
- package/dist/output/stdout.js +24 -0
- package/dist/output/text/command-format.js +13 -19
- package/dist/output/text/helpers.js +1 -1
- package/dist/output/text/index.js +2 -5
- package/dist/output/text.js +4 -3
- package/dist/registry/resolve.js +37 -10
- package/dist/scripts/akm-migrate-node.js +15197 -11351
- package/dist/scripts/akm-migrate.js +15514 -11668
- package/dist/setup/semantic-assets.js +2 -2
- package/dist/setup/setup.js +3 -3
- package/dist/setup/steps/connection.js +2 -3
- package/dist/setup/steps/tasks.js +29 -36
- package/dist/sources/providers/git-install.js +17 -11
- package/dist/sources/providers/git-provider.js +12 -5
- package/dist/sources/providers/git-stash.js +38 -16
- package/dist/sources/snapshot-fetchers/website-ingest.js +3 -3
- package/dist/storage/repositories/embedding-salvage-repository.js +184 -0
- package/dist/storage/repositories/index-connection.js +3 -1
- package/dist/storage/repositories/index-entries-repository.js +68 -77
- package/dist/storage/repositories/index-entry-schema.js +25 -16
- package/dist/storage/repositories/index-fts-repository.js +263 -29
- package/dist/storage/repositories/index-meta-repository.js +29 -0
- package/dist/storage/repositories/index-schema.js +122 -115
- package/dist/storage/repositories/index-utility-repository.js +1 -1
- package/dist/storage/repositories/index-vec-repository.js +435 -22
- package/dist/tasks/activation-config.js +90 -0
- package/dist/tasks/backends/cron.js +9 -0
- package/dist/tasks/backends/launchd.js +1 -0
- package/dist/tasks/backends/schtasks.js +2 -0
- package/dist/tasks/embedded.js +4 -5
- package/dist/tasks/scheduler-binding.js +2 -2
- package/dist/tasks/scheduler-sync-preview.js +8 -1
- package/dist/tasks/scheduler-sync.js +19 -10
- package/dist/tasks/source/parse-task-source.js +10 -113
- package/dist/tasks/source/project-v4.js +2 -2
- package/dist/tasks/source/task-source-v4.js +4 -12
- package/dist/tasks/source/task-to-v3.js +4 -12
- package/dist/tasks/source/task-to-v4.js +40 -7
- package/docs/migration/README.md +1 -0
- package/docs/migration/release-notes/0.9.15.md +36 -34
- package/docs/migration/release-notes/0.9.16.md +60 -98
- package/docs/migration/release-notes/README.md +0 -5
- package/docs/migration/v0.9.1-to-v0.9.2.md +6 -9
- package/docs/reference/cli.md +124 -122
- package/docs/reference/configuration.md +137 -133
- package/docs/reference/data-and-telemetry.md +1 -2
- package/docs/reference/tasks.md +34 -29
- package/package.json +1 -1
- package/schemas/akm-config.json +170 -6
- package/schemas/akm-task.json +1 -2
- package/dist/commands/sources/index-status.js +0 -99
- package/dist/core/hash.js +0 -18
- package/dist/indexer/drain.js +0 -306
- package/dist/indexer/embedding-identity.js +0 -20
- package/dist/indexer/enrich.js +0 -260
- package/dist/indexer/reconcile.js +0 -890
- package/dist/indexer/scan/parse-file.js +0 -66
- package/dist/indexer/units/unit.js +0 -159
- package/dist/llm/embedders/provider-limits.js +0 -288
- package/dist/storage/repositories/files-repository.js +0 -181
- package/dist/storage/repositories/units-repository.js +0 -510
|
@@ -24,18 +24,17 @@ import { systemErrorCode } from "../../core/system-error.js";
|
|
|
24
24
|
import { allowsFragmentRef, defaultRendererRegistry } from "../../core/type-presentation.js";
|
|
25
25
|
import { normalizeEmbeddingEndpoint } from "../../llm/embedders/remote.js";
|
|
26
26
|
import { assertIndexPathReadable, closeDatabase, openExistingDatabase, } from "../../storage/repositories/index-connection.js";
|
|
27
|
-
import { getAllEntries, getBaseBeliefStatesForDerivedTwins, getEntryCount, getPositiveFeedbackCountsByIds, } from "../../storage/repositories/index-entries-repository.js";
|
|
28
|
-
import { getIndexedMarkdownFragment, getIndexedMarkdownFragments, } from "../../storage/repositories/index-fts-repository.js";
|
|
27
|
+
import { getAllEntries, getBaseBeliefStatesForDerivedTwins, getEntryById, getEntryCount, getPositiveFeedbackCountsByIds, } from "../../storage/repositories/index-entries-repository.js";
|
|
28
|
+
import { getIndexedMarkdownFragment, getIndexedMarkdownFragments, searchFts, } from "../../storage/repositories/index-fts-repository.js";
|
|
29
29
|
import { getMeta } from "../../storage/repositories/index-meta-repository.js";
|
|
30
|
-
import {
|
|
30
|
+
import { getEmbeddingCount, searchVec } from "../../storage/repositories/index-vec-repository.js";
|
|
31
31
|
import { getCurrentWorkflowScopeKey } from "../../workflows/authoring/scope-key.js";
|
|
32
|
-
import { deriveObservedEmbeddingIdentity } from "../embedding-identity.js";
|
|
33
32
|
import { ensureIndex } from "../ensure-index.js";
|
|
34
33
|
import { collectGraphRelatedHit, loadGraphBoostContext } from "../graph/graph-boost.js";
|
|
35
34
|
import { isProposedQuality } from "../passes/metadata.js";
|
|
36
35
|
import { resolveProjectContext } from "../walk/project-context.js";
|
|
37
36
|
import { buildLexicalQueryPlan, parseRefPrefixQuery, parseRetiredTypePrefixQuery, } from "./fts-query.js";
|
|
38
|
-
import { applyRankingRules,
|
|
37
|
+
import { applyRankingRules, combineSearchScores, lexicalNameMatchTier, normalizeFtsScores } from "./ranking.js";
|
|
39
38
|
import { typeBoostFor } from "./ranking-contributors.js";
|
|
40
39
|
import { attachSearchHitAttribution, copySearchHitAttribution, getSearchHitAttribution } from "./search-attribution.js";
|
|
41
40
|
import { enrichSearchHit } from "./search-hit-enrichers.js";
|
|
@@ -226,39 +225,6 @@ export function canonicalContentTieKey(entry) {
|
|
|
226
225
|
const source = (body || entry.description || "").replace(/^ +| +$/g, "");
|
|
227
226
|
return Buffer.from(asciiCaseFold(source), "utf8").toString("hex");
|
|
228
227
|
}
|
|
229
|
-
/**
|
|
230
|
-
* Priority rank for `RankedEntryInput.lexicalMatch` — lower is stronger
|
|
231
|
-
* evidence. `undefined` (a pure-semantic hit with no lexical component at
|
|
232
|
-
* all) ranks weakest, below even a relaxed OR-pool recovery.
|
|
233
|
-
*
|
|
234
|
-
* Named-mechanism fix (fix-ranking-derived-outranks-primary): the exact →
|
|
235
|
-
* prefix → relaxed tier ladder (`searchUnitsLexicalScoped` in this file) is
|
|
236
|
-
* computed and carried on every candidate as `lexicalMatch`, but nothing
|
|
237
|
-
* downstream ever CONSULTED it as ranking evidence — `fuseByEntry` scores
|
|
238
|
-
* every tier on the same `stableFtsScore` magnitude scale (deliberately, so a
|
|
239
|
-
* relaxed hit that topped up the candidate pool floors at 0.3 instead of
|
|
240
|
-
* racing on rank), and the final comparator below sorted purely by that
|
|
241
|
-
* magnitude. `stableFtsScore`'s [0.3, 0.8] compression then flattens a large
|
|
242
|
-
* raw-BM25 gap between an all-token exact match and a two-of-three relaxed
|
|
243
|
-
* match to a few thousandths (e.g. 0.7148 vs 0.7053 for a ~6x BM25 gap) — well
|
|
244
|
-
* inside the swing of any single additive ranking contributor (alias-ranking
|
|
245
|
-
* alone is +0.3) or a belief-state ceiling. So a contributor or a ceiling,
|
|
246
|
-
* neither of which is supposed to do more than nudge, ends up DECIDING an
|
|
247
|
-
* ordering that the lexical tier — which already told us conclusively that
|
|
248
|
-
* one candidate matched every query token and the other did not — should
|
|
249
|
-
* have decided.
|
|
250
|
-
*
|
|
251
|
-
* This is the same escape hatch `aNameTier === 3` below already uses for a
|
|
252
|
-
* perfect name match, generalized to the tier ladder: exact tier is stronger
|
|
253
|
-
* evidence than prefix, which is stronger than relaxed, independent of the
|
|
254
|
-
* compressed magnitude gap between them. It sits after the name-tier-3 gate
|
|
255
|
-
* (an exact full name equality is stronger evidence still) and before the
|
|
256
|
-
* score comparison it used to lose to.
|
|
257
|
-
*/
|
|
258
|
-
const LEXICAL_TIER_RANK = { exact: 0, prefix: 1, relaxed: 2 };
|
|
259
|
-
function lexicalTierRank(tier) {
|
|
260
|
-
return tier === undefined ? 3 : LEXICAL_TIER_RANK[tier];
|
|
261
|
-
}
|
|
262
228
|
function buildSearchResultComparator(query) {
|
|
263
229
|
const queryTokens = buildLexicalQueryPlan(query).tokens.map((token) => token.toLowerCase());
|
|
264
230
|
const displayScore = (score) => Math.round(displaySearchScore(score) * 10000) / 10000;
|
|
@@ -271,9 +237,6 @@ function buildSearchResultComparator(query) {
|
|
|
271
237
|
if (nameDiff !== 0)
|
|
272
238
|
return nameDiff;
|
|
273
239
|
}
|
|
274
|
-
const tierDiff = lexicalTierRank(a.lexicalMatch) - lexicalTierRank(b.lexicalMatch);
|
|
275
|
-
if (tierDiff !== 0)
|
|
276
|
-
return tierDiff;
|
|
277
240
|
const scoreDiff = displayScore(b.score) - displayScore(a.score);
|
|
278
241
|
if (scoreDiff !== 0)
|
|
279
242
|
return scoreDiff;
|
|
@@ -282,8 +245,9 @@ function buildSearchResultComparator(query) {
|
|
|
282
245
|
return rawScoreDiff;
|
|
283
246
|
// Ceiling values are intentionally allowed to demote visibility, but not
|
|
284
247
|
// to erase relevance. Prefer the score before a relaxed body-only ceiling;
|
|
285
|
-
// a later belief-state ceiling
|
|
286
|
-
// Belief-only ceilings fall back to
|
|
248
|
+
// a later belief-state ceiling has its own minScore handoff and must not
|
|
249
|
+
// overwrite this ordering evidence. Belief-only ceilings fall back to
|
|
250
|
+
// their `preCeilingScore`.
|
|
287
251
|
const preCeilingRelevance = (item) => item.preRelaxedCeilingScore ?? item.preCeilingScore ?? item.score;
|
|
288
252
|
const ceilingDiff = stableRankScore(preCeilingRelevance(b)) - stableRankScore(preCeilingRelevance(a));
|
|
289
253
|
if (ceilingDiff !== 0)
|
|
@@ -361,13 +325,35 @@ async function searchDatabase(db, query, searchType, limit, stashDir, allSourceD
|
|
|
361
325
|
mode: "keyword",
|
|
362
326
|
};
|
|
363
327
|
}
|
|
364
|
-
// Start the async embedding request without awaiting, then run
|
|
365
|
-
//
|
|
366
|
-
// in-flight.
|
|
328
|
+
// Start the async embedding request without awaiting, then run FTS
|
|
329
|
+
// synchronously while the HTTP/local embedding request is in-flight.
|
|
367
330
|
const typeFilter = searchType === "any" ? undefined : searchType;
|
|
368
|
-
const { embedMs, mode, semanticWarning
|
|
331
|
+
const { ftsResults, embeddingScores, embedMs, mode, semanticWarning } = await collectSearchSignals(db, query, limit * 3, typeFilter, defaultExcludes, config);
|
|
369
332
|
const tRank0 = Date.now();
|
|
370
|
-
|
|
333
|
+
// ── Score normalization ──────────────────────────────────────────────
|
|
334
|
+
// Stable bounded BM25 transform + cosine similarity with weighted addition
|
|
335
|
+
// (FTS 0.7, vector 0.3). The lexical transform is per-row, so widening the
|
|
336
|
+
// candidate set cannot alter a pre-existing row's base score.
|
|
337
|
+
const ftsScoreMap = normalizeFtsScores(ftsResults);
|
|
338
|
+
// Build embedding score map (cosine similarities already 0-1)
|
|
339
|
+
const embedScoreMap = new Map();
|
|
340
|
+
if (embeddingScores) {
|
|
341
|
+
for (const [id, cosine] of embeddingScores) {
|
|
342
|
+
embedScoreMap.set(id, cosine);
|
|
343
|
+
}
|
|
344
|
+
}
|
|
345
|
+
// ── Combine FTS + vector scores ──────────────────────────────────────
|
|
346
|
+
const scored = combineSearchScores({
|
|
347
|
+
ftsScoreMap,
|
|
348
|
+
embedScoreMap,
|
|
349
|
+
getEntryById: (id) => getEntryById(db, id) ?? undefined,
|
|
350
|
+
typeFilter,
|
|
351
|
+
// #627 — also exclude default-hidden types from the vector-only branch so a
|
|
352
|
+
// session asset that is a top-k vector neighbor (but not an FTS match) does
|
|
353
|
+
// not leak into default ('any') results. defaultExcludes is already []
|
|
354
|
+
// unless this is the untyped path without includeExcludedTypes.
|
|
355
|
+
excludeTypes: defaultExcludes,
|
|
356
|
+
}).filter(hasIndexedProvenance);
|
|
371
357
|
// ── Scoring Phase ──────────────────────────────────────────────────────
|
|
372
358
|
// Apply boosts as multiplicative factors (all boosts in a single phase
|
|
373
359
|
// so that sort order and displayed scores are always consistent).
|
|
@@ -433,15 +419,21 @@ async function searchDatabase(db, query, searchType, limit, stashDir, allSourceD
|
|
|
433
419
|
positiveFeedbackCounts,
|
|
434
420
|
scopeKey,
|
|
435
421
|
});
|
|
436
|
-
//
|
|
437
|
-
//
|
|
438
|
-
//
|
|
439
|
-
//
|
|
440
|
-
//
|
|
441
|
-
//
|
|
442
|
-
|
|
422
|
+
// ── minScore floor ──────────────────────────────────────────────────────
|
|
423
|
+
// Drop semantic-only hits (cosine-only, no FTS match) whose score falls
|
|
424
|
+
// below the configured floor. FTS hits and hybrid hits are always kept.
|
|
425
|
+
// Default floor: 0.2. Set search.minScore = 0 in config to disable.
|
|
426
|
+
// Judged on the PRE-ceiling score when a demoting belief state clamped the
|
|
427
|
+
// item (`preCeilingScore`): the belief ceilings can sit below this floor
|
|
428
|
+
// (archived 0.15 < 0.2), and a demotion must rank the hit last, not
|
|
429
|
+
// silently remove a result that would otherwise have listed.
|
|
430
|
+
const minScore = config.search?.minScore ?? 0.2;
|
|
431
|
+
const preFilter = minScore > 0
|
|
432
|
+
? scored.filter((item) => item.rankingMode !== "semantic" || (item.preCeilingScore ?? item.score) >= minScore)
|
|
433
|
+
: scored;
|
|
434
|
+
preFilter.sort(buildSearchResultComparator(query));
|
|
443
435
|
// Deduplicate by file path — keep only the highest-scored entry per file.
|
|
444
|
-
const deduped = deduplicateByPath(
|
|
436
|
+
const deduped = deduplicateByPath(preFilter);
|
|
445
437
|
// Source → scope → proposed-quality → derived-twin belief inheritance →
|
|
446
438
|
// belief: the post-candidate filter chain shared with enumerateEntries (see
|
|
447
439
|
// applyEntryFilters). Applied AFTER ranking so filtering narrows the result
|
|
@@ -483,7 +475,6 @@ async function searchDatabase(db, query, searchType, limit, stashDir, allSourceD
|
|
|
483
475
|
rankingMode,
|
|
484
476
|
lexicalMatch: ranked.lexicalMatch,
|
|
485
477
|
fragmentId: ranked.fragmentId,
|
|
486
|
-
matchedUnit: ranked.matchedUnit,
|
|
487
478
|
indexedFragment: ranked.fragmentId ? (selectedFragmentByEntryId.get(ranked.id) ?? null) : undefined,
|
|
488
479
|
defaultStashDir: stashDir,
|
|
489
480
|
allSourceDirs,
|
|
@@ -498,298 +489,24 @@ async function searchDatabase(db, query, searchType, limit, stashDir, allSourceD
|
|
|
498
489
|
}));
|
|
499
490
|
return { embedMs, rankMs, hits, mode, semanticWarning };
|
|
500
491
|
}
|
|
501
|
-
// ── Units search (index-redesign-contract.md B3) ────────────────────────────
|
|
502
|
-
//
|
|
503
|
-
// Every write path (reconcile, and `indexWrittenAssets` for a just-written
|
|
504
|
-
// asset) populates `unit_texts`/`units_fts`/`entry_units` atomically with the
|
|
505
|
-
// `entries` row itself (B1's contract), so there is exactly one search path:
|
|
506
|
-
// lexical `units_fts` fused with semantic `units_vec` by evidence magnitude
|
|
507
|
-
// (`ranking.ts`'s `fuseByEntry` — see its doc for why magnitude, not rank
|
|
508
|
-
// fusion). There is no longer a coverage check to branch on — B5a's generation bump
|
|
509
|
-
// (index-schema.ts) discards `entries` outright on an incompatible schema, so
|
|
510
|
-
// a readable `entries` row always has its `entry_units` sibling.
|
|
511
|
-
/**
|
|
512
|
-
* `units_fts`/`units_vec` are keyed by UNIT, not by entry, and one entry can
|
|
513
|
-
* own several units (its structured-fields card plus one per Markdown
|
|
514
|
-
* fragment). Retrieving only `candidateLimit` units therefore yields fewer
|
|
515
|
-
* than `candidateLimit` distinct entries once grouped — this scales the
|
|
516
|
-
* requested `k` by the corpus's observed mean so entry-level recall stays
|
|
517
|
-
* comparable to the old per-entry candidate pool. 1 is the floor for a
|
|
518
|
-
* corpus with no `entry_units` rows yet (nothing to divide by).
|
|
519
|
-
*/
|
|
520
|
-
function meanUnitsPerEntry(db) {
|
|
521
|
-
const row = db
|
|
522
|
-
.prepare("SELECT AVG(cnt) AS mean FROM (SELECT COUNT(*) AS cnt FROM entry_units GROUP BY entry_id)")
|
|
523
|
-
.get();
|
|
524
|
-
const mean = row?.mean;
|
|
525
|
-
return typeof mean === "number" && Number.isFinite(mean) && mean > 0 ? mean : 1;
|
|
526
|
-
}
|
|
527
|
-
/**
|
|
528
|
-
* Build the `unit_hash IN (...)` clause that pushes a type predicate into the
|
|
529
|
-
* SQL BEFORE the candidate cap (item 3 — the confirmed defect: applying
|
|
530
|
-
* `typeFilter`/`excludeTypes` in JS after `fuseByEntry` filtered a pool that
|
|
531
|
-
* `LIMIT` already truncated could drop every eligible candidate). A unit is
|
|
532
|
-
* eligible if it has an owning entry (via `entry_units` → `entries`) that
|
|
533
|
-
* satisfies both predicates at once — the same entry, not independently
|
|
534
|
-
* matched rows — which is the correct reading for a unit hash shared by more
|
|
535
|
-
* than one entry (content-addressed reuse).
|
|
536
|
-
*/
|
|
537
|
-
function buildUnitTypeClause(typeOpts) {
|
|
538
|
-
if (!typeOpts?.typeFilter?.length && !typeOpts?.excludeTypes?.length)
|
|
539
|
-
return null;
|
|
540
|
-
const clauses = [];
|
|
541
|
-
const params = [];
|
|
542
|
-
if (typeOpts.typeFilter?.length) {
|
|
543
|
-
clauses.push(`e.type IN (${typeOpts.typeFilter.map(() => "?").join(",")})`);
|
|
544
|
-
params.push(...typeOpts.typeFilter);
|
|
545
|
-
}
|
|
546
|
-
if (typeOpts.excludeTypes?.length) {
|
|
547
|
-
clauses.push(`e.type NOT IN (${typeOpts.excludeTypes.map(() => "?").join(",")})`);
|
|
548
|
-
params.push(...typeOpts.excludeTypes);
|
|
549
|
-
}
|
|
550
|
-
return {
|
|
551
|
-
sql: `unit_hash IN (SELECT eu.unit_hash FROM entry_units eu JOIN entries e ON e.id = eu.entry_id WHERE ${clauses.join(" AND ")})`,
|
|
552
|
-
params,
|
|
553
|
-
};
|
|
554
|
-
}
|
|
555
|
-
function runUnitsFtsQuery(db, ftsQuery, lexicalMatch, k, kind, typeOpts) {
|
|
556
|
-
// `kind` filters via a subquery against `unit_texts` rather than joining
|
|
557
|
-
// (and aliasing) `units_fts` directly — FTS5's `bm25()` auxiliary function
|
|
558
|
-
// must name the exact identifier `units_fts` is referenced by in the FROM
|
|
559
|
-
// clause, so aliasing it would mean threading that alias through `bm25()`
|
|
560
|
-
// too. The subquery keeps `units_fts` unaliased and lets the MATCH still
|
|
561
|
-
// drive the query through FTS5's own index (`unit_texts_kind` then narrows
|
|
562
|
-
// it, see `files-repository.ts`). The type predicate (item 3) is pushed in
|
|
563
|
-
// the same way, and — critically — BEFORE the `LIMIT`, so an ineligible
|
|
564
|
-
// unit never occupies a slot a genuinely eligible one needed.
|
|
565
|
-
const conditions = ["units_fts MATCH ?"];
|
|
566
|
-
const params = [ftsQuery];
|
|
567
|
-
if (kind) {
|
|
568
|
-
conditions.push("unit_hash IN (SELECT unit_hash FROM unit_texts WHERE kind = ?)");
|
|
569
|
-
params.push(kind);
|
|
570
|
-
}
|
|
571
|
-
const typeClause = buildUnitTypeClause(typeOpts);
|
|
572
|
-
if (typeClause) {
|
|
573
|
-
conditions.push(typeClause.sql);
|
|
574
|
-
params.push(...typeClause.params);
|
|
575
|
-
}
|
|
576
|
-
params.push(k);
|
|
577
|
-
const rows = db
|
|
578
|
-
.prepare(`SELECT unit_hash AS unitHash, bm25(units_fts) AS score
|
|
579
|
-
FROM units_fts
|
|
580
|
-
WHERE ${conditions.join(" AND ")}
|
|
581
|
-
ORDER BY score ASC
|
|
582
|
-
LIMIT ?`)
|
|
583
|
-
.all(...params);
|
|
584
|
-
// Competition ranking (ties share a rank) rather than strict sequential
|
|
585
|
-
// position: SQLite gives no deterministic secondary order for an exact
|
|
586
|
-
// bm25 tie, and the final ranking comparator's content-based tie-break
|
|
587
|
-
// (`canonicalContentTieKey`) needs an exact score tie to survive to ever
|
|
588
|
-
// run — a strict `index + 1` would silently turn "these two units tied on
|
|
589
|
-
// relevance" into "this one wins".
|
|
590
|
-
let rank = 0;
|
|
591
|
-
let previousScore;
|
|
592
|
-
return rows.map((row, index) => {
|
|
593
|
-
if (previousScore === undefined || row.score !== previousScore)
|
|
594
|
-
rank = index + 1;
|
|
595
|
-
previousScore = row.score;
|
|
596
|
-
return { unitHash: row.unitHash, rank, bm25: row.score, lexicalMatch };
|
|
597
|
-
});
|
|
598
|
-
}
|
|
599
|
-
/**
|
|
600
|
-
* `units_fts` bm25 lexical search over unit text, ranked best-first,
|
|
601
|
-
* optionally scoped to one unit `kind`. Mirrors `searchFts`'s own exact →
|
|
602
|
-
* prefix → relaxed fallback (`index-fts-repository.ts`), but as a PRIORITY
|
|
603
|
-
* ORDER rather than an early exit (item 2): a unit is a card or one Markdown
|
|
604
|
-
* section, so a conjunctive query is rarely satisfied by any single unit —
|
|
605
|
-
* stopping at the first non-empty tier let one incidental hit (e.g. a pasted
|
|
606
|
-
* stack trace quoting every query token) suppress the far larger, more
|
|
607
|
-
* relevant relaxed pool. Instead, take the exact hits, then top up with
|
|
608
|
-
* prefix hits, then relaxed hits, until `k` is reached — each hit keeps the
|
|
609
|
-
* tier it came from in `lexicalMatch`. Magnitude fusion (`ranking.ts`'s
|
|
610
|
-
* `fuseByEntry`) is what makes topping up safe: a relaxed-tier junk match now
|
|
611
|
-
* scores at `stableFtsScore`'s 0.3 floor instead of near the top of a rank
|
|
612
|
-
* list.
|
|
613
|
-
*
|
|
614
|
-
* A genuine bm25 tie at the `k` boundary is never split across the cutoff:
|
|
615
|
-
* `addTier` finishes the whole tied group even if that pushes the result
|
|
616
|
-
* past `k`, because the final ranking comparator's content-based tie-break
|
|
617
|
-
* depends on that exact score tie surviving into `fuseByEntry`'s output.
|
|
618
|
-
* Ties are only tracked WITHIN one tier's own query — bm25 from different
|
|
619
|
-
* MATCH queries (exact vs. prefix vs. relaxed) is not comparable, so a
|
|
620
|
-
* later tier always starts its own fresh rank sequence, offset to continue
|
|
621
|
-
* numbering after the tiers already taken.
|
|
622
|
-
*/
|
|
623
|
-
function searchUnitsLexicalScoped(db, query, k, kind, typeOpts) {
|
|
624
|
-
if (k <= 0)
|
|
625
|
-
return [];
|
|
626
|
-
const plan = buildLexicalQueryPlan(query);
|
|
627
|
-
if (!plan.exact)
|
|
628
|
-
return [];
|
|
629
|
-
const hits = [];
|
|
630
|
-
const seen = new Set();
|
|
631
|
-
let rankOffset = 0;
|
|
632
|
-
const addTier = (tierHits) => {
|
|
633
|
-
let lastRank;
|
|
634
|
-
for (const hit of tierHits) {
|
|
635
|
-
if (seen.has(hit.unitHash))
|
|
636
|
-
continue;
|
|
637
|
-
if (hits.length >= k && hit.rank !== lastRank)
|
|
638
|
-
break;
|
|
639
|
-
seen.add(hit.unitHash);
|
|
640
|
-
hits.push({ ...hit, rank: hit.rank + rankOffset });
|
|
641
|
-
lastRank = hit.rank;
|
|
642
|
-
}
|
|
643
|
-
const tierMaxRank = tierHits[tierHits.length - 1]?.rank ?? 0;
|
|
644
|
-
rankOffset += tierMaxRank;
|
|
645
|
-
};
|
|
646
|
-
addTier(runUnitsFtsQuery(db, plan.exact, "exact", k, kind, typeOpts));
|
|
647
|
-
if (hits.length < k && plan.exactPrefix) {
|
|
648
|
-
addTier(runUnitsFtsQuery(db, plan.exactPrefix, "prefix", k, kind, typeOpts));
|
|
649
|
-
}
|
|
650
|
-
if (hits.length < k && plan.relaxed) {
|
|
651
|
-
addTier(runUnitsFtsQuery(db, plan.relaxed, "relaxed", k, kind, typeOpts));
|
|
652
|
-
}
|
|
653
|
-
return hits;
|
|
654
|
-
}
|
|
655
|
-
/**
|
|
656
|
-
* `units_fts` bm25 lexical search over EVERY unit, kind-agnostic — the
|
|
657
|
-
* original single-pool query, kept for callers that want one flat
|
|
658
|
-
* entry-level lexical ranking rather than the card/fragment split
|
|
659
|
-
* `collectSearchSignals` uses (below): `searchEntriesLexical`'s
|
|
660
|
-
* deterministic-only canary scoring for collapse-detector, which has no use
|
|
661
|
-
* for field emphasis.
|
|
662
|
-
*/
|
|
663
|
-
export function searchUnitsLexical(db, query, k) {
|
|
664
|
-
return searchUnitsLexicalScoped(db, query, k);
|
|
665
|
-
}
|
|
666
|
-
/**
|
|
667
|
-
* `units_fts` bm25 lexical search over BOTH kind-scoped pools (`"card"`,
|
|
668
|
-
* `"fragment"`) at once (index-redesign-contract.md B5f item 2). Structural
|
|
669
|
-
* field emphasis: `"card"` units hold name/description/tags/hints and are
|
|
670
|
-
* few (one per entry), so a name match ranks near the top of a SMALL pool
|
|
671
|
-
* instead of racing every fragment's body text in one shared BM25 ranking —
|
|
672
|
-
* the same effect the old per-column BM25 weights (name 10x, description
|
|
673
|
-
* 5x, ...) bought through tuning, gotten here from the units' own structure
|
|
674
|
-
* instead.
|
|
675
|
-
*
|
|
676
|
-
* Each pool runs its OWN exact → prefix → relaxed priority-order ladder
|
|
677
|
-
* (item 2 — `searchUnitsLexicalScoped`'s own doc), rather than sharing one
|
|
678
|
-
* tier decision as an earlier revision did: sharing let an incidental
|
|
679
|
-
* fragment-exact match (e.g. a pasted stack trace quoting every query token)
|
|
680
|
-
* lock the card pool out of ever escalating to its own relaxed recovery, so
|
|
681
|
-
* a well-named relevant entry disappeared behind an unrelated log dump.
|
|
682
|
-
* Magnitude fusion (`ranking.ts`'s `fuseByEntry`) is what makes independent
|
|
683
|
-
* ladders safe: a relaxed-tier junk match now scores at `stableFtsScore`'s
|
|
684
|
-
* 0.3 floor instead of competing on rank, so a stray fragment-side escalation
|
|
685
|
-
* can no longer crowd out a genuine card-side exact hit the way it would
|
|
686
|
-
* have under rank fusion.
|
|
687
|
-
*/
|
|
688
|
-
export function searchUnitsLexicalPair(db, query, k, typeOpts) {
|
|
689
|
-
return {
|
|
690
|
-
card: searchUnitsLexicalScoped(db, query, k, "card", typeOpts),
|
|
691
|
-
fragment: searchUnitsLexicalScoped(db, query, k, "fragment", typeOpts),
|
|
692
|
-
};
|
|
693
|
-
}
|
|
694
|
-
/** Count of `units` rows for the active identity. */
|
|
695
|
-
function getUnitVectorCount(db, identity) {
|
|
696
|
-
try {
|
|
697
|
-
const row = db.prepare("SELECT COUNT(*) AS cnt FROM units WHERE identity = ?").get(identity);
|
|
698
|
-
return row?.cnt ?? 0;
|
|
699
|
-
}
|
|
700
|
-
catch {
|
|
701
|
-
// The design doc's migration story has units_fts populated (lexical
|
|
702
|
-
// ready) before the first embedding drain completes (`units` empty or
|
|
703
|
-
// absent) — an expected transient state, not a fault. Lexical-only
|
|
704
|
-
// results are the correct behavior until the drain catches up.
|
|
705
|
-
return 0;
|
|
706
|
-
}
|
|
707
|
-
}
|
|
708
|
-
async function tryUnitVecScores(db, query, k, config, typeOpts) {
|
|
709
|
-
if (config.semanticSearchMode === "off")
|
|
710
|
-
return { hits: null };
|
|
711
|
-
const identity = getMeta(db, "embeddingIdentity");
|
|
712
|
-
if (!identity || getUnitVectorCount(db, identity) === 0)
|
|
713
|
-
return { hits: null };
|
|
714
|
-
try {
|
|
715
|
-
const { embed } = await import("../../llm/embedder.js");
|
|
716
|
-
const queryEmbedding = await embed(query, config.embedding);
|
|
717
|
-
// item 5 — a query embedded under a different identity than the index
|
|
718
|
-
// must not be trusted, even when it happens to come back the same width
|
|
719
|
-
// (768/1024/1536 are all common across otherwise-unrelated models): a
|
|
720
|
-
// width match alone is not a vector-space match. `deriveObservedEmbeddingIdentity`
|
|
721
|
-
// is the same derivation `drain.ts` uses to learn/verify the identity it
|
|
722
|
-
// is embedding units under; there is no server-reported model for a
|
|
723
|
-
// single query `embed()` call (only `embedBatch`'s `onBatch` threads that
|
|
724
|
-
// through from the provider's response), so this derives from the
|
|
725
|
-
// CURRENT config the same way drain does whenever the provider's
|
|
726
|
-
// response echoes the configured model — the case `embedding.model`/
|
|
727
|
-
// `embedding.endpoint` being edited since the last index actually
|
|
728
|
-
// exercises. A genuine width mismatch is already safe (sqlite-vec throws
|
|
729
|
-
// below); this catches the same-width, different-model case that would
|
|
730
|
-
// otherwise silently compare incompatible vector spaces.
|
|
731
|
-
const observedIdentity = deriveObservedEmbeddingIdentity(config.embedding, undefined, queryEmbedding.length);
|
|
732
|
-
if (observedIdentity !== identity) {
|
|
733
|
-
return { hits: null, warning: buildIdentityMismatchWarning(config) };
|
|
734
|
-
}
|
|
735
|
-
return { hits: searchUnits(db, queryEmbedding, k, identity, typeOpts) };
|
|
736
|
-
}
|
|
737
|
-
catch (error) {
|
|
738
|
-
return { hits: null, warning: buildVectorFallbackWarning(config, error) };
|
|
739
|
-
}
|
|
740
|
-
}
|
|
741
492
|
async function collectSearchSignals(db, query, candidateLimit, typeFilter, excludeTypes, config) {
|
|
742
493
|
const startedAt = Date.now();
|
|
743
|
-
const
|
|
744
|
-
|
|
745
|
-
|
|
746
|
-
|
|
747
|
-
// `unitK` by `LIMIT`/`k` let a whole type-excluded pool (e.g. 100 `session`
|
|
748
|
-
// cards exactly matching the query) crowd a genuinely-matching entry of a
|
|
749
|
-
// different type out of the candidate window entirely. `fuseByEntry`'s own
|
|
750
|
-
// filter stays as a cheap guard, not the mechanism.
|
|
751
|
-
const typeOpts = { typeFilter: typeFilter ? [typeFilter] : undefined, excludeTypes };
|
|
752
|
-
const semanticPromise = tryUnitVecScores(db, query, unitK, config, typeOpts);
|
|
753
|
-
// index-redesign-contract.md B5f item 2 — two kind-scoped lexical lists,
|
|
754
|
-
// not one mixed pool: a card (name/description/tags/hints) match ranks
|
|
755
|
-
// within its own small pool instead of competing against every fragment's
|
|
756
|
-
// body text on raw BM25, so field emphasis falls out of the units'
|
|
757
|
-
// structure rather than tuned per-column weights. Each pool runs its own
|
|
758
|
-
// priority-order ladder — see `searchUnitsLexicalPair`'s own doc.
|
|
759
|
-
const { card: cardLexicalHits, fragment: fragmentLexicalHits } = searchUnitsLexicalPair(db, query, unitK, typeOpts);
|
|
760
|
-
const semanticResult = await semanticPromise;
|
|
761
|
-
const mode = semanticResult.warning
|
|
494
|
+
const embeddingPromise = tryVecScores(db, query, candidateLimit, config);
|
|
495
|
+
const ftsResults = searchFts(db, query, candidateLimit, typeFilter, excludeTypes);
|
|
496
|
+
const embeddingResult = await embeddingPromise;
|
|
497
|
+
const mode = embeddingResult.warning
|
|
762
498
|
? "fts-fallback"
|
|
763
|
-
:
|
|
499
|
+
: embeddingResult.scores !== null
|
|
764
500
|
? "semantic"
|
|
765
501
|
: "keyword";
|
|
766
|
-
const unitScored = fuseByEntry(db, cardLexicalHits, fragmentLexicalHits, semanticResult.hits ?? [], typeOpts);
|
|
767
502
|
return {
|
|
503
|
+
ftsResults,
|
|
504
|
+
embeddingScores: embeddingResult.scores,
|
|
768
505
|
embedMs: Date.now() - startedAt,
|
|
769
506
|
mode,
|
|
770
|
-
semanticWarning:
|
|
771
|
-
unitScored,
|
|
507
|
+
semanticWarning: embeddingResult.warning,
|
|
772
508
|
};
|
|
773
509
|
}
|
|
774
|
-
/**
|
|
775
|
-
* Entry-level lexical-only search over units, best match first — for
|
|
776
|
-
* consumers that need ranked entries without semantic fusion (e.g.
|
|
777
|
-
* collapse-detector's canary scoring, which is deterministic-only by design:
|
|
778
|
-
* see `src/commands/improve/collapse-detector.ts`). The `units_fts` card unit
|
|
779
|
-
* carries name/description/tags/hints, so an entry-level lexical search is a
|
|
780
|
-
* units query grouped by entry — the same grouping `collectSearchSignals`
|
|
781
|
-
* uses, with an empty semantic list so `fuseByEntry`'s magnitude fusion
|
|
782
|
-
* degenerates to a pure lexical-bm25 ordering.
|
|
783
|
-
*/
|
|
784
|
-
export function searchEntriesLexical(db, query, k) {
|
|
785
|
-
const unitK = Math.max(1, Math.round(k * meanUnitsPerEntry(db)));
|
|
786
|
-
const lexicalHits = searchUnitsLexical(db, query, unitK);
|
|
787
|
-
// One flat kind-agnostic pool, not the card/fragment split
|
|
788
|
-
// `collectSearchSignals` uses — deliberately: this is the kind-agnostic
|
|
789
|
-
// single-list mode `fuseByEntry` still supports for a caller with no use
|
|
790
|
-
// for field emphasis (see `searchUnitsLexical`'s own doc).
|
|
791
|
-
return fuseByEntry(db, lexicalHits, [], []).sort((a, b) => b.score - a.score);
|
|
792
|
-
}
|
|
793
510
|
/**
|
|
794
511
|
* The no-hits tip. A query in the retired `<type>:` / `<type>:<prefix>/` browse
|
|
795
512
|
* grammar gets the conceptId spelling that replaces it: without this it comes
|
|
@@ -971,6 +688,33 @@ function matchBeliefFilter(beliefState, filter) {
|
|
|
971
688
|
beliefState === "archived");
|
|
972
689
|
}
|
|
973
690
|
// ── Vector scorer ───────────────────────────────────────────────────────────
|
|
691
|
+
async function tryVecScores(db, query, k, config) {
|
|
692
|
+
if (config.semanticSearchMode === "off")
|
|
693
|
+
return { scores: null };
|
|
694
|
+
// A real-time completeness fact, not a cached verdict: skip the network
|
|
695
|
+
// round trip only when the index has never embedded anything. A PARTIAL
|
|
696
|
+
// failure (some entries embedded, one write degraded) still attempts —
|
|
697
|
+
// and if the endpoint is genuinely down, the failure surfaces as a live
|
|
698
|
+
// `semanticWarning` below instead of silently skipping with no signal.
|
|
699
|
+
if (getEmbeddingCount(db) === 0)
|
|
700
|
+
return { scores: null };
|
|
701
|
+
try {
|
|
702
|
+
const { embed } = await import("../../llm/embedder.js");
|
|
703
|
+
const queryEmbedding = await embed(query, config.embedding);
|
|
704
|
+
const vecResults = searchVec(db, queryEmbedding, k);
|
|
705
|
+
const scores = new Map();
|
|
706
|
+
for (const { id, distance } of vecResults) {
|
|
707
|
+
// Convert L2 distance to cosine similarity (vectors are normalized).
|
|
708
|
+
// Guard against NaN/Infinity from sqlite-vec edge cases.
|
|
709
|
+
const raw = 1 - (distance * distance) / 2;
|
|
710
|
+
scores.set(id, Number.isFinite(raw) ? Math.max(0, raw) : 0);
|
|
711
|
+
}
|
|
712
|
+
return { scores };
|
|
713
|
+
}
|
|
714
|
+
catch (error) {
|
|
715
|
+
return { scores: null, warning: buildVectorFallbackWarning(config, error) };
|
|
716
|
+
}
|
|
717
|
+
}
|
|
974
718
|
function buildVectorFallbackWarning(config, error) {
|
|
975
719
|
const endpoint = safeEmbeddingEndpoint(config);
|
|
976
720
|
const reason = classifyVectorFailure(error);
|
|
@@ -982,21 +726,6 @@ function buildVectorFallbackWarning(config, error) {
|
|
|
982
726
|
const unavailable = reason === "connection failed" ? `cannot reach ${target}` : `${target} is unavailable`;
|
|
983
727
|
return `Vector search unavailable: ${unavailable} (${reason}) — falling back to keyword search.`;
|
|
984
728
|
}
|
|
985
|
-
/**
|
|
986
|
-
* item 5 — same shape as {@link buildVectorFallbackWarning}, for the case
|
|
987
|
-
* where embedding itself succeeded but the query was embedded under a
|
|
988
|
-
* different identity than the index (`embedding.model`/`embedding.endpoint`
|
|
989
|
-
* edited since the last index run).
|
|
990
|
-
*/
|
|
991
|
-
function buildIdentityMismatchWarning(config) {
|
|
992
|
-
const endpoint = safeEmbeddingEndpoint(config);
|
|
993
|
-
const target = endpoint
|
|
994
|
-
? `embedding endpoint ${endpoint}`
|
|
995
|
-
: config.embedding?.endpoint
|
|
996
|
-
? "configured embedding endpoint"
|
|
997
|
-
: "local embedding model";
|
|
998
|
-
return `Vector search unavailable: ${target} is embedding queries under a different identity than the index was built with (embedding.model/embedding.endpoint changed since the last index) — falling back to keyword search.`;
|
|
999
|
-
}
|
|
1000
729
|
/**
|
|
1001
730
|
* Name the useful endpoint without ever carrying URL userinfo, query secrets,
|
|
1002
731
|
* or fragments into a warning. Invalid authored values fail closed.
|
|
@@ -1063,36 +792,24 @@ export async function buildDbHit(input) {
|
|
|
1063
792
|
? (input.bundleId ?? undefined)
|
|
1064
793
|
: undefined);
|
|
1065
794
|
const parentRef = resolveSearchHitRef(input.entry, input, defaultBundleId);
|
|
1066
|
-
//
|
|
1067
|
-
//
|
|
1068
|
-
//
|
|
1069
|
-
|
|
1070
|
-
// every consumer (a judgment, a stored `derivedFrom`, a copy-pasted CLI
|
|
1071
|
-
// command) that names the bare entry. A consumer that genuinely wants the
|
|
1072
|
-
// matched fragment's own ref reads `selectedRef` below instead — computed
|
|
1073
|
-
// exactly the way `ref` itself used to be, so its availability (gated by
|
|
1074
|
-
// `allowsFragmentRef`) is unchanged; only the PRIMARY ref stopped carrying it.
|
|
1075
|
-
const ref = parentRef;
|
|
795
|
+
// Fragments prove lexical relevance, but executable assets must retain the
|
|
796
|
+
// parent ref consumed by their advertised action (for example workflow run).
|
|
797
|
+
// The central type-presentation contract opts those types out explicitly.
|
|
798
|
+
const ref = input.fragmentId && allowsFragmentRef(input.entry.type) ? `${parentRef}#${input.fragmentId}` : parentRef;
|
|
1076
799
|
const editable = isEditable(absolutePath, input.config, input.sources);
|
|
1077
800
|
const indexedFragment = input.indexedFragment === undefined
|
|
1078
801
|
? input.fragmentId && input.db
|
|
1079
802
|
? getIndexedMarkdownFragment(input.db, input.itemRef, input.fragmentId)
|
|
1080
803
|
: undefined
|
|
1081
804
|
: (input.indexedFragment ?? undefined);
|
|
1082
|
-
|
|
1083
|
-
// parent ref consumed by their advertised action (for example workflow run).
|
|
1084
|
-
// The central type-presentation contract opts those types out explicitly.
|
|
1085
|
-
const selectedRef = input.fragmentId && allowsFragmentRef(input.entry.type) ? `${parentRef}#${input.fragmentId}` : undefined;
|
|
805
|
+
const selectedRef = input.fragmentId && ref !== parentRef ? `${parentRef}#${input.fragmentId}` : undefined;
|
|
1086
806
|
const parentEstimatedTokens = typeof input.entry.fileSize === "number"
|
|
1087
807
|
? Math.round(input.entry.fileSize / 4)
|
|
1088
808
|
: indexedFragment
|
|
1089
809
|
? Math.round(indexedFragment.parentChars / 4)
|
|
1090
810
|
: undefined;
|
|
1091
811
|
const fragmentEstimatedTokens = indexedFragment ? Math.round(indexedFragment.fragmentChars / 4) : undefined;
|
|
1092
|
-
|
|
1093
|
-
// is always the parent's — a caller that wants the fragment's own size reads
|
|
1094
|
-
// `fragmentEstimatedTokens` from the `selectedRef` block below.
|
|
1095
|
-
const estimatedTokens = parentEstimatedTokens;
|
|
812
|
+
const estimatedTokens = selectedRef === ref && fragmentEstimatedTokens !== undefined ? fragmentEstimatedTokens : parentEstimatedTokens;
|
|
1096
813
|
const hit = {
|
|
1097
814
|
type: input.entry.type,
|
|
1098
815
|
name: input.entry.name,
|
|
@@ -1144,7 +861,6 @@ export async function buildDbHit(input) {
|
|
|
1144
861
|
// hit. Omitted when the hit has no FTS component (pure-semantic hybrid
|
|
1145
862
|
// contribution).
|
|
1146
863
|
...(input.lexicalMatch ? { matchStage: input.lexicalMatch } : {}),
|
|
1147
|
-
...(input.matchedUnit ? { matchedUnit: input.matchedUnit } : {}),
|
|
1148
864
|
};
|
|
1149
865
|
attachDbHitAttribution(hit, input);
|
|
1150
866
|
if (input.entry.derivedFrom) {
|
|
@@ -106,25 +106,20 @@ function beliefStateBoost(item) {
|
|
|
106
106
|
* stash-conventions-code-spec.md — corrections demotion).
|
|
107
107
|
*
|
|
108
108
|
* Why the additive {@link beliefStateBoost} penalties alone are not enough:
|
|
109
|
-
* keyword base scores have a bounded lexical floor (`
|
|
110
|
-
*
|
|
111
|
-
*
|
|
112
|
-
*
|
|
113
|
-
*
|
|
114
|
-
*
|
|
115
|
-
* the corrections pattern's point ("so the ranker demotes the stale version
|
|
116
|
-
* instead of letting it outrank your fix").
|
|
109
|
+
* keyword base scores have a bounded lexical floor (`normalizeFtsScores`),
|
|
110
|
+
* while the boost sum then MULTIPLIES the base (`score *= 1 + boostSum`,
|
|
111
|
+
* {@link applyScoreContributors}). A superseded incumbent can still earn
|
|
112
|
+
* enough independent boosts to outrank its own correction, so additive
|
|
113
|
+
* penalties alone cannot guarantee the corrections pattern's point ("so the
|
|
114
|
+
* ranker demotes the stale version instead of letting it outrank your fix").
|
|
117
115
|
*
|
|
118
116
|
* The ceilings guarantee the demotion while keeping flagged entries VISIBLE:
|
|
119
|
-
* un-demoted keyword hits floor at
|
|
120
|
-
*
|
|
121
|
-
*
|
|
122
|
-
*
|
|
123
|
-
*
|
|
124
|
-
*
|
|
125
|
-
* additive-penalty severity order pinned in
|
|
126
|
-
* tests/integration/belief-state-phase1a.test.ts: deprecated (mildest) >
|
|
127
|
-
* superseded > contradicted > archived.
|
|
117
|
+
* un-demoted keyword hits floor at a 0.3 base, so any un-demoted hit outranks
|
|
118
|
+
* a ceilinged one; demoted entries still list (belief FILTERING stays a
|
|
119
|
+
* separate opt-in axis, `--belief`), and scores already below a ceiling keep
|
|
120
|
+
* their relative ordering. Ceiling order mirrors the additive-penalty
|
|
121
|
+
* severity order pinned in tests/integration/belief-state-phase1a.test.ts:
|
|
122
|
+
* deprecated (mildest) > superseded > contradicted > archived.
|
|
128
123
|
*/
|
|
129
124
|
const BELIEF_STATE_SCORE_CEILINGS = {
|
|
130
125
|
deprecated: 0.28,
|
|
@@ -139,10 +134,10 @@ const BELIEF_STATE_SCORE_CEILINGS = {
|
|
|
139
134
|
* order and displayed scores stay consistent (single scoring pipeline).
|
|
140
135
|
*
|
|
141
136
|
* When the ceiling clamps, the pre-clamp score is recorded as
|
|
142
|
-
* `preCeilingScore` so db-search's
|
|
143
|
-
*
|
|
144
|
-
*
|
|
145
|
-
* silently
|
|
137
|
+
* `preCeilingScore` so db-search's semantic-only `minScore` floor can judge
|
|
138
|
+
* the hit by what it would have scored WITHOUT the demotion — a ceiling below
|
|
139
|
+
* the floor (archived 0.15 < default minScore 0.2) must demote a hit to last
|
|
140
|
+
* place, never silently drop it from the results.
|
|
146
141
|
*/
|
|
147
142
|
export function applyBeliefStateScoreCeiling(item) {
|
|
148
143
|
const state = item.entry.beliefState;
|