@modusensus/dsh-mneme 0.4.6 → 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/api.js CHANGED
@@ -103,7 +103,7 @@ export function createApi(ctx, service, settings, commands, embedder, semantic =
103
103
  try {
104
104
  const url = new URL(req.url, "http://localhost");
105
105
  const q = url.searchParams.get("q") ?? "";
106
- const limit = Number(url.searchParams.get("limit") ?? 20);
106
+ const limit = Number(url.searchParams.get("topK") ?? url.searchParams.get("limit") ?? 20);
107
107
  // mode selects the recall strategy (defaults to auto):
108
108
  // auto (default) keyword first, vector fills remaining slots
109
109
  // hybrid vector first, keyword fills remaining slots; scores of
@@ -306,6 +306,122 @@ export function createApi(ctx, service, settings, commands, embedder, semantic =
306
306
  }
307
307
  });
308
308
 
309
+ // --- ego graph: 1-2 hop neighborhood of one entity (graph panel P1) ---
310
+ // Read-only like list/search/semantic, so it stays open when apiToken is set.
311
+ // BFS from the root entity over entity_relations (both directions; the
312
+ // idx_relations_from/to indexes keep a 2-hop walk in the tens of ms even
313
+ // for a few thousand nodes). `distance` on each node is the hop count from
314
+ // the root so the UI can shade the frontier. The API is graph-traversal
315
+ // only — nodes carry no attr payload; hover summaries come from
316
+ // /semantic/graph/entity-attrs.
317
+ register({
318
+ kind: "exact",
319
+ path: "/api/dsh-mneme/semantic/graph/ego",
320
+ handler(req, res) {
321
+ try {
322
+ const url = new URL(req.url, "http://localhost");
323
+ const name = (url.searchParams.get("entity") ?? "").trim();
324
+ if (!name) {
325
+ sendJson(res, 400, { error: "missing-entity" });
326
+ return;
327
+ }
328
+ const root = service.findEntityByName?.(name);
329
+ if (!root) {
330
+ sendJson(res, 404, { error: "entity-not-found" });
331
+ return;
332
+ }
333
+ const depth = Math.max(1, Math.min(2, Number(url.searchParams.get("depth") ?? 1) || 1));
334
+ const limit = Math.max(1, Math.min(100, Number(url.searchParams.get("limit") ?? 40) || 40));
335
+
336
+ const nodes = new Map([[root.id, { ...root, distance: 0 }]]);
337
+ let frontier = [root.id];
338
+ for (let d = 1; d <= depth && nodes.size < limit; d++) {
339
+ const next = [];
340
+ for (const id of frontier) {
341
+ for (const rel of service.getRelations?.(id) ?? []) {
342
+ const other = rel.from_entity === id ? rel.to_entity : rel.from_entity;
343
+ if (nodes.has(other) || nodes.size >= limit) continue;
344
+ const entity = service.findEntityById?.(other);
345
+ if (!entity) continue;
346
+ nodes.set(other, { ...entity, distance: d });
347
+ next.push(other);
348
+ }
349
+ }
350
+ frontier = next;
351
+ }
352
+
353
+ // Collect every relation whose endpoints both survived the limit cut;
354
+ // each edge is visited twice (once per endpoint) so dedupe by id.
355
+ const edgeMap = new Map();
356
+ for (const id of nodes.keys()) {
357
+ for (const rel of service.getRelations?.(id) ?? []) {
358
+ if (nodes.has(rel.from_entity) && nodes.has(rel.to_entity)) {
359
+ edgeMap.set(rel.id, rel);
360
+ }
361
+ }
362
+ }
363
+
364
+ sendJson(res, 200, {
365
+ root: { id: root.id, name: root.name, type: root.type ?? null, mention_count: root.mention_count ?? 1 },
366
+ nodes: [...nodes.values()].map((n) => ({
367
+ id: n.id,
368
+ name: n.name,
369
+ type: n.type ?? null,
370
+ mention_count: n.mention_count ?? 1,
371
+ distance: n.distance
372
+ })),
373
+ edges: [...edgeMap.values()].map((e) => ({
374
+ id: e.id,
375
+ from: e.from_entity,
376
+ to: e.to_entity,
377
+ relation_type: e.relation_type,
378
+ memory_id: e.memory_id ?? null,
379
+ created_at: e.created_at
380
+ }))
381
+ });
382
+ } catch {
383
+ sendJson(res, 500, { error: "internal" });
384
+ }
385
+ }
386
+ });
387
+
388
+ // --- entity attrs: current valid attrs for one entity (graph hover panel) ---
389
+ // Read-only; mirrors getCurrentAttrs (valid_until IS NULL). Also used as the
390
+ // graph panel's fallback list when the ego graph is too sparse to draw.
391
+ register({
392
+ kind: "exact",
393
+ path: "/api/dsh-mneme/semantic/graph/entity-attrs",
394
+ handler(req, res) {
395
+ try {
396
+ const url = new URL(req.url, "http://localhost");
397
+ const name = (url.searchParams.get("entity") ?? "").trim();
398
+ if (!name) {
399
+ sendJson(res, 400, { error: "missing-entity" });
400
+ return;
401
+ }
402
+ const entity = service.findEntityByName?.(name);
403
+ if (!entity) {
404
+ sendJson(res, 404, { error: "entity-not-found" });
405
+ return;
406
+ }
407
+ const attrs = service.getCurrentAttrs?.(entity.id) ?? [];
408
+ sendJson(res, 200, {
409
+ entity: { id: entity.id, name: entity.name, type: entity.type ?? null, mention_count: entity.mention_count ?? 1 },
410
+ attrs: Array.isArray(attrs)
411
+ ? attrs.map((a) => ({
412
+ key: a.attr_key,
413
+ value: a.attr_value,
414
+ confidence: a.confidence ?? null,
415
+ valid_from: a.valid_from ?? null
416
+ }))
417
+ : []
418
+ });
419
+ } catch {
420
+ sendJson(res, 500, { error: "internal" });
421
+ }
422
+ }
423
+ });
424
+
309
425
  // --- health: mirror sync state (F-NEW-03 / v0.3.6) ---
310
426
  // Auth-gated; only returns a sanitized error code (never raw last_error which
311
427
  // may leak paths/token-like strings/internal hosts). On state read failure it
package/src/config.js CHANGED
@@ -17,7 +17,7 @@ export const Config = z.object({
17
17
  dreamDelayMs: z.natural().min(0).max(60000).default(2000),
18
18
  dreamProvider: z.string(),
19
19
  dreamModel: z.string(),
20
- dreamMaxTokens: z.natural().min(256).max(131072).default(4096),
20
+ dreamMaxTokens: z.natural().min(256).max(131072).default(8192),
21
21
  // Pass-through reasoning effort for dream's LLM calls. 'none' (default)
22
22
  // omits the field so the provider's own default applies; low/medium/high
23
23
  // are forwarded verbatim. Useful to cap reasoning spend on thinking-type
@@ -92,6 +92,35 @@ export const Config = z.object({
92
92
  // rule-based pick to fill/dedupe. Empty query / no vector → legacy behavior.
93
93
  hybridInject: z.boolean().default(true),
94
94
 
95
+ // --- recall optimization (v0.5.0) ----------------------------------------
96
+ // BM25 third recall path beside vector + LIKE keyword (1.1): per-token IDF
97
+ // scoring recalls rows whose query terms are scattered — identifiers, code
98
+ // fragments, mixed CJK/ASCII — where substring LIKE cannot match.
99
+ bm25SearchEnabled: z.boolean().default(true),
100
+ // Query-aware vector cutoff (1.2) replacing the fixed 0.65: entity:/attr:
101
+ // prefixes loosen to 0.5, short queries tighten to 0.7, long queries loosen
102
+ // to 0.6, and a decisive top-1/top-5 score gap loosens to 0.5 so the tail
103
+ // still reaches the reranker. Off = legacy fixed threshold behavior.
104
+ adaptiveThresholdEnabled: z.boolean().default(true),
105
+ // Session-scoped hot memory (1.3): the latest N dialogue rounds rendered
106
+ // ahead of the long-term recall block — short-term context that never
107
+ // enters the memory store.
108
+ hotMemoryEnabled: z.boolean().default(true),
109
+ hotMemoryRounds: z.natural().min(1).max(50).default(5),
110
+ hotMemoryMaxTokens: z.natural().min(200).max(32000).default(2000),
111
+ // Topic-ranked injection (2.2): when a query vector is available the whole
112
+ // injection candidate list is re-ordered by similarity to the current
113
+ // query instead of keeping the rule-based order.
114
+ selectiveInjectEnabled: z.boolean().default(true),
115
+ // Search-time semantic dedup (2.3): greedy pass over the merged candidate
116
+ // list dropping rows whose embedding cosine-similarity to an already-kept
117
+ // row exceeds the threshold — duplicates are filtered at recall time
118
+ // instead of waiting for a dream consolidation. Opt-in aggressive mode:
119
+ // small embedding models can collapse legitimately distinct rows, so the
120
+ // default keeps every recalled row.
121
+ searchSemanticDedup: z.boolean().default(false),
122
+ searchSemanticDedupThreshold: z.number().min(0.5).max(1).default(0.95),
123
+
95
124
  // --- semantic: rerank layer (v0.2) --------------------------------------
96
125
  // Opt-in by default (item ⑥): the local cross-encoder pulls in onnxruntime
97
126
  // (transformers.js) at init, so a bare install must not load it. Only an
package/src/dream.js CHANGED
@@ -3,6 +3,43 @@ import { clusterMemories, findPotentialConflicts } from "./dream/clustering.js";
3
3
  import { createHash, randomUUID } from "node:crypto";
4
4
  export { validateDecisions, applyDecisions };
5
5
 
6
+
7
+ // Extract the first JSON array from LLM output, tolerating markdown fences,
8
+ // leading/trailing prose, and common wrapper noise. Returns an array or null.
9
+ function extractJsonArray(text) {
10
+ if (typeof text !== "string" || text.trim().length === 0) return null;
11
+
12
+ // 1. Strip markdown code fences (```json ... ``` or ``` ... ```).
13
+ let cleaned = text.replace(/```(?:json)?\s*([\s\S]*?)```/gi, "$1");
14
+ cleaned = cleaned.trim();
15
+
16
+ // 2. Find the first '[' and the matching last ']' that yields valid JSON.
17
+ const start = cleaned.indexOf("[");
18
+ if (start === -1) return null;
19
+ for (let end = cleaned.lastIndexOf("]"); end > start; end = cleaned.lastIndexOf("]", end - 1)) {
20
+ const candidate = cleaned.slice(start, end + 1);
21
+ try {
22
+ return JSON.parse(candidate);
23
+ } catch {
24
+ // Light repair: remove trailing commas before ] or }.
25
+ try {
26
+ const repaired = candidate.replace(/,(\s*[}\]])/g, "$1");
27
+ return JSON.parse(repaired);
28
+ } catch {
29
+ // keep searching backwards
30
+ }
31
+ }
32
+ }
33
+
34
+ // 3. Fallback: a broader regex extraction.
35
+ try {
36
+ const match = cleaned.match(/\[[\s\S]*\]/);
37
+ if (match) return JSON.parse(match[0]);
38
+ } catch {
39
+ // fall through
40
+ }
41
+ return null;
42
+ }
6
43
  const SUMMARY_PROMPT = `你是记忆库摘要助手。根据整理后的记忆,生成一段 150-200 字的记忆库总览,覆盖:用户偏好、活跃项目、关键决策。之后作为会话上下文注入。只输出摘要文本,不要其他内容。`;
7
44
 
8
45
  const CONSOLIDATION_PROMPT = `你是记忆库整理助手。下面是全部记忆条目(id、类型、标题、内容、重要性、更新时间)。
@@ -579,18 +616,10 @@ export function createDreamScheduler({ onRun, thresholdCount = 10, thresholdChar
579
616
  return finish({ ok: false, error: "llm failed", summary: false });
580
617
  }
581
618
 
582
- let decisions;
583
- try {
584
- const start = decisionText.indexOf("[");
585
- const end = decisionText.lastIndexOf("]");
586
- if (start === -1 || end <= start) {
587
- logger?.warn?.("dsh-mneme dream: no json array in llm output");
588
- return finish({ ok: false, error: "no json array in llm output", summary: false });
589
- }
590
- decisions = JSON.parse(decisionText.slice(start, end + 1));
591
- } catch {
592
- logger?.warn?.("dsh-mneme dream: invalid decisions json");
593
- return finish({ ok: false, error: "invalid decisions json", summary: false });
619
+ const decisions = extractJsonArray(decisionText);
620
+ if (!Array.isArray(decisions)) {
621
+ logger?.warn?.(`dsh-mneme dream: no json array in llm output (raw length ${decisionText?.length ?? 0})`);
622
+ return finish({ ok: false, error: "no json array in llm output", summary: false });
594
623
  }
595
624
  const { ok, errors } = validateDecisions(decisions, snapshot, {
596
625
  maxUpdatePerRun: config.reflectionUpdateMaxPerRun,
@@ -0,0 +1,46 @@
1
+ // Session-scoped hot memory (v0.5.0 召回率优化 1.3): a short-term buffer of
2
+ // the latest dialogue rounds, kept strictly apart from the long-term memory
3
+ // store. The injector renders it ahead of the long-term recall block so the
4
+ // agent sees "what we were just talking about" without those rounds ever
5
+ // being persisted as memories. Bounded two ways: maxRounds (count) and
6
+ // maxTokens (budget) — whichever evicts first.
7
+
8
+ // CJK-aware token estimate: one Chinese character ≈ 0.6 tokens (clustering
9
+ // behavior of mainstream tokenizers), one ASCII char ≈ 0.25.
10
+ export function estimateTokens(text) {
11
+ const s = String(text ?? "");
12
+ let cjk = 0;
13
+ for (const ch of s) if (ch >= "\u4e00" && ch <= "\u9fff") cjk++;
14
+ return Math.ceil(cjk * 0.6 + (s.length - cjk) * 0.25);
15
+ }
16
+
17
+ /**
18
+ * @param {{maxRounds?: number, maxTokens?: number}} opts
19
+ * @returns {{add(round: {query: string, response?: string}): void,
20
+ * getContext(): string,
21
+ * rounds(): Array, clear(): void}}
22
+ */
23
+ export function createHotMemory({ maxRounds = 5, maxTokens = 2000 } = {}) {
24
+ const buffer = [];
25
+
26
+ function totalTokens() {
27
+ return buffer.reduce(
28
+ (sum, r) => sum + estimateTokens(`Q: ${r.query}\nA: ${r.response ?? ""}`),
29
+ 0
30
+ );
31
+ }
32
+
33
+ return {
34
+ add(round) {
35
+ if (!round?.query) return;
36
+ buffer.push({ query: String(round.query), response: String(round.response ?? "") });
37
+ while (buffer.length > maxRounds) buffer.shift();
38
+ while (buffer.length > 1 && totalTokens() > maxTokens) buffer.shift();
39
+ },
40
+ getContext() {
41
+ return buffer.map((r) => `Q: ${r.query}\nA: ${r.response ?? ""}`).join("\n\n");
42
+ },
43
+ rounds: () => [...buffer],
44
+ clear() { buffer.length = 0; }
45
+ };
46
+ }
package/src/inject.js CHANGED
@@ -1,3 +1,5 @@
1
+ import { createHotMemory } from "./hot-memory.js";
2
+
1
3
  // Best-effort extraction of the current user's latest message text from the
2
4
  // live session, for semantic-first injection (Bug4). The system-prompt
3
5
  // interpolator renders synchronously, so this walks the already-materialized
@@ -25,6 +27,50 @@ function lastUserQuery(ctx) {
25
27
  return "";
26
28
  }
27
29
 
30
+ // Hot-memory round extraction (v0.5.0 1.3): pairs each user/message with the
31
+ // next assistant reply from the materialized session log. Tolerates shapes
32
+ // where assistant events carry a different type tag — anything whose payload
33
+ // has content parts and is not a user message counts as a reply. Best-effort:
34
+ // returns [] on any failure, and the hot block simply does not render.
35
+ function extractRounds(ctx, maxRounds) {
36
+ try {
37
+ const events = ctx?.agent?.session?.events;
38
+ if (!Array.isArray(events) || events.length === 0) return [];
39
+ const rounds = [];
40
+ let pendingQuery = null;
41
+ const textOf = (event) => {
42
+ const parts = event?.data?.content;
43
+ if (!Array.isArray(parts)) return "";
44
+ return parts
45
+ .map((p) => (typeof p === "string" ? p : p?.text ?? ""))
46
+ .filter(Boolean)
47
+ .join("\n")
48
+ .trim();
49
+ };
50
+ for (const event of events) {
51
+ const kind = event?.data?.source?.kind;
52
+ const isUser = event?.type === "user/message" && (kind === undefined || kind === "user");
53
+ if (isUser) {
54
+ if (pendingQuery) rounds.push({ query: pendingQuery, response: "" });
55
+ pendingQuery = textOf(event).slice(0, 500);
56
+ continue;
57
+ }
58
+ // Only assistant-originated events close a round; tool/system events
59
+ // carrying text must not be mistaken for the model's reply.
60
+ const isAssistant = typeof event?.type === "string" && event.type.includes("assistant")
61
+ || kind === "assistant";
62
+ const body = isAssistant ? textOf(event) : "";
63
+ if (!body || !pendingQuery) continue;
64
+ rounds.push({ query: pendingQuery, response: body.slice(0, 800) });
65
+ pendingQuery = null;
66
+ }
67
+ if (pendingQuery) rounds.push({ query: pendingQuery, response: "" });
68
+ return rounds.slice(-maxRounds);
69
+ } catch {
70
+ return [];
71
+ }
72
+ }
73
+
28
74
  export function createInjector(ctx, service, settings, config) {
29
75
  const maxItems = config.maxInjectedItems ?? 5;
30
76
  const threshold = config.importanceThreshold ?? 3;
@@ -36,6 +82,35 @@ export function createInjector(ctx, service, settings, config) {
36
82
  const MAX_CONTENT = 300;
37
83
  const MAX_BLOCK = 1500;
38
84
 
85
+ // Compressed injection (v0.5.0 2.1): a sleep-demoted row already carries its
86
+ // summary in `content` with the original parked in `_full_content` — inject
87
+ // the summary verbatim instead of re-truncating the (already short) text.
88
+ // Regular long rows keep the hard truncate.
89
+ function injectMemory(m, maxLength = MAX_CONTENT) {
90
+ if (m?._full_content) return String(m.content ?? "");
91
+ const text = String(m?.content ?? "");
92
+ return text.length <= maxLength ? text : `${text.slice(0, maxLength)}…`;
93
+ }
94
+
95
+ // Hot memory (v0.5.0 1.3): the latest rounds of THIS session, rebuilt from
96
+ // the materialized event log on every render — stateless, so it survives
97
+ // session switches and never persists anywhere.
98
+ const hot = createHotMemory({
99
+ maxRounds: config.hotMemoryRounds ?? 5,
100
+ maxTokens: config.hotMemoryMaxTokens ?? 2000
101
+ });
102
+
103
+ function renderHotContext(ctx) {
104
+ if (config.hotMemoryEnabled === false) return "";
105
+ const rounds = extractRounds(ctx, config.hotMemoryRounds ?? 5);
106
+ if (!rounds.length) return "";
107
+ hot.clear();
108
+ for (const r of rounds) hot.add(r);
109
+ const body = hot.getContext();
110
+ if (!body) return "";
111
+ return `[短期上下文] 最近对话(共 ${rounds.length} 轮):\n${body}`;
112
+ }
113
+
39
114
  function render(candidates) {
40
115
  if (!candidates.length) return "";
41
116
  const header = "[记忆库] 来自 dsh-mneme 的跨会话记忆(用户偏好与高优先级项目/决策):";
@@ -48,8 +123,7 @@ export function createInjector(ctx, service, settings, config) {
48
123
  ? "[verified] "
49
124
  : "";
50
125
  const title = `${m.title}(重要性 ${m.importance})`;
51
- let content = String(m.content ?? "");
52
- if (content.length > MAX_CONTENT) content = `${content.slice(0, MAX_CONTENT)}…`;
126
+ const content = injectMemory(m);
53
127
  const full = `- [${m.type}] ${verified}${title}:${content}`;
54
128
  if (budget - full.length >= 0) {
55
129
  lines.push(full);
@@ -108,7 +182,14 @@ export function createInjector(ctx, service, settings, config) {
108
182
  if (query) prefetchQueryVector(query);
109
183
  const queryVector = queryVectorCache.get(query);
110
184
  const candidates = service.injectCandidates({ query, queryVector, maxItems, threshold });
111
- return render(candidates);
185
+ // Hot memory (v0.5.0 1.3) leads the single memory block: the agent
186
+ // sees the short-term rounds first, then the cross-session recall —
187
+ // the documented injection order 1→2. Folding it here (instead of a
188
+ // separate context) keeps the prompt assembly stable at two blocks.
189
+ const hotText = renderHotContext(ctx);
190
+ const body = render(candidates);
191
+ if (!hotText) return body;
192
+ return body ? `${hotText}\n\n${body}` : hotText;
112
193
  }
113
194
  }),
114
195
  ctx.systemPrompt.context({
@@ -22,7 +22,13 @@ function modelHash(model) {
22
22
 
23
23
  /** Lazy default loader: dynamic import keeps module load cheap. */
24
24
  async function defaultPipelineLoader(task, model, options) {
25
- const { pipeline } = await import("@huggingface/transformers");
25
+ const { env, pipeline } = await import("@huggingface/transformers");
26
+ // issue #13: transformers.js's get_tokenizer_files() drops the caller's
27
+ // cache_dir when it pre-checks tokenizer_config.json metadata, so the HEAD
28
+ // request falls back to env.cacheDir and hits the network even when the
29
+ // model is fully cached locally. Mirroring the cache_dir onto env.cacheDir
30
+ // makes that pre-check resolve locally too — fully offline loading.
31
+ if (options?.cache_dir) env.cacheDir = options.cache_dir;
26
32
  return pipeline(task, model, options);
27
33
  }
28
34
 
@@ -0,0 +1,22 @@
1
+ // Adaptive vector threshold (v0.5.0 召回率优化 1.2): replaces the fixed
2
+ // vectorSearchThreshold=0.65 with a query-aware cutoff.
3
+ // entity:/attr: prefixes → 0.5 (entity recall is name-driven; loosen)
4
+ // very short queries → 0.7 (<5 chars match almost anything; tighten)
5
+ // very long queries → 0.6 (semantically specific; loosen a little)
6
+ // head-gap rule → when the top-1 vs top-5 candidate gap exceeds
7
+ // 0.3 the head is decisive — loosen to 0.5 so
8
+ // the tail still reaches the reranker
9
+ // otherwise → 0.65 (the legacy default)
10
+ // Pure and total: same inputs, same cutoff, no store access.
11
+ export function adaptiveThreshold(query, candidates = []) {
12
+ const q = String(query ?? "");
13
+ if (q.startsWith("entity:") || q.startsWith("attr:")) return 0.5;
14
+ if (q.length > 0 && q.length < 5) return 0.7;
15
+ if (q.length > 50) return 0.6;
16
+ const scores = (Array.isArray(candidates) ? candidates : [])
17
+ .map((c) => (typeof c?._score === "number" ? c._score : typeof c?.score === "number" ? c.score : 0))
18
+ .filter((s) => s > 0)
19
+ .sort((a, b) => b - a);
20
+ if (scores.length >= 5 && scores[0] - scores[4] > 0.3) return 0.5;
21
+ return 0.65;
22
+ }
@@ -0,0 +1,96 @@
1
+ // BM25 sparse retrieval (v0.5.0 召回率优化 1.1): the third recall path beside
2
+ // vector search and the LIKE keyword scan. The LIKE path only matches full
3
+ // substrings, so a multi-term query ("rust 异步 tokio") misses rows whose
4
+ // terms are scattered. BM25 scores per-token overlap with IDF weighting,
5
+ // which is exactly the gap: identifiers, code fragments and mixed CJK/ASCII
6
+ // queries recall rows the substring scan cannot see.
7
+
8
+ // Tokenizer: ASCII words keep their shape (identifiers like "dsh-mneme" or
9
+ // "ZFS_4421" survive as whole tokens); CJK runs become sliding bigrams
10
+ // (unigram only for single characters), the standard workaround for BM25's
11
+ // whitespace tokenization on Chinese.
12
+ export function tokenize(text) {
13
+ const raw = String(text ?? "").toLowerCase();
14
+ const tokens = [];
15
+ const ascii = raw.match(/[a-z0-9_]+/g) ?? [];
16
+ tokens.push(...ascii);
17
+ const cjkRuns = raw.match(/[\u4e00-\u9fff]+/g) ?? [];
18
+ for (const run of cjkRuns) {
19
+ if (run.length === 1) { tokens.push(run); continue; }
20
+ for (let i = 0; i < run.length - 1; i++) tokens.push(run.slice(i, i + 2));
21
+ }
22
+ return tokens;
23
+ }
24
+
25
+ const K1 = 1.5; // term-frequency saturation
26
+ const B = 0.75; // length normalization
27
+
28
+ /**
29
+ * Build a BM25 index over documents: [{id, title, content}].
30
+ * Returns { score, search }:
31
+ * score(query, doc) — per-spec ad-hoc scoring (re-tokenizes the doc)
32
+ * search(query, {limit}) — precomputed-tf ranking, scores normalized to
33
+ * [0,1] by the max so BM25 hits can weight-blend with vector/keyword
34
+ * scores on one scale. Rows the query does not touch at all are dropped.
35
+ */
36
+ export function createBM25Index(documents) {
37
+ const docs = Array.isArray(documents) ? documents.filter(Boolean) : [];
38
+ const N = docs.length;
39
+ const df = new Map();
40
+ const prepared = docs.map((doc) => {
41
+ const tokens = tokenize(`${doc.title ?? ""} ${doc.content ?? ""}`);
42
+ const tf = new Map();
43
+ for (const t of tokens) tf.set(t, (tf.get(t) ?? 0) + 1);
44
+ for (const t of tf.keys()) df.set(t, (df.get(t) ?? 0) + 1);
45
+ return { doc, tf, len: tokens.length };
46
+ });
47
+ const avgLen = N ? prepared.reduce((s, p) => s + p.len, 0) / N : 0 || 1;
48
+
49
+ const idf = (t) => {
50
+ const n = df.get(t) ?? 0;
51
+ return Math.log((N - n + 0.5) / (n + 0.5) + 1);
52
+ };
53
+
54
+ function scorePrepared(queryTokens, p) {
55
+ let score = 0;
56
+ for (const t of queryTokens) {
57
+ const f = p.tf.get(t);
58
+ if (!f) continue;
59
+ const norm = p.len ? K1 * (1 - B + B * (p.len / avgLen)) : K1;
60
+ score += idf(t) * ((f * (K1 + 1)) / (f + norm));
61
+ }
62
+ return score;
63
+ }
64
+
65
+ return {
66
+ score(query, doc) {
67
+ const tokens = tokenize(`${doc?.title ?? ""} ${doc?.content ?? ""}`);
68
+ const tf = new Map();
69
+ for (const t of tokens) tf.set(t, (tf.get(t) ?? 0) + 1);
70
+ const len = tokens.length;
71
+ // Ad-hoc scoring can't see corpus df; fall back to tf-only saturation
72
+ // (df is approximated as 1 so idf ≈ log(N - 0.5 + 1) is constant).
73
+ let score = 0;
74
+ for (const t of tokenize(query)) {
75
+ const f = tf.get(t);
76
+ if (!f) continue;
77
+ const norm = len ? K1 * (1 - B + B * (len / avgLen)) : K1;
78
+ score += idf(t) * ((f * (K1 + 1)) / (f + norm));
79
+ }
80
+ return score;
81
+ },
82
+ search(query, { limit = 20 } = {}) {
83
+ const qTokens = tokenize(query);
84
+ if (!qTokens.length || !N) return [];
85
+ const scored = [];
86
+ for (const p of prepared) {
87
+ const s = scorePrepared(qTokens, p);
88
+ if (s > 0) scored.push({ row: p.doc, raw: s });
89
+ }
90
+ scored.sort((a, b) => b.raw - a.raw);
91
+ const top = scored.slice(0, limit);
92
+ const max = top[0]?.raw || 1;
93
+ return top.map(({ row, raw }) => ({ ...row, score: max ? raw / max : 0 }));
94
+ }
95
+ };
96
+ }