sensemaking 0.23.0 → 0.24.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (193) hide show
  1. package/README.md +13 -9
  2. package/dist/cjs/cli/index.d.cts +1 -1
  3. package/dist/cjs/cli/index.d.ts +1 -1
  4. package/dist/cjs/cli/index.js +1 -1
  5. package/dist/cjs/cli/index.js.map +1 -1
  6. package/dist/cjs/cli/named.js +6 -2
  7. package/dist/cjs/cli/named.js.map +1 -1
  8. package/dist/cjs/cli/search.js +6 -2
  9. package/dist/cjs/cli/search.js.map +1 -1
  10. package/dist/cjs/cli/shared.d.cts +2 -0
  11. package/dist/cjs/cli/shared.d.ts +2 -0
  12. package/dist/cjs/cli/shared.js +24 -0
  13. package/dist/cjs/cli/shared.js.map +1 -1
  14. package/dist/cjs/commands/map.js +2 -2
  15. package/dist/cjs/commands/map.js.map +1 -1
  16. package/dist/cjs/commands/search.d.cts +10 -0
  17. package/dist/cjs/commands/search.d.ts +10 -0
  18. package/dist/cjs/commands/search.js +488 -180
  19. package/dist/cjs/commands/search.js.map +1 -1
  20. package/dist/cjs/output/output.js +45 -8
  21. package/dist/cjs/output/output.js.map +1 -1
  22. package/dist/cjs/output/search-error.d.cts +1 -0
  23. package/dist/cjs/output/search-error.d.ts +1 -0
  24. package/dist/cjs/output/search-error.js +28 -4
  25. package/dist/cjs/output/search-error.js.map +1 -1
  26. package/dist/cjs/scan/frontmatter.js +1 -1
  27. package/dist/cjs/scan/frontmatter.js.map +1 -1
  28. package/dist/cjs/scan/index.js +1 -0
  29. package/dist/cjs/scan/index.js.map +1 -1
  30. package/dist/cjs/store/batch.d.cts +4 -0
  31. package/dist/cjs/store/batch.d.ts +4 -0
  32. package/dist/cjs/store/batch.js +80 -0
  33. package/dist/cjs/store/batch.js.map +1 -0
  34. package/dist/cjs/store/duckdb/batch.d.cts +0 -4
  35. package/dist/cjs/store/duckdb/batch.d.ts +0 -4
  36. package/dist/cjs/store/duckdb/batch.js +5 -27
  37. package/dist/cjs/store/duckdb/batch.js.map +1 -1
  38. package/dist/cjs/store/duckdb/fieldStats.js +4 -7
  39. package/dist/cjs/store/duckdb/fieldStats.js.map +1 -1
  40. package/dist/cjs/store/duckdb/lexical.d.cts +2 -2
  41. package/dist/cjs/store/duckdb/lexical.d.ts +2 -2
  42. package/dist/cjs/store/duckdb/lexical.js +215 -64
  43. package/dist/cjs/store/duckdb/lexical.js.map +1 -1
  44. package/dist/cjs/store/duckdb/reconcile.js +7 -2
  45. package/dist/cjs/store/duckdb/reconcile.js.map +1 -1
  46. package/dist/cjs/store/duckdb/store.js +1 -3
  47. package/dist/cjs/store/duckdb/store.js.map +1 -1
  48. package/dist/cjs/store/duckdb/vectors.js +13 -8
  49. package/dist/cjs/store/duckdb/vectors.js.map +1 -1
  50. package/dist/cjs/store/native.js +5 -4
  51. package/dist/cjs/store/native.js.map +1 -1
  52. package/dist/cjs/store/open.js +7 -0
  53. package/dist/cjs/store/open.js.map +1 -1
  54. package/dist/cjs/store/shared.js.map +1 -1
  55. package/dist/cjs/store/sqlite/lexical.js +1 -4
  56. package/dist/cjs/store/sqlite/lexical.js.map +1 -1
  57. package/dist/cjs/store/sqlite/store.js +1 -4
  58. package/dist/cjs/store/sqlite/store.js.map +1 -1
  59. package/dist/cjs/store/sqlite/vectors.js +78 -24
  60. package/dist/cjs/store/sqlite/vectors.js.map +1 -1
  61. package/dist/cjs/store/turso/connection.d.cts +3 -0
  62. package/dist/cjs/store/turso/connection.d.ts +3 -0
  63. package/dist/cjs/store/turso/connection.js +302 -31
  64. package/dist/cjs/store/turso/connection.js.map +1 -1
  65. package/dist/cjs/store/turso/lexical-text.d.cts +1 -0
  66. package/dist/cjs/store/turso/lexical-text.d.ts +1 -0
  67. package/dist/cjs/store/turso/lexical-text.js +40 -0
  68. package/dist/cjs/store/turso/lexical-text.js.map +1 -0
  69. package/dist/cjs/store/turso/lexical.js +89 -16
  70. package/dist/cjs/store/turso/lexical.js.map +1 -1
  71. package/dist/cjs/store/turso/native.d.cts +4 -0
  72. package/dist/cjs/store/turso/native.d.ts +4 -0
  73. package/dist/cjs/store/turso/native.js +9 -0
  74. package/dist/cjs/store/turso/native.js.map +1 -1
  75. package/dist/cjs/store/turso/open.d.cts +1 -1
  76. package/dist/cjs/store/turso/open.d.ts +1 -1
  77. package/dist/cjs/store/turso/open.js +83 -10
  78. package/dist/cjs/store/turso/open.js.map +1 -1
  79. package/dist/cjs/store/turso/reconcile.d.cts +1 -1
  80. package/dist/cjs/store/turso/reconcile.d.ts +1 -1
  81. package/dist/cjs/store/turso/reconcile.js +6 -2
  82. package/dist/cjs/store/turso/reconcile.js.map +1 -1
  83. package/dist/cjs/store/turso/sql-functions.d.cts +1 -0
  84. package/dist/cjs/store/turso/sql-functions.d.ts +1 -0
  85. package/dist/cjs/store/turso/sql-functions.js +174 -0
  86. package/dist/cjs/store/turso/sql-functions.js.map +1 -0
  87. package/dist/cjs/store/turso/store.js +105 -4
  88. package/dist/cjs/store/turso/store.js.map +1 -1
  89. package/dist/cjs/store/turso/vectors.js +50 -14
  90. package/dist/cjs/store/turso/vectors.js.map +1 -1
  91. package/dist/cjs/store/types.d.cts +1 -2
  92. package/dist/cjs/store/types.d.ts +1 -2
  93. package/dist/cjs/store/types.js +1 -4
  94. package/dist/cjs/store/types.js.map +1 -1
  95. package/dist/cjs/store/vectors.d.cts +7 -0
  96. package/dist/cjs/store/vectors.d.ts +7 -0
  97. package/dist/cjs/store/vectors.js +8 -0
  98. package/dist/cjs/store/vectors.js.map +1 -1
  99. package/dist/cjs/text/segment.d.cts +13 -0
  100. package/dist/cjs/text/segment.d.ts +13 -0
  101. package/dist/cjs/text/segment.js +109 -4
  102. package/dist/cjs/text/segment.js.map +1 -1
  103. package/dist/esm/cli/index.d.ts +1 -1
  104. package/dist/esm/cli/index.js +1 -1
  105. package/dist/esm/cli/index.js.map +1 -1
  106. package/dist/esm/cli/named.js +6 -2
  107. package/dist/esm/cli/named.js.map +1 -1
  108. package/dist/esm/cli/search.js +5 -1
  109. package/dist/esm/cli/search.js.map +1 -1
  110. package/dist/esm/cli/shared.d.ts +2 -0
  111. package/dist/esm/cli/shared.js +19 -0
  112. package/dist/esm/cli/shared.js.map +1 -1
  113. package/dist/esm/commands/map.js +2 -2
  114. package/dist/esm/commands/map.js.map +1 -1
  115. package/dist/esm/commands/search.d.ts +10 -0
  116. package/dist/esm/commands/search.js +271 -91
  117. package/dist/esm/commands/search.js.map +1 -1
  118. package/dist/esm/output/output.js +27 -4
  119. package/dist/esm/output/output.js.map +1 -1
  120. package/dist/esm/output/search-error.d.ts +1 -0
  121. package/dist/esm/output/search-error.js +17 -1
  122. package/dist/esm/output/search-error.js.map +1 -1
  123. package/dist/esm/scan/frontmatter.js +1 -1
  124. package/dist/esm/scan/frontmatter.js.map +1 -1
  125. package/dist/esm/scan/index.js +1 -0
  126. package/dist/esm/scan/index.js.map +1 -1
  127. package/dist/esm/store/batch.d.ts +4 -0
  128. package/dist/esm/store/batch.js +21 -0
  129. package/dist/esm/store/batch.js.map +1 -0
  130. package/dist/esm/store/duckdb/batch.d.ts +0 -4
  131. package/dist/esm/store/duckdb/batch.js +2 -19
  132. package/dist/esm/store/duckdb/batch.js.map +1 -1
  133. package/dist/esm/store/duckdb/fieldStats.js +4 -7
  134. package/dist/esm/store/duckdb/fieldStats.js.map +1 -1
  135. package/dist/esm/store/duckdb/lexical.d.ts +2 -2
  136. package/dist/esm/store/duckdb/lexical.js +94 -40
  137. package/dist/esm/store/duckdb/lexical.js.map +1 -1
  138. package/dist/esm/store/duckdb/reconcile.js +2 -2
  139. package/dist/esm/store/duckdb/reconcile.js.map +1 -1
  140. package/dist/esm/store/duckdb/store.js +2 -4
  141. package/dist/esm/store/duckdb/store.js.map +1 -1
  142. package/dist/esm/store/duckdb/vectors.js +26 -13
  143. package/dist/esm/store/duckdb/vectors.js.map +1 -1
  144. package/dist/esm/store/native.js +5 -4
  145. package/dist/esm/store/native.js.map +1 -1
  146. package/dist/esm/store/open.js +7 -0
  147. package/dist/esm/store/open.js.map +1 -1
  148. package/dist/esm/store/shared.js +1 -2
  149. package/dist/esm/store/shared.js.map +1 -1
  150. package/dist/esm/store/sqlite/lexical.js +2 -5
  151. package/dist/esm/store/sqlite/lexical.js.map +1 -1
  152. package/dist/esm/store/sqlite/store.js +1 -4
  153. package/dist/esm/store/sqlite/store.js.map +1 -1
  154. package/dist/esm/store/sqlite/vectors.js +48 -19
  155. package/dist/esm/store/sqlite/vectors.js.map +1 -1
  156. package/dist/esm/store/turso/connection.d.ts +3 -0
  157. package/dist/esm/store/turso/connection.js +61 -5
  158. package/dist/esm/store/turso/connection.js.map +1 -1
  159. package/dist/esm/store/turso/lexical-text.d.ts +1 -0
  160. package/dist/esm/store/turso/lexical-text.js +12 -0
  161. package/dist/esm/store/turso/lexical-text.js.map +1 -0
  162. package/dist/esm/store/turso/lexical.js +56 -21
  163. package/dist/esm/store/turso/lexical.js.map +1 -1
  164. package/dist/esm/store/turso/native.d.ts +4 -0
  165. package/dist/esm/store/turso/native.js +8 -0
  166. package/dist/esm/store/turso/native.js.map +1 -1
  167. package/dist/esm/store/turso/open.d.ts +1 -1
  168. package/dist/esm/store/turso/open.js +7 -7
  169. package/dist/esm/store/turso/open.js.map +1 -1
  170. package/dist/esm/store/turso/reconcile.d.ts +1 -1
  171. package/dist/esm/store/turso/reconcile.js +6 -2
  172. package/dist/esm/store/turso/reconcile.js.map +1 -1
  173. package/dist/esm/store/turso/sql-functions.d.ts +1 -0
  174. package/dist/esm/store/turso/sql-functions.js +179 -0
  175. package/dist/esm/store/turso/sql-functions.js.map +1 -0
  176. package/dist/esm/store/turso/store.js +32 -7
  177. package/dist/esm/store/turso/store.js.map +1 -1
  178. package/dist/esm/store/turso/vectors.js +19 -10
  179. package/dist/esm/store/turso/vectors.js.map +1 -1
  180. package/dist/esm/store/types.d.ts +1 -2
  181. package/dist/esm/store/types.js +6 -8
  182. package/dist/esm/store/types.js.map +1 -1
  183. package/dist/esm/store/vectors.d.ts +7 -0
  184. package/dist/esm/store/vectors.js +8 -2
  185. package/dist/esm/store/vectors.js.map +1 -1
  186. package/dist/esm/text/segment.d.ts +13 -0
  187. package/dist/esm/text/segment.js +70 -0
  188. package/dist/esm/text/segment.js.map +1 -1
  189. package/package.json +4 -3
  190. package/schema.json +1 -1
  191. package/skills/sense/EXAMPLES.md +3 -3
  192. package/skills/sense/SKILL.md +6 -6
  193. package/skills/sense-setup/SKILL.md +1 -1
@@ -1,3 +1,4 @@
1
+ import { stemmer } from 'stemmer';
1
2
  // Word boundaries FTS5's tokenizers cannot find on their own. Contract: seg(query) must appear in
2
3
  // seg(document) for any substring query; word mode is context-dependent and breaks that, so grapheme mode (UAX #29) is used.
3
4
  // Scripts written without word spaces: a closed set of writing systems.
@@ -12,6 +13,70 @@ const HAS_RUN = new RegExp(`[${UNSPACED_SCRIPTS}]`, 'u');
12
13
  export function hasUnspacedRun(text) {
13
14
  return HAS_RUN.test(text);
14
15
  }
16
+ // Search tokens share the native separator contract: letters, marks, and numbers form tokens;
17
+ // underscore, apostrophe, and hyphen separate them. Unspaced-script runs stay whole so callers
18
+ // can apply substring semantics, while adjacent Latin text still gets its own token.
19
+ export function searchTokens(text) {
20
+ const tokens = [];
21
+ for (const match of text.matchAll(/[\p{L}\p{N}][\p{L}\p{M}\p{N}]*/gu)){
22
+ const start = match.index;
23
+ const end = start + match[0].length;
24
+ let cursor = start;
25
+ for (const run of text.slice(start, end).matchAll(RUN)){
26
+ const runStart = start + run.index;
27
+ if (runStart > cursor) tokens.push({
28
+ text: text.slice(cursor, runStart),
29
+ start: cursor,
30
+ end: runStart,
31
+ unspaced: false
32
+ });
33
+ const runEnd = runStart + run[0].length;
34
+ tokens.push({
35
+ text: run[0],
36
+ start: runStart,
37
+ end: runEnd,
38
+ unspaced: true
39
+ });
40
+ cursor = runEnd;
41
+ }
42
+ if (cursor < end) tokens.push({
43
+ text: text.slice(cursor, end),
44
+ start: cursor,
45
+ end,
46
+ unspaced: false
47
+ });
48
+ }
49
+ return tokens;
50
+ }
51
+ // Phrase membership is checked after native ranking. A CJK token matches inside one authored
52
+ // run; Latin tokens compare folded Porter stems. Each call scans one field once per candidate.
53
+ export function matchesSearchPhrase(text, phrase) {
54
+ const haystack = searchTokens(text);
55
+ if (phrase.length === 0 || haystack.length < phrase.length) return false;
56
+ for(let start = 0; start <= haystack.length - phrase.length; start++){
57
+ let matches = true;
58
+ for(let i = 0; i < phrase.length; i++){
59
+ const query = phrase[i];
60
+ const authored = haystack[start + i];
61
+ if (query.unspaced) {
62
+ if (!authored.unspaced || !authored.text.includes(query.text)) matches = false;
63
+ } else if (authored.unspaced || stemmer(foldForSearch(authored.text)) !== stemmer(foldForSearch(query.text))) {
64
+ matches = false;
65
+ }
66
+ if (!matches) break;
67
+ }
68
+ if (matches) return true;
69
+ }
70
+ return false;
71
+ }
72
+ // Character spans of unspaced-script runs in text, the same boundary the index segments on --
73
+ // the snippet marker uses this to keep substring semantics inside a run and word/stem matching outside it.
74
+ export function unspacedRuns(text) {
75
+ return Array.from(text.matchAll(RUN), (m)=>({
76
+ start: m.index,
77
+ end: m.index + m[0].length
78
+ }));
79
+ }
15
80
  // Grapheme clusters, ECMA-402/UAX #29: base char plus its marks, ZWJ sequences, Hangul jamo.
16
81
  // Built on first use and kept: construction is 6.6 ms, and a tree with no unspaced-script run
17
82
  // never segments at all, so no command pays it at module load.
@@ -87,3 +152,8 @@ export function segmentMatch(terms) {
87
152
  }
88
153
  return out;
89
154
  }
155
+ // FTS tokenizers compare letters without combining marks. Keep the authored string for output and
156
+ // offsets, but use this key when JS needs to compare a matched word to its source segment.
157
+ export function foldForSearch(text) {
158
+ return text.normalize('NFD').replace(/\p{M}/gu, '').toLowerCase();
159
+ }
@@ -1 +1 @@
1
- {"version":3,"sources":["/Users/kevin/Dev/OpenSource/ai/sensemaking/src/text/segment.ts"],"sourcesContent":["// Word boundaries FTS5's tokenizers cannot find on their own. Contract: seg(query) must appear in\n// seg(document) for any substring query; word mode is context-dependent and breaks that, so grapheme mode (UAX #29) is used.\n\n// Scripts written without word spaces: a closed set of writing systems.\n// Script_Extensions, not Script: the katakana long-vowel mark ー is Script=Common.\nexport const UNSPACED_SCRIPTS = '\\\\p{scx=Han}\\\\p{scx=Hiragana}\\\\p{scx=Katakana}\\\\p{scx=Thai}\\\\p{scx=Khmer}\\\\p{scx=Lao}\\\\p{scx=Myanmar}';\n// A run is script BASE characters with their combining marks attached; a bare mark after a\n// Latin letter (decomposed é) never starts one.\nconst RUN = new RegExp(`((?:[${UNSPACED_SCRIPTS}]\\\\p{M}*)+)`, 'gu');\nconst HAS_RUN = new RegExp(`[${UNSPACED_SCRIPTS}]`, 'u');\n\n// Whether text holds a script that marks no word boundaries, the predicate every store's\n// sidecar population and query split turns on.\nexport function hasUnspacedRun(text: string): boolean {\n return HAS_RUN.test(text);\n}\n// Grapheme clusters, ECMA-402/UAX #29: base char plus its marks, ZWJ sequences, Hangul jamo.\n// Built on first use and kept: construction is 6.6 ms, and a tree with no unspaced-script run\n// never segments at all, so no command pays it at module load.\nlet graphemeSegmenter: Intl.Segmenter | undefined;\n// unicode61 drops Unicode punctuation as a separator. Some of it (。、「」) has Script_Extensions\n// into an unspaced script, so RUN keeps it -- a grapheme matching this becomes a split point.\nconst PUNCTUATION = /\\p{P}/u;\n// Token barrier between separate runs (or a punctuation split within one) so their graphemes\n// are never phrase-adjacent. U+A7F7: a letter (so FTS5 keeps it as a token) no one types.\nconst BARRIER = 'ꟷ';\n\nfunction graphemes(run: string): string[] {\n graphemeSegmenter ??= new Intl.Segmenter(undefined, { granularity: 'grapheme' });\n return Array.from(graphemeSegmenter.segment(run), (s) => s.segment);\n}\n\n// A run's graphemes, cut into punctuation-free groups at every punctuation grapheme (dropped,\n// matching unicode61). The one place punctuation is classified, so index and query agree.\nfunction splitOnPunctuation(run: string): string[][] {\n const groups: string[][] = [[]];\n for (const g of graphemes(run)) {\n if (PUNCTUATION.test(g)) groups.push([]);\n else groups[groups.length - 1].push(g);\n }\n return groups.filter((g) => g.length > 0);\n}\n\n// Index side: '' when the field has no unspaced-script run. Otherwise each run explodes into\n// its graphemes, barrier-delimited from its neighbors and from a punctuation split within itself.\nexport function segmentField(text: string): string {\n if (!HAS_RUN.test(text)) return '';\n const out = text.replace(RUN, (run) => {\n const body = splitOnPunctuation(run)\n .map((g) => g.join(' '))\n .join(` ${BARRIER} `);\n return ` ${BARRIER} ${body} ${BARRIER} `;\n });\n return out.replace(/\\s+/g, ' ').trim();\n}\n\n// A `title:`/`summary:`/`text:` qualifier directly before a run that is about to become a\n// quoted grapheme phrase, so the rewrite can retarget it at the matching `_seg` column.\nconst QUALIFIER = /(^|[\\s(])(-?)(title|summary|text)\\s*:\\s*$/;\n\n// Raw title/summary/text drop punctuation as unicode61's token separator, so `数数` and `数。数`\n// falsely match there; only the barriered `_seg` columns are safe from it, hence this fallback target.\nconst SIDECAR_COLUMNS = '{title_seg summary_seg text_seg}:';\n\n// A run's punctuation-free groups, each its own quoted phrase (bare token if one grapheme),\n// space-joined -- the same split points segmentField barriers, so query and index agree.\nfunction runQuery(run: string, columnPrefix: string): string {\n return splitOnPunctuation(run)\n .map((g) => `${columnPrefix}${g.length > 1 ? `\"${g.join(' ')}\"` : g[0]}`)\n .join(' ');\n}\n\n// Query side: each unspaced run becomes phrases of its graphemes, matching segmentField. A\n// qualifier maps to its `_seg` column; unqualified maps to all three (SIDECAR_COLUMNS).\nexport function segmentMatch(terms: string): string {\n if (!HAS_RUN.test(terms)) return terms;\n let out = '';\n let quoted = false;\n const pieces = terms.split(RUN); // split keeps captured runs at odd indices\n for (let i = 0; i < pieces.length; i++) {\n if (i % 2 === 0) {\n for (const ch of pieces[i]) if (ch === '\"') quoted = !quoted;\n out += pieces[i];\n continue;\n }\n if (quoted) {\n out += pieces[i]; // an author's phrase is matched as written\n continue;\n }\n const m = out.match(QUALIFIER);\n if (m) {\n out = `${out.slice(0, m.index)}${m[1]}${runQuery(pieces[i], `${m[2]}${m[3]}_seg:`)}`;\n } else {\n out += runQuery(pieces[i], SIDECAR_COLUMNS);\n }\n }\n return out;\n}\n"],"names":["UNSPACED_SCRIPTS","RUN","RegExp","HAS_RUN","hasUnspacedRun","text","test","graphemeSegmenter","PUNCTUATION","BARRIER","graphemes","run","Intl","Segmenter","undefined","granularity","Array","from","segment","s","splitOnPunctuation","groups","g","push","length","filter","segmentField","out","replace","body","map","join","trim","QUALIFIER","SIDECAR_COLUMNS","runQuery","columnPrefix","segmentMatch","terms","quoted","pieces","split","i","ch","m","match","slice","index"],"mappings":"AAAA,kGAAkG;AAClG,6HAA6H;AAE7H,wEAAwE;AACxE,kFAAkF;AAClF,OAAO,MAAMA,mBAAmB,wGAAwG;AACxI,2FAA2F;AAC3F,gDAAgD;AAChD,MAAMC,MAAM,IAAIC,OAAO,CAAC,KAAK,EAAEF,iBAAiB,WAAW,CAAC,EAAE;AAC9D,MAAMG,UAAU,IAAID,OAAO,CAAC,CAAC,EAAEF,iBAAiB,CAAC,CAAC,EAAE;AAEpD,yFAAyF;AACzF,+CAA+C;AAC/C,OAAO,SAASI,eAAeC,IAAY;IACzC,OAAOF,QAAQG,IAAI,CAACD;AACtB;AACA,6FAA6F;AAC7F,8FAA8F;AAC9F,+DAA+D;AAC/D,IAAIE;AACJ,8FAA8F;AAC9F,8FAA8F;AAC9F,MAAMC,cAAc;AACpB,6FAA6F;AAC7F,0FAA0F;AAC1F,MAAMC,UAAU;AAEhB,SAASC,UAAUC,GAAW;IAC5BJ,8BAAAA,+BAAAA,oBAAAA,oBAAsB,IAAIK,KAAKC,SAAS,CAACC,WAAW;QAAEC,aAAa;IAAW;IAC9E,OAAOC,MAAMC,IAAI,CAACV,kBAAkBW,OAAO,CAACP,MAAM,CAACQ,IAAMA,EAAED,OAAO;AACpE;AAEA,8FAA8F;AAC9F,0FAA0F;AAC1F,SAASE,mBAAmBT,GAAW;IACrC,MAAMU,SAAqB;QAAC,EAAE;KAAC;IAC/B,KAAK,MAAMC,KAAKZ,UAAUC,KAAM;QAC9B,IAAIH,YAAYF,IAAI,CAACgB,IAAID,OAAOE,IAAI,CAAC,EAAE;aAClCF,MAAM,CAACA,OAAOG,MAAM,GAAG,EAAE,CAACD,IAAI,CAACD;IACtC;IACA,OAAOD,OAAOI,MAAM,CAAC,CAACH,IAAMA,EAAEE,MAAM,GAAG;AACzC;AAEA,6FAA6F;AAC7F,kGAAkG;AAClG,OAAO,SAASE,aAAarB,IAAY;IACvC,IAAI,CAACF,QAAQG,IAAI,CAACD,OAAO,OAAO;IAChC,MAAMsB,MAAMtB,KAAKuB,OAAO,CAAC3B,KAAK,CAACU;QAC7B,MAAMkB,OAAOT,mBAAmBT,KAC7BmB,GAAG,CAAC,CAACR,IAAMA,EAAES,IAAI,CAAC,MAClBA,IAAI,CAAC,CAAC,CAAC,EAAEtB,QAAQ,CAAC,CAAC;QACtB,OAAO,CAAC,CAAC,EAAEA,QAAQ,CAAC,EAAEoB,KAAK,CAAC,EAAEpB,QAAQ,CAAC,CAAC;IAC1C;IACA,OAAOkB,IAAIC,OAAO,CAAC,QAAQ,KAAKI,IAAI;AACtC;AAEA,0FAA0F;AAC1F,wFAAwF;AACxF,MAAMC,YAAY;AAElB,4FAA4F;AAC5F,uGAAuG;AACvG,MAAMC,kBAAkB;AAExB,4FAA4F;AAC5F,yFAAyF;AACzF,SAASC,SAASxB,GAAW,EAAEyB,YAAoB;IACjD,OAAOhB,mBAAmBT,KACvBmB,GAAG,CAAC,CAACR,IAAM,GAAGc,eAAed,EAAEE,MAAM,GAAG,IAAI,CAAC,CAAC,EAAEF,EAAES,IAAI,CAAC,KAAK,CAAC,CAAC,GAAGT,CAAC,CAAC,EAAE,EAAE,EACvES,IAAI,CAAC;AACV;AAEA,2FAA2F;AAC3F,wFAAwF;AACxF,OAAO,SAASM,aAAaC,KAAa;IACxC,IAAI,CAACnC,QAAQG,IAAI,CAACgC,QAAQ,OAAOA;IACjC,IAAIX,MAAM;IACV,IAAIY,SAAS;IACb,MAAMC,SAASF,MAAMG,KAAK,CAACxC,MAAM,2CAA2C;IAC5E,IAAK,IAAIyC,IAAI,GAAGA,IAAIF,OAAOhB,MAAM,EAAEkB,IAAK;QACtC,IAAIA,IAAI,MAAM,GAAG;YACf,KAAK,MAAMC,MAAMH,MAAM,CAACE,EAAE,CAAE,IAAIC,OAAO,KAAKJ,SAAS,CAACA;YACtDZ,OAAOa,MAAM,CAACE,EAAE;YAChB;QACF;QACA,IAAIH,QAAQ;YACVZ,OAAOa,MAAM,CAACE,EAAE,EAAE,2CAA2C;YAC7D;QACF;QACA,MAAME,IAAIjB,IAAIkB,KAAK,CAACZ;QACpB,IAAIW,GAAG;YACLjB,MAAM,GAAGA,IAAImB,KAAK,CAAC,GAAGF,EAAEG,KAAK,IAAIH,CAAC,CAAC,EAAE,GAAGT,SAASK,MAAM,CAACE,EAAE,EAAE,GAAGE,CAAC,CAAC,EAAE,GAAGA,CAAC,CAAC,EAAE,CAAC,KAAK,CAAC,GAAG;QACtF,OAAO;YACLjB,OAAOQ,SAASK,MAAM,CAACE,EAAE,EAAER;QAC7B;IACF;IACA,OAAOP;AACT"}
1
+ {"version":3,"sources":["/Users/kevin/Dev/OpenSource/ai/sensemaking/src/text/segment.ts"],"sourcesContent":["import { stemmer } from 'stemmer';\n\n// Word boundaries FTS5's tokenizers cannot find on their own. Contract: seg(query) must appear in\n// seg(document) for any substring query; word mode is context-dependent and breaks that, so grapheme mode (UAX #29) is used.\n\n// Scripts written without word spaces: a closed set of writing systems.\n// Script_Extensions, not Script: the katakana long-vowel mark ー is Script=Common.\nexport const UNSPACED_SCRIPTS = '\\\\p{scx=Han}\\\\p{scx=Hiragana}\\\\p{scx=Katakana}\\\\p{scx=Thai}\\\\p{scx=Khmer}\\\\p{scx=Lao}\\\\p{scx=Myanmar}';\n// A run is script BASE characters with their combining marks attached; a bare mark after a\n// Latin letter (decomposed é) never starts one.\nconst RUN = new RegExp(`((?:[${UNSPACED_SCRIPTS}]\\\\p{M}*)+)`, 'gu');\nconst HAS_RUN = new RegExp(`[${UNSPACED_SCRIPTS}]`, 'u');\n\n// Whether text holds a script that marks no word boundaries, the predicate every store's\n// sidecar population and query split turns on.\nexport function hasUnspacedRun(text: string): boolean {\n return HAS_RUN.test(text);\n}\n\nexport interface SearchToken {\n text: string;\n start: number;\n end: number;\n unspaced: boolean;\n}\n\n// Search tokens share the native separator contract: letters, marks, and numbers form tokens;\n// underscore, apostrophe, and hyphen separate them. Unspaced-script runs stay whole so callers\n// can apply substring semantics, while adjacent Latin text still gets its own token.\nexport function searchTokens(text: string): SearchToken[] {\n const tokens: SearchToken[] = [];\n for (const match of text.matchAll(/[\\p{L}\\p{N}][\\p{L}\\p{M}\\p{N}]*/gu)) {\n const start = match.index;\n const end = start + match[0].length;\n let cursor = start;\n for (const run of text.slice(start, end).matchAll(RUN)) {\n const runStart = start + run.index;\n if (runStart > cursor) tokens.push({ text: text.slice(cursor, runStart), start: cursor, end: runStart, unspaced: false });\n const runEnd = runStart + run[0].length;\n tokens.push({ text: run[0], start: runStart, end: runEnd, unspaced: true });\n cursor = runEnd;\n }\n if (cursor < end) tokens.push({ text: text.slice(cursor, end), start: cursor, end, unspaced: false });\n }\n return tokens;\n}\n\n// Phrase membership is checked after native ranking. A CJK token matches inside one authored\n// run; Latin tokens compare folded Porter stems. Each call scans one field once per candidate.\nexport function matchesSearchPhrase(text: string, phrase: SearchToken[]): boolean {\n const haystack = searchTokens(text);\n if (phrase.length === 0 || haystack.length < phrase.length) return false;\n for (let start = 0; start <= haystack.length - phrase.length; start++) {\n let matches = true;\n for (let i = 0; i < phrase.length; i++) {\n const query = phrase[i];\n const authored = haystack[start + i];\n if (query.unspaced) {\n if (!authored.unspaced || !authored.text.includes(query.text)) matches = false;\n } else if (authored.unspaced || stemmer(foldForSearch(authored.text)) !== stemmer(foldForSearch(query.text))) {\n matches = false;\n }\n if (!matches) break;\n }\n if (matches) return true;\n }\n return false;\n}\n\n// Character spans of unspaced-script runs in text, the same boundary the index segments on --\n// the snippet marker uses this to keep substring semantics inside a run and word/stem matching outside it.\nexport function unspacedRuns(text: string): Array<{ start: number; end: number }> {\n return Array.from(text.matchAll(RUN), (m) => ({ start: m.index, end: m.index + m[0].length }));\n}\n// Grapheme clusters, ECMA-402/UAX #29: base char plus its marks, ZWJ sequences, Hangul jamo.\n// Built on first use and kept: construction is 6.6 ms, and a tree with no unspaced-script run\n// never segments at all, so no command pays it at module load.\nlet graphemeSegmenter: Intl.Segmenter | undefined;\n// unicode61 drops Unicode punctuation as a separator. Some of it (。、「」) has Script_Extensions\n// into an unspaced script, so RUN keeps it -- a grapheme matching this becomes a split point.\nconst PUNCTUATION = /\\p{P}/u;\n// Token barrier between separate runs (or a punctuation split within one) so their graphemes\n// are never phrase-adjacent. U+A7F7: a letter (so FTS5 keeps it as a token) no one types.\nconst BARRIER = 'ꟷ';\n\nfunction graphemes(run: string): string[] {\n graphemeSegmenter ??= new Intl.Segmenter(undefined, { granularity: 'grapheme' });\n return Array.from(graphemeSegmenter.segment(run), (s) => s.segment);\n}\n\n// A run's graphemes, cut into punctuation-free groups at every punctuation grapheme (dropped,\n// matching unicode61). The one place punctuation is classified, so index and query agree.\nfunction splitOnPunctuation(run: string): string[][] {\n const groups: string[][] = [[]];\n for (const g of graphemes(run)) {\n if (PUNCTUATION.test(g)) groups.push([]);\n else groups[groups.length - 1].push(g);\n }\n return groups.filter((g) => g.length > 0);\n}\n\n// Index side: '' when the field has no unspaced-script run. Otherwise each run explodes into\n// its graphemes, barrier-delimited from its neighbors and from a punctuation split within itself.\nexport function segmentField(text: string): string {\n if (!HAS_RUN.test(text)) return '';\n const out = text.replace(RUN, (run) => {\n const body = splitOnPunctuation(run)\n .map((g) => g.join(' '))\n .join(` ${BARRIER} `);\n return ` ${BARRIER} ${body} ${BARRIER} `;\n });\n return out.replace(/\\s+/g, ' ').trim();\n}\n\n// A `title:`/`summary:`/`text:` qualifier directly before a run that is about to become a\n// quoted grapheme phrase, so the rewrite can retarget it at the matching `_seg` column.\nconst QUALIFIER = /(^|[\\s(])(-?)(title|summary|text)\\s*:\\s*$/;\n\n// Raw title/summary/text drop punctuation as unicode61's token separator, so `数数` and `数。数`\n// falsely match there; only the barriered `_seg` columns are safe from it, hence this fallback target.\nconst SIDECAR_COLUMNS = '{title_seg summary_seg text_seg}:';\n\n// A run's punctuation-free groups, each its own quoted phrase (bare token if one grapheme),\n// space-joined -- the same split points segmentField barriers, so query and index agree.\nfunction runQuery(run: string, columnPrefix: string): string {\n return splitOnPunctuation(run)\n .map((g) => `${columnPrefix}${g.length > 1 ? `\"${g.join(' ')}\"` : g[0]}`)\n .join(' ');\n}\n\n// Query side: each unspaced run becomes phrases of its graphemes, matching segmentField. A\n// qualifier maps to its `_seg` column; unqualified maps to all three (SIDECAR_COLUMNS).\nexport function segmentMatch(terms: string): string {\n if (!HAS_RUN.test(terms)) return terms;\n let out = '';\n let quoted = false;\n const pieces = terms.split(RUN); // split keeps captured runs at odd indices\n for (let i = 0; i < pieces.length; i++) {\n if (i % 2 === 0) {\n for (const ch of pieces[i]) if (ch === '\"') quoted = !quoted;\n out += pieces[i];\n continue;\n }\n if (quoted) {\n out += pieces[i]; // an author's phrase is matched as written\n continue;\n }\n const m = out.match(QUALIFIER);\n if (m) {\n out = `${out.slice(0, m.index)}${m[1]}${runQuery(pieces[i], `${m[2]}${m[3]}_seg:`)}`;\n } else {\n out += runQuery(pieces[i], SIDECAR_COLUMNS);\n }\n }\n return out;\n}\n\n// FTS tokenizers compare letters without combining marks. Keep the authored string for output and\n// offsets, but use this key when JS needs to compare a matched word to its source segment.\nexport function foldForSearch(text: string): string {\n return text.normalize('NFD').replace(/\\p{M}/gu, '').toLowerCase();\n}\n"],"names":["stemmer","UNSPACED_SCRIPTS","RUN","RegExp","HAS_RUN","hasUnspacedRun","text","test","searchTokens","tokens","match","matchAll","start","index","end","length","cursor","run","slice","runStart","push","unspaced","runEnd","matchesSearchPhrase","phrase","haystack","matches","i","query","authored","includes","foldForSearch","unspacedRuns","Array","from","m","graphemeSegmenter","PUNCTUATION","BARRIER","graphemes","Intl","Segmenter","undefined","granularity","segment","s","splitOnPunctuation","groups","g","filter","segmentField","out","replace","body","map","join","trim","QUALIFIER","SIDECAR_COLUMNS","runQuery","columnPrefix","segmentMatch","terms","quoted","pieces","split","ch","normalize","toLowerCase"],"mappings":"AAAA,SAASA,OAAO,QAAQ,UAAU;AAElC,kGAAkG;AAClG,6HAA6H;AAE7H,wEAAwE;AACxE,kFAAkF;AAClF,OAAO,MAAMC,mBAAmB,wGAAwG;AACxI,2FAA2F;AAC3F,gDAAgD;AAChD,MAAMC,MAAM,IAAIC,OAAO,CAAC,KAAK,EAAEF,iBAAiB,WAAW,CAAC,EAAE;AAC9D,MAAMG,UAAU,IAAID,OAAO,CAAC,CAAC,EAAEF,iBAAiB,CAAC,CAAC,EAAE;AAEpD,yFAAyF;AACzF,+CAA+C;AAC/C,OAAO,SAASI,eAAeC,IAAY;IACzC,OAAOF,QAAQG,IAAI,CAACD;AACtB;AASA,8FAA8F;AAC9F,+FAA+F;AAC/F,qFAAqF;AACrF,OAAO,SAASE,aAAaF,IAAY;IACvC,MAAMG,SAAwB,EAAE;IAChC,KAAK,MAAMC,SAASJ,KAAKK,QAAQ,CAAC,oCAAqC;QACrE,MAAMC,QAAQF,MAAMG,KAAK;QACzB,MAAMC,MAAMF,QAAQF,KAAK,CAAC,EAAE,CAACK,MAAM;QACnC,IAAIC,SAASJ;QACb,KAAK,MAAMK,OAAOX,KAAKY,KAAK,CAACN,OAAOE,KAAKH,QAAQ,CAACT,KAAM;YACtD,MAAMiB,WAAWP,QAAQK,IAAIJ,KAAK;YAClC,IAAIM,WAAWH,QAAQP,OAAOW,IAAI,CAAC;gBAAEd,MAAMA,KAAKY,KAAK,CAACF,QAAQG;gBAAWP,OAAOI;gBAAQF,KAAKK;gBAAUE,UAAU;YAAM;YACvH,MAAMC,SAASH,WAAWF,GAAG,CAAC,EAAE,CAACF,MAAM;YACvCN,OAAOW,IAAI,CAAC;gBAAEd,MAAMW,GAAG,CAAC,EAAE;gBAAEL,OAAOO;gBAAUL,KAAKQ;gBAAQD,UAAU;YAAK;YACzEL,SAASM;QACX;QACA,IAAIN,SAASF,KAAKL,OAAOW,IAAI,CAAC;YAAEd,MAAMA,KAAKY,KAAK,CAACF,QAAQF;YAAMF,OAAOI;YAAQF;YAAKO,UAAU;QAAM;IACrG;IACA,OAAOZ;AACT;AAEA,6FAA6F;AAC7F,+FAA+F;AAC/F,OAAO,SAASc,oBAAoBjB,IAAY,EAAEkB,MAAqB;IACrE,MAAMC,WAAWjB,aAAaF;IAC9B,IAAIkB,OAAOT,MAAM,KAAK,KAAKU,SAASV,MAAM,GAAGS,OAAOT,MAAM,EAAE,OAAO;IACnE,IAAK,IAAIH,QAAQ,GAAGA,SAASa,SAASV,MAAM,GAAGS,OAAOT,MAAM,EAAEH,QAAS;QACrE,IAAIc,UAAU;QACd,IAAK,IAAIC,IAAI,GAAGA,IAAIH,OAAOT,MAAM,EAAEY,IAAK;YACtC,MAAMC,QAAQJ,MAAM,CAACG,EAAE;YACvB,MAAME,WAAWJ,QAAQ,CAACb,QAAQe,EAAE;YACpC,IAAIC,MAAMP,QAAQ,EAAE;gBAClB,IAAI,CAACQ,SAASR,QAAQ,IAAI,CAACQ,SAASvB,IAAI,CAACwB,QAAQ,CAACF,MAAMtB,IAAI,GAAGoB,UAAU;YAC3E,OAAO,IAAIG,SAASR,QAAQ,IAAIrB,QAAQ+B,cAAcF,SAASvB,IAAI,OAAON,QAAQ+B,cAAcH,MAAMtB,IAAI,IAAI;gBAC5GoB,UAAU;YACZ;YACA,IAAI,CAACA,SAAS;QAChB;QACA,IAAIA,SAAS,OAAO;IACtB;IACA,OAAO;AACT;AAEA,8FAA8F;AAC9F,2GAA2G;AAC3G,OAAO,SAASM,aAAa1B,IAAY;IACvC,OAAO2B,MAAMC,IAAI,CAAC5B,KAAKK,QAAQ,CAACT,MAAM,CAACiC,IAAO,CAAA;YAAEvB,OAAOuB,EAAEtB,KAAK;YAAEC,KAAKqB,EAAEtB,KAAK,GAAGsB,CAAC,CAAC,EAAE,CAACpB,MAAM;QAAC,CAAA;AAC7F;AACA,6FAA6F;AAC7F,8FAA8F;AAC9F,+DAA+D;AAC/D,IAAIqB;AACJ,8FAA8F;AAC9F,8FAA8F;AAC9F,MAAMC,cAAc;AACpB,6FAA6F;AAC7F,0FAA0F;AAC1F,MAAMC,UAAU;AAEhB,SAASC,UAAUtB,GAAW;IAC5BmB,8BAAAA,+BAAAA,oBAAAA,oBAAsB,IAAII,KAAKC,SAAS,CAACC,WAAW;QAAEC,aAAa;IAAW;IAC9E,OAAOV,MAAMC,IAAI,CAACE,kBAAkBQ,OAAO,CAAC3B,MAAM,CAAC4B,IAAMA,EAAED,OAAO;AACpE;AAEA,8FAA8F;AAC9F,0FAA0F;AAC1F,SAASE,mBAAmB7B,GAAW;IACrC,MAAM8B,SAAqB;QAAC,EAAE;KAAC;IAC/B,KAAK,MAAMC,KAAKT,UAAUtB,KAAM;QAC9B,IAAIoB,YAAY9B,IAAI,CAACyC,IAAID,OAAO3B,IAAI,CAAC,EAAE;aAClC2B,MAAM,CAACA,OAAOhC,MAAM,GAAG,EAAE,CAACK,IAAI,CAAC4B;IACtC;IACA,OAAOD,OAAOE,MAAM,CAAC,CAACD,IAAMA,EAAEjC,MAAM,GAAG;AACzC;AAEA,6FAA6F;AAC7F,kGAAkG;AAClG,OAAO,SAASmC,aAAa5C,IAAY;IACvC,IAAI,CAACF,QAAQG,IAAI,CAACD,OAAO,OAAO;IAChC,MAAM6C,MAAM7C,KAAK8C,OAAO,CAAClD,KAAK,CAACe;QAC7B,MAAMoC,OAAOP,mBAAmB7B,KAC7BqC,GAAG,CAAC,CAACN,IAAMA,EAAEO,IAAI,CAAC,MAClBA,IAAI,CAAC,CAAC,CAAC,EAAEjB,QAAQ,CAAC,CAAC;QACtB,OAAO,CAAC,CAAC,EAAEA,QAAQ,CAAC,EAAEe,KAAK,CAAC,EAAEf,QAAQ,CAAC,CAAC;IAC1C;IACA,OAAOa,IAAIC,OAAO,CAAC,QAAQ,KAAKI,IAAI;AACtC;AAEA,0FAA0F;AAC1F,wFAAwF;AACxF,MAAMC,YAAY;AAElB,4FAA4F;AAC5F,uGAAuG;AACvG,MAAMC,kBAAkB;AAExB,4FAA4F;AAC5F,yFAAyF;AACzF,SAASC,SAAS1C,GAAW,EAAE2C,YAAoB;IACjD,OAAOd,mBAAmB7B,KACvBqC,GAAG,CAAC,CAACN,IAAM,GAAGY,eAAeZ,EAAEjC,MAAM,GAAG,IAAI,CAAC,CAAC,EAAEiC,EAAEO,IAAI,CAAC,KAAK,CAAC,CAAC,GAAGP,CAAC,CAAC,EAAE,EAAE,EACvEO,IAAI,CAAC;AACV;AAEA,2FAA2F;AAC3F,wFAAwF;AACxF,OAAO,SAASM,aAAaC,KAAa;IACxC,IAAI,CAAC1D,QAAQG,IAAI,CAACuD,QAAQ,OAAOA;IACjC,IAAIX,MAAM;IACV,IAAIY,SAAS;IACb,MAAMC,SAASF,MAAMG,KAAK,CAAC/D,MAAM,2CAA2C;IAC5E,IAAK,IAAIyB,IAAI,GAAGA,IAAIqC,OAAOjD,MAAM,EAAEY,IAAK;QACtC,IAAIA,IAAI,MAAM,GAAG;YACf,KAAK,MAAMuC,MAAMF,MAAM,CAACrC,EAAE,CAAE,IAAIuC,OAAO,KAAKH,SAAS,CAACA;YACtDZ,OAAOa,MAAM,CAACrC,EAAE;YAChB;QACF;QACA,IAAIoC,QAAQ;YACVZ,OAAOa,MAAM,CAACrC,EAAE,EAAE,2CAA2C;YAC7D;QACF;QACA,MAAMQ,IAAIgB,IAAIzC,KAAK,CAAC+C;QACpB,IAAItB,GAAG;YACLgB,MAAM,GAAGA,IAAIjC,KAAK,CAAC,GAAGiB,EAAEtB,KAAK,IAAIsB,CAAC,CAAC,EAAE,GAAGwB,SAASK,MAAM,CAACrC,EAAE,EAAE,GAAGQ,CAAC,CAAC,EAAE,GAAGA,CAAC,CAAC,EAAE,CAAC,KAAK,CAAC,GAAG;QACtF,OAAO;YACLgB,OAAOQ,SAASK,MAAM,CAACrC,EAAE,EAAE+B;QAC7B;IACF;IACA,OAAOP;AACT;AAEA,kGAAkG;AAClG,2FAA2F;AAC3F,OAAO,SAASpB,cAAczB,IAAY;IACxC,OAAOA,KAAK6D,SAAS,CAAC,OAAOf,OAAO,CAAC,WAAW,IAAIgB,WAAW;AACjE"}
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "sensemaking",
3
- "version": "0.23.0",
4
- "description": "Query and search your markdown notes with context-aware progressive disclosure: SQL over frontmatter, links, and text, plus semantic search and link-graph ranking. No server, no build step",
3
+ "version": "0.24.0",
4
+ "description": "Query and search your markdown notes with context-aware progressive disclosure: SQL over frontmatter, links, and text, plus semantic search and link-graph ranking. No server, no build step.",
5
5
  "keywords": [
6
6
  "markdown",
7
7
  "frontmatter",
@@ -81,12 +81,13 @@
81
81
  "version": "tsds version"
82
82
  },
83
83
  "dependencies": {
84
- "@huggingface/tokenizers": "^0.1.3",
84
+ "@huggingface/tokenizers": "^0.2.0",
85
85
  "franc-min": "^6.2.0",
86
86
  "install-module-linked": "^2.0.0",
87
87
  "markdown-it": "^15.0.1",
88
88
  "markdown-it-footnote": "^4.0.0",
89
89
  "markdown-it-task-lists": "^2.1.1",
90
+ "stemmer": "^2.0.1",
90
91
  "tinypool": "^2.1.2",
91
92
  "yaml": "^2.9.0"
92
93
  },
package/schema.json CHANGED
@@ -104,7 +104,7 @@
104
104
  "store": {
105
105
  "type": "string",
106
106
  "enum": ["sqlite", "duckdb", "turso"],
107
- "description": "Backing store engine, defaulting to \"sqlite\" (zero-dependency, Node's built-in SQLite). \"duckdb\" and \"turso\" are experimental: the first command that opens such a tree installs its engine package on its own (@duckdb/node-api, a one-time native download of about 110 MB; @tursodatabase/database, much smaller). The same commands and table names run on all three; what does not port is FTS5 syntax. Under duckdb and turso, `search` text and raw MATCH reject FTS5's prefix (foo*), boolean (AND/OR/NOT), NEAR, initial-token (^), and column-filter (title:foo) operators with a named error (STORE_CAPABILITY_MISSING) that says how to rephrase or set \"store\" to \"sqlite\"; bare words and quoted phrases work on all three. Raw SQL is a per-store dialect: the tables are portable everywhere, but sqlite's FTS5 MATCH, snippet(), and bm25() do not run under duckdb (its fts extension has its own functions) or turso (Tantivy fts_match/fts_score), so saved queries written in FTS5 syntax are sqlite dialect. The has/basename/segment functions are registered on sqlite and duckdb but not turso, whose client cannot register SQL functions at all. `sense watch` requires sqlite; on duckdb and turso it errors at start naming the fix. Each store keeps its own cache file (.sense/cache.db, .sense/cache.duckdb, .sense/cache.turso.db); switching stores rebuilds the index rather than migrating it."
107
+ "description": "Backing store engine, defaulting to \"sqlite\" (zero-dependency, Node's built-in SQLite). \"duckdb\" and \"turso\" are experimental: the first command that opens such a tree installs its engine package on its own (@duckdb/node-api, a one-time native download of about 110 MB; @tursodatabase/database, much smaller). The same commands and table names run on all three; what does not port is FTS5 syntax. Under duckdb and turso, `search` text and raw MATCH reject FTS5's prefix (foo*), boolean (AND/OR/NOT), NEAR, initial-token (^), and column-filter (title:foo) operators with a named error (STORE_CAPABILITY_MISSING) that says how to rephrase or set \"store\" to \"sqlite\"; bare words and quoted phrases work on all three. Raw SQL is a per-store dialect: the tables are portable everywhere, but sqlite's FTS5 MATCH, snippet(), and bm25() do not run under duckdb (its fts extension has its own functions) or turso (Tantivy fts_match/fts_score), so saved queries written in FTS5 syntax are sqlite dialect. The has/basename/segment functions: has and basename run on all three (turso rewrites them into portable SQL rather than registering them); segment runs on sqlite and duckdb only, so a query calling segment under turso fails with a named error (STORE_CAPABILITY_MISSING) saying to rephrase or set \"store\" to sqlite. `sense watch` runs on all three; under duckdb and turso, which lock the cache file per connection, a concurrent command waits out the watcher's current cycle instead of failing. Each store keeps its own cache file (.sense/cache.db, .sense/cache.duckdb, .sense/cache.turso.db); switching stores rebuilds the index rather than migrating it."
108
108
  },
109
109
  "queries": {
110
110
  "type": "object",
@@ -11,9 +11,9 @@ sense search "pricing OR billing OR invoicing" --k 10 --format json
11
11
  ```json
12
12
  [
13
13
  { "path": "notes/pricing-model.md", "title": "Pricing model",
14
- "summary": "Tiered per-seat pricing; floor and discount rules", "hit": "«Pricing» floor for annual…", "via": "match+link", "score": 0.0333, "lines": null },
14
+ "summary": "Tiered per-seat pricing; floor and discount rules", "snippets": ["«Pricing» floor for annual…"], "via": "match+link", "score": 0.0333, "lines": null },
15
15
  { "path": "notes/renewal-playbook.md", "title": "Renewal playbook",
16
- "summary": "Renewal sequence and owners", "hit": "…«billing» contact confirms the PO…", "via": "match", "score": 0.0313, "lines": "L81-140" }
16
+ "summary": "Renewal sequence and owners", "snippets": ["…«billing» contact confirms the PO…"], "via": "match", "score": 0.0313, "lines": "L81-140" }
17
17
  ]
18
18
  ```
19
19
 
@@ -79,7 +79,7 @@ sense search "children dying from poor nutrition" --k 3 --format json
79
79
  ```json
80
80
  [
81
81
  { "path": "notes/malnutrition-outcomes.md", "title": "Malnutrition outcomes",
82
- "summary": "Stunting and mortality by region", "hit": null, "via": "vector", "score": 0.0167, "similarity": 0.61, "lines": "L14-52" }
82
+ "summary": "Stunting and mortality by region", "snippets": [], "via": "vector", "score": 0.0167, "similarity": 0.61, "lines": "L14-52" }
83
83
  ]
84
84
  ```
85
85
 
@@ -7,7 +7,7 @@ description: "Query a markdown tree with the sense CLI: filter notes by frontmat
7
7
 
8
8
  SQL over a markdown tree, kept fresh by a filesystem check on every query. Every file becomes rows in `frontmatter` (one column per key, plus `path`/`_mtime`/`_ctime`/`_size`/`_rank`/`_parse_error`; `_ctime` is filesystem birthtime, which a clone or copy resets just like `_mtime`), `content` (`title`, `summary`, `text`, `path`; an FTS5 index on the default store, with machine-written `title_seg`/`summary_seg`/`text_seg` sidecars used for matching Chinese, Japanese, Thai, Khmer, Lao, and Burmese text, not for reading), `links` (`src`, `target`, `dst`, `embed`; `NULL` dst = dead link; `embed` 1 for `![[...]]` embeds, 0 for links; one row per distinct written target and kind with alias and anchor stripped, so `[[Foo]]` and `[[Foo|alias]]` are one row while `[[Foo]]` and `[[notes/Foo]]` are two rows that can share a `dst`, and a target both linked and embedded is a row of each kind; extraction matches Obsidian's own graph: comments, code, and link-syntax text yield no rows, `[[#Anchor]]` is a self-edge, a frontmatter value that is exactly `[[X]]` is a link, and a basename collision resolves to the linking note itself, else the shortest path), `tags` (`path`, `tag`: frontmatter and inline `#tags` merged and deduplicated; nested tags stored full, so `book/scifi` matches `tag = 'book' OR tag LIKE 'book/%'`), `sections` (heading outline with line ranges and token estimates), and `preset_files` (`path`, `preset`: which presets cover which files). Features add their own storage; `map` and `status` report which are on.
9
9
 
10
- **Stores.** The config's `store` key picks the backing store: `sqlite` (default, zero-dependency, Node's built-in SQLite), or the experimental `duckdb` and `turso` (the first command that opens such a tree installs that engine's package on its own: `@duckdb/node-api`, a one-time native download of about 110 MB, or the much smaller `@tursodatabase/database`). The tables, `?` placeholders, quoted identifiers, and the `scope` binding are the same on all three, so ordinary frontmatter SQL ports as written. Two things do not port. First, FTS5: `content` is an FTS5 table on sqlite and a plain table on the other two, so hand-written `MATCH`, `snippet()`, `bm25()`, and sqlite's date-function forms run only on sqlite (duckdb has its own fts functions and date syntax; turso has Tantivy's `fts_match`/`fts_score`), and `search` text under duckdb and turso rejects FTS5's prefix (`foo*`), boolean (`AND`/`OR`/`NOT`), `NEAR`, initial-token (`^`), and column-filter (`title:foo`) operators with a named error (STORE_CAPABILITY_MISSING) that says how to rephrase or set `store` to `sqlite`; bare words and quoted phrases work on all three. Second, the `has`/`basename`/`segment` functions are registered on sqlite and duckdb but not turso, whose client cannot register SQL functions at all, so a query calling them under turso fails with `no such function`. `sense watch` runs on all three; under duckdb and turso, which lock the cache file per connection, a concurrent command waits out the watcher's current cycle instead of failing.
10
+ **Stores.** The config's `store` key picks the backing store: `sqlite` (default, zero-dependency, Node's built-in SQLite), or the experimental `duckdb` and `turso` (the first command that opens such a tree installs that engine's package on its own: `@duckdb/node-api`, a one-time native download of about 110 MB, or the much smaller `@tursodatabase/database`). The tables, `?` placeholders, quoted identifiers, and the `scope` binding are the same on all three, so ordinary frontmatter SQL ports as written. Two things do not port. First, FTS5: `content` is an FTS5 table on sqlite and a plain table on the other two, so hand-written `MATCH`, `snippet()`, `bm25()`, and sqlite's date-function forms run only on sqlite (duckdb has its own fts functions and date syntax; turso has Tantivy's `fts_match`/`fts_score`), and `search` text under duckdb and turso rejects FTS5's prefix (`foo*`), boolean (`AND`/`OR`/`NOT`), `NEAR`, initial-token (`^`), and column-filter (`title:foo`) operators with a named error (STORE_CAPABILITY_MISSING) that says how to rephrase or set `store` to `sqlite`; bare words and quoted phrases work on all three. Second, of the `has`/`basename`/`segment` functions, `has` and `basename` run on all three (turso rewrites them into portable SQL rather than registering them), while `segment` runs on sqlite and duckdb only, so a query calling `segment` under turso fails with a named error (STORE_CAPABILITY_MISSING) saying to rephrase or set `store` to sqlite. `sense watch` runs on all three; under duckdb and turso, which lock the cache file per connection, a concurrent command waits out the watcher's current cycle instead of failing.
11
11
 
12
12
  ## What each tool is for
13
13
 
@@ -31,7 +31,7 @@ Output defaults to a table, built for humans; `--format json` returns the same r
31
31
  |---|---|---|---|
32
32
  | `words` | the text itself (BM25) | every literal occurrence: identifiers, error strings, names, exact phrases; deterministic | anything phrased differently, e.g. a note saying "compensation floor" for the query "minimum pay" |
33
33
  | `links` | connections authors wrote | what the tree's own structure treats as related: context the matching text never restates | notes nobody linked |
34
- | `vectors` | per-chunk meaning-vectors | notes sharing no words with the query: paraphrase, the category for an instance, a concept restated | proving presence. A vector row cannot show the words occur anywhere; `hit` can |
34
+ | `vectors` | per-chunk meaning-vectors | notes sharing no words with the query: paraphrase, the category for an instance, a concept restated | proving presence. A vector row cannot show the words occur anywhere; `snippets` can |
35
35
 
36
36
  Each signal in the map carries a weight (`{"words": 1, "links": 1, "vectors": 1}` is the default when the key is omitted entirely); presence turns a signal on, and the number scales its share of the fused ranking. Equal weight (1 for everything) is what every number elsewhere in this doc describes. A weight above 1 pulls the fused ranking toward that signal's own ordering: measured on nfcorpus with the default static model and on MIRACL zh with an HTTP encoder, equal-weight fusion helps English nfcorpus but ranks the encoder's MIRACL-zh results below its own cosine-only ranking (benchmark/reports/2026-08-27-embedding-model-selection.md, weight-sweep table). There is no single weight that is right for both, so a preset that leans on a strong encoder is a candidate for a higher `vectors` weight, checked against that table rather than assumed.
37
37
 
@@ -54,13 +54,13 @@ sense --list | status | download
54
54
  ```
55
55
 
56
56
  - Terms pass verbatim to FTS5 MATCH. Bare words AND-join (one absent word means zero rows), so write `OR` yourself when you want any-word matching; double-quote punctuated terms (`"customer-facing"`, `"founder's"`); invalid syntax is an error, not a rewrite. The same rules apply to search commands you write into subagent briefs.
57
- - When a search misses, both sides have levers. Lexical: OR-in synonyms and concrete instances (the index only knows the words in the files; a note about a specific tool rarely names its category), raise `--k` (a row costs tens of tokens), widen the scope (`--preset`, or `--include` for an ad-hoc glob). Vector: restate the concept in different words. Vectors rank the meaning of the whole query text, so a rephrase moves them even when every literal word still misses. Then pivot through the nearest hit with `related <path>` to walk its meaning-neighbourhood. Vector-only rows in a miss are themselves evidence: `similarity` and the snippet say whether the concept exists in the tree under other words. Each widening adds candidates and dilutes ranking, so the noise trade-off runs both ways.
57
+ - When a search misses, both sides have levers. Lexical: OR-in synonyms and concrete instances (the index only knows the words in the files; a note about a specific tool rarely names its category), raise `--k` (a row costs tens of tokens), widen the scope (`--preset`, or `--include` for an ad-hoc glob). Vector: restate the concept in different words. Vectors rank the meaning of the whole query text, so a rephrase moves them even when every literal word still misses. Then pivot through the nearest hit with `related <path>` to walk its meaning-neighbourhood. Vector-only rows in a miss are leads rather than lexical evidence: use `similarity` and the `lines` range to decide what to read; their `snippets` list is empty because the query words did not match. Each widening adds candidates and dilutes ranking, so the noise trade-off runs both ways.
58
58
  - A frontmatter query enumerates its matches deterministically; search ranks by term overlap, so results shift as phrasing shifts. Trade-off: a query needs a known field, search doesn't.
59
- - The `via` column says what produced each row: `match` (words hit), `link` (connected to notes that hit), `vector` (near in meaning), and combinations. The `lines` column, when set, points at the section that earned the row (the best-matching chunk on vector rows, the term cluster's section on large lexical notes) and is a direct `Read` range; null means the whole note is the reference.
59
+ - The `via` column says what produced each row: `match` (words hit), `link` (connected to notes that hit), `vector` (near in meaning), and combinations. The `lines` column, when set, points at the section that earned the row (the best-matching chunk on vector rows, the term cluster's section on large lexical notes) and is a direct `Read` range; null means the whole note is the reference. A row's `snippets` holds the marked passages that earned the lexical match (`[]` on a row that never contained the terms); `--snippet-char-limit` (default 80, the passage's rendered length, marks and ellipses counted) and `--snippet-count-limit` (default 1, passages per note) bound them, on `search` and a saved search alike.
60
60
  - Scope is one vocabulary shared by `search`, `map`, `peek`, `path`, and `related`: bare command uses the config's `default` preset; `--preset <name>` picks another (unknown names error, listing what's declared); `--include <glob>` and `--exclude <glob>` are ad-hoc globs for one command, each overriding its own side of the preset, so one does not clear the other; `--no-exclude` drops the preset's `exclude` for one command, the only way to widen past it without editing config (it widens the query scope, not the index: a file no preset covers is never indexed). `--where` takes any SQL condition against frontmatter alias `f`, not only field equality: `"f.status = 'active' AND has(f.tags, 'x')"`, `"datetime(f.created) >= datetime(?)"`. There is no whole-index flag: a broad `default` preset, or a declared `all` preset (`include ["**/*"]`), is the whole tree. `sense status` shows every preset with its coverage. `sql` scopes differently: it runs over the whole index by default, and `--preset <name>` *binds* the scope as a temporary `scope(path)` table your statement joins, rather than filtering behind the query's back (`JOIN scope ON scope."path" = f.path`). Naming a preset without joining `scope` is a usage error, since it would return everything while reading as scoped. Without the flag, join `preset_files` directly, which is the same coverage under a preset name you write into the SQL.
61
61
  - `score` is a rank-fusion value: it ranks rows within one result set and is not comparable across queries, not a relevance magnitude. It encodes how many signals fired and at what rank, so a perfect lexical hit and a weak vector-only hit can read the same number. With vectors active, rows carry `similarity`: the cosine (-1 to 1) of the query against that file's best-matching chunk (the same chunk the `lines` range points at). It orders vector evidence within a result set; the range it spans depends on the corpus and the embedding model, and compresses on small trees, where even a nonsense query has a moderately near neighbour somewhere. Compare similarities within a result set rather than against a fixed cutoff carried between trees.
62
62
  - Vector rankings read the whole chunk, boilerplate included, so on a tree whose notes are mostly one template with a line of unique text (a directory of plugins, people, or assets), the shared scaffolding dominates every vector and `related` returns near-ties at the top of the range for any seed. Two signs, both visible in the output already: the top `similarity` values sit within a hair of each other, and the same few notes come back for unrelated seeds. Compare neighbour lists across two unlike seeds when a tree looks like this; matching lists mean the ranking is reading the template, not the content, and `search` over the distinguishing words is the answer instead.
63
- - Absence evidence lives in the labels: a preset whose `signals` exclude `vectors` (or a tree with no `embed` block) returns 0 rows when the words are nowhere in it. Default `search` always returns up to `k` rows (nearest-neighbour search has a nearest neighbour for any input), so a result of only `via: vector` rows IS the absence signal for the words themselves; `similarity` and the snippet are the evidence for judging whether a vector row is a real conceptual hit.
63
+ - Absence evidence lives in the labels: a preset whose `signals` exclude `vectors` (or a tree with no `embed` block) returns 0 rows when the words are nowhere in it. Default `search` always returns up to `k` rows (nearest-neighbour search has a nearest neighbour for any input), so a result of only `via: vector` rows is the absence signal for the words themselves. Judge whether a vector row is a useful conceptual lead from `similarity`, then read its `lines` range; it has no lexical snippet.
64
64
  - A `queries` entry names the verb it runs, mirroring the two commands: `"dead-links": { "sql": "SELECT src, target FROM links WHERE dst IS NULL AND lower(target) NOT GLOB '*.[a-z0-9]*'" }` runs as `sense dead-links`, and `"hot": { "search": "pricing OR billing", "preset": "raw", "k": 20 }` runs as `sense hot` with its settings baked in, so repeat runs need no flags. An invocation-level `--preset`, `--k`, or `--where` overrides a saved search's value; `--list` labels each entry `(sql)` or `(search)`.
65
65
  - Running an entry is how it is validated: a typo'd column, stale SQL, bad FTS5 syntax or an unknown preset errors and exits nonzero. A parameterised entry validates with any argument, since SQL is prepared before parameters bind (`sense by-tag zzz` reports `no such column` if the column is wrong, `(0 rows)` if it is right). To sweep a whole config after editing it, read the exit code: 0 ran, 2 means it needs parameters (re-run it with any argument to validate the SQL), anything else is broken.
66
66
 
@@ -112,7 +112,7 @@ WHERE a.dst = ? AND b.dst IS NOT NULL AND b.dst <> a.dst;
112
112
  - `content MATCH` only works against the fts5 table by its own name, never through an alias or a view: `FROM content c ... WHERE c MATCH 'x'` fails with `no such column: c`. This is why `--preset` binds a table to join rather than shadowing the tables.
113
113
  - `content MATCH` takes FTS5 syntax: `a OR b`, `"phrase"`, `pref*`, `NEAR(a b, 5)`, `summary: term`. Stemmed; markdown stripped at index time. Double-quote any term with punctuation. Bare `customer-facing` errors (`-` reads as a column filter), bare apostrophes are syntax errors: write `"customer-facing"`, `"founder's"`. This is the default sqlite store's grammar: under `duckdb` and `turso` the operator forms (`a OR b`, `pref*`, `NEAR`, `^`, column filters) are a named error naming the rephrase, and `MATCH` itself does not run (see Stores).
114
114
  - A language written without word spaces (Chinese, Japanese, Thai, Khmer, Lao, Burmese) is indexed per grapheme and searched as an ordered grapheme phrase against the `_seg` sidecar columns: substring semantics, what `grep` gives, a query matches wherever its exact text occurs, including inside a longer run (`京都` matches `东京都政府`, correctly, because it's there at position 2), and needs no minimum length. No decision is needed for these languages. Hand-written SQL is not rewritten for you, so a raw `content MATCH '数据库'` finds nothing: write `content MATCH segment(?)` and bind the terms. `segment()` returns text with no such run unchanged, so it is safe to leave in a query whatever the tree's language.
115
- - Rank with `ORDER BY bm25(content, 10.0, 5.0, 1.0)` (title > summary > body); the full form `bm25(content, 10.0, 5.0, 1.0, 0, 10.0, 5.0, 1.0)` mirrors the same weights onto the `_seg` sidecars, so a title hit found through `title_seg` ranks like one found through `title` (the three-weight form still runs, FTS5 defaults unnamed columns to 1.0, but ranks a sidecar match at body weight). Excerpt with `snippet(content, 2, '«', '»', '…', 10)`, naming the `text` column explicitly: `-1` means best column, which can surface a machine-spaced sidecar as the excerpt (`search` itself always names column 2). snippet() re-tokenizes each matched doc and its cost grows superlinearly with doc size, measured ~10 s per query on a tree holding one 1 MB note. `search` bounds this itself (docs past 16 KB get an equivalent excerpt another way); in hand-written SQL, guard it: `CASE WHEN length(text) <= 16384 THEN snippet(...) END`, or select `title`/`summary` instead of an excerpt.
115
+ - Rank with `ORDER BY bm25(content, 10.0, 5.0, 1.0)` (title > summary > body); the full form `bm25(content, 10.0, 5.0, 1.0, 0, 10.0, 5.0, 1.0)` mirrors the same weights onto the `_seg` sidecars, so a title hit found through `title_seg` ranks like one found through `title` (the three-weight form still runs, FTS5 defaults unnamed columns to 1.0, but ranks a sidecar match at body weight). In hand-written SQLite SQL, `snippet(content, 2, '«', '»', '…', 10)` names the authored `text` column explicitly; `-1` can surface a machine-spaced sidecar instead. `search` does not call SQLite's `snippet()`: it computes bounded passages in shared code for every store. A raw SQL `snippet()` re-tokenizes its matched document, so guard large text with `CASE WHEN length(text) <= 16384 THEN snippet(...) END`, or select `title`/`summary`. Treat historical timing as diagnostic until a current identical-work sitting replaces it.
116
116
  - Select `content.title`/`content.summary` (always exist, empty when absent) rather than `f.title`/`f.summary` (discovered columns; error on trees that never declare them).
117
117
  - Frontmatter values keep their YAML type: strings are TEXT, whole numbers and booleans are INTEGER (`true` stores as 1, so `WHERE flag = 1` matches and `WHERE flag = 'true'` matches nothing), fractions are REAL, lists and maps are JSON text. On the `duckdb` store the discovered columns are VARIANT: homogeneous keys, which are nearly all of them, compare identically, but a numeric predicate against a key that holds numbers in some notes and text in others raises a comparison error where sqlite orders by storage class silently. `map` prints the observed type per field, and a field showing two types (`integer,text`) has drifted across notes. A list key written with no items (`tags:` above a bare `-`) is a list holding one null, stored as the JSON text `[null]`: `IS NULL` does not find it (the column holds a string), `json_each` yields one empty member per such row, and `map` counts it as covered because the key is present. `has(tags, 'x')` reads it correctly as no match. To separate written-but-empty from absent, compare against the text: `WHERE tags = '[null]'`.
118
118
  - **Dead links need the attachment filter.** `dst IS NULL` alone is not "broken link": a wikilink to anything that is not markdown (`[[Board.base]]`, `![[Pasted image.png]]`, `[[spec.pdf]]`) can never resolve, because sense indexes markdown and resolution only tries the exact path or `+.md`. Those are out of the index's universe, not broken. On a 1,400-note Obsidian vault the unfiltered query returns 143 rows where 14 are real. Exclude anything carrying a file extension, as in the recipe above, and widen the exclusion if your notes have dotted titles (`[[Node.js]]` carries one too, so a stricter list, `'*.png'`, `'*.pdf'`, `'*.base'`, and whatever else your vault attaches, is safer on a tree whose titles use dots). Scope it with `preset_files` as well: template and skill files are full of `[[Note Name]]` examples that are deliberately unresolved.
@@ -10,7 +10,7 @@ Querying an existing tree is the `sense` skill. This one covers making a tree: i
10
10
  ## Setup
11
11
 
12
12
  - `npm install -g sensemaking`, then `sense init` at the tree root writes `sense.config.json`: two presets (`default`, and `large` showing what a big tree tunes) and an `embed` block naming the model. The model fetches once per machine at the first vector search (progress on stderr); `sense download` prefetches it instead where that timing matters (CI, air-gapped setup). Config discovery walks up from cwd; `--config <path>` overrides.
13
- - **Backing store.** The config's `store` key: `sqlite` (default, zero-dependency, Node's built-in SQLite), or the experimental `duckdb` and `turso`, whose engine package the first command that opens such a tree installs on its own (`@duckdb/node-api`, a one-time native download of about 110 MB; `@tursodatabase/database`, much smaller). The same commands and tables run on all three. Two things do not port, and each one decides a tree. **FTS5 syntax:** under `duckdb` and `turso`, `search` text and raw `MATCH` reject FTS5's prefix, boolean, `NEAR`, initial-token and column-filter operators with a named error, and sqlite's FTS5 SQL (`MATCH`, `snippet()`, `bm25()`) does not run, so saved queries written in that syntax are sqlite dialect; a tree whose saved queries or search vocabulary depend on FTS5 operators stays on `sqlite`. **SQL functions:** `has`/`basename`/`segment` are registered on `sqlite` and `duckdb` but not `turso`, whose client cannot register them at all, so a tree whose queries call them stays off `turso`. `sense watch` runs on all three; `duckdb` and `turso` lock their cache file per connection, so a concurrent command waits out the watcher's current cycle instead of failing. Each store keeps its own cache file (`.sense/cache.db`, `.sense/cache.duckdb`, `.sense/cache.turso.db`); switching stores is a rebuild, not a migration. Indexing speed is the other axis, and it does not follow from any of the above: sqlite builds a cold index fastest and turso slowest, by a wide margin on a large tree, and no setting closes that gap, since turso's engine costs more per write and more again to maintain each index. What turso buys instead is concurrent writers, non-blocking I/O and encryption, none of which a one-shot command uses, so reach for it for those rather than for speed. Per-store figures: BENCHMARKING.md in the sensemaking repo.
13
+ - **Backing store.** The config's `store` key: `sqlite` (default, zero-dependency, Node's built-in SQLite), or the experimental `duckdb` and `turso`, whose engine package the first command that opens such a tree installs on its own (`@duckdb/node-api`, a one-time native download of about 110 MB; `@tursodatabase/database`, much smaller). The same commands and tables run on all three. Two things do not port, and each one decides a tree. **FTS5 syntax:** under `duckdb` and `turso`, `search` text and raw `MATCH` reject FTS5's prefix, boolean, `NEAR`, initial-token and column-filter operators with a named error, and sqlite's FTS5 SQL (`MATCH`, `snippet()`, `bm25()`) does not run, so saved queries written in that syntax are sqlite dialect; a tree whose saved queries or search vocabulary depend on FTS5 operators stays on `sqlite`. **SQL functions:** `has`/`basename` run on all three (`turso` rewrites them into portable SQL rather than registering them); `segment` runs on `sqlite` and `duckdb` only, so a tree whose queries call `segment` stays off `turso`. `sense watch` runs on all three; `duckdb` and `turso` lock their cache file per connection, so a concurrent command waits out the watcher's current cycle instead of failing. Each store keeps its own cache file (`.sense/cache.db`, `.sense/cache.duckdb`, `.sense/cache.turso.db`); switching stores is a rebuild, not a migration. Speed is a separate axis and must come from a current identical-work measurement. Historical per-store rows can include different ranked candidates and are diagnostics, not an overall store leaderboard. The cache remains an ordinary database file, so a tree indexed under duckdb is readable by DuckDB's ecosystem, while turso brings concurrent access, non-blocking I/O and encryption. Pick for the capabilities you need, then review the current evidence in BENCHMARKING.md.
14
14
  - Globs resolve relative to the config file, never the cwd.
15
15
  - `sense status` and `sense map` show each preset's coverage (files matched, embedded count), so what a config actually indexes is always visible in output. A config edit that changes coverage rebuilds the cache and names the preset that caused it on stderr.
16
16