sensemaking 0.9.5 → 0.10.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (156) hide show
  1. package/README.md +24 -18
  2. package/dist/cjs/cli/check.js +7 -13
  3. package/dist/cjs/cli/check.js.map +1 -1
  4. package/dist/cjs/cli/download.d.cts +3 -0
  5. package/dist/cjs/cli/download.d.ts +3 -0
  6. package/dist/cjs/cli/download.js +214 -0
  7. package/dist/cjs/cli/download.js.map +1 -0
  8. package/dist/cjs/cli/index.d.cts +6 -3
  9. package/dist/cjs/cli/index.d.ts +6 -3
  10. package/dist/cjs/cli/index.js +23 -5
  11. package/dist/cjs/cli/index.js.map +1 -1
  12. package/dist/cjs/cli/init.js +4 -1
  13. package/dist/cjs/cli/init.js.map +1 -1
  14. package/dist/cjs/cli/map.js +2 -2
  15. package/dist/cjs/cli/map.js.map +1 -1
  16. package/dist/cjs/cli/named.js +10 -18
  17. package/dist/cjs/cli/named.js.map +1 -1
  18. package/dist/cjs/cli/path.d.cts +3 -0
  19. package/dist/cjs/cli/path.d.ts +3 -0
  20. package/dist/cjs/cli/path.js +161 -0
  21. package/dist/cjs/cli/path.js.map +1 -0
  22. package/dist/cjs/cli/peek.js +3 -2
  23. package/dist/cjs/cli/peek.js.map +1 -1
  24. package/dist/cjs/cli/related.d.cts +3 -0
  25. package/dist/cjs/cli/related.d.ts +3 -0
  26. package/dist/cjs/cli/related.js +279 -0
  27. package/dist/cjs/cli/related.js.map +1 -0
  28. package/dist/cjs/cli/search.js +3 -7
  29. package/dist/cjs/cli/search.js.map +1 -1
  30. package/dist/cjs/cli/shared.d.cts +3 -1
  31. package/dist/cjs/cli/shared.d.ts +3 -1
  32. package/dist/cjs/cli/shared.js +28 -5
  33. package/dist/cjs/cli/shared.js.map +1 -1
  34. package/dist/cjs/cli/sql.d.cts +3 -0
  35. package/dist/cjs/cli/sql.d.ts +3 -0
  36. package/dist/cjs/cli/{query.js → sql.js} +4 -4
  37. package/dist/cjs/cli/sql.js.map +1 -0
  38. package/dist/cjs/cli/status.js +9 -2
  39. package/dist/cjs/cli/status.js.map +1 -1
  40. package/dist/cjs/cli.js +14 -15
  41. package/dist/cjs/cli.js.map +1 -1
  42. package/dist/cjs/commands.d.cts +12 -4
  43. package/dist/cjs/commands.d.ts +12 -4
  44. package/dist/cjs/commands.js +138 -58
  45. package/dist/cjs/commands.js.map +1 -1
  46. package/dist/cjs/config.d.cts +6 -4
  47. package/dist/cjs/config.d.ts +6 -4
  48. package/dist/cjs/config.js +137 -58
  49. package/dist/cjs/config.js.map +1 -1
  50. package/dist/cjs/db.js +14 -34
  51. package/dist/cjs/db.js.map +1 -1
  52. package/dist/cjs/errors.d.cts +1 -1
  53. package/dist/cjs/errors.d.ts +1 -1
  54. package/dist/cjs/errors.js.map +1 -1
  55. package/dist/cjs/features/embed.d.cts +14 -0
  56. package/dist/cjs/features/embed.d.ts +14 -0
  57. package/dist/cjs/features/embed.js +248 -54
  58. package/dist/cjs/features/embed.js.map +1 -1
  59. package/dist/cjs/features/links.js +7 -16
  60. package/dist/cjs/features/links.js.map +1 -1
  61. package/dist/cjs/features/rank.js +2 -6
  62. package/dist/cjs/features/rank.js.map +1 -1
  63. package/dist/cjs/index.d.cts +1 -1
  64. package/dist/cjs/index.d.ts +1 -1
  65. package/dist/cjs/index.js.map +1 -1
  66. package/dist/cjs/output.js +11 -14
  67. package/dist/cjs/output.js.map +1 -1
  68. package/dist/cjs/progress.js +2 -4
  69. package/dist/cjs/progress.js.map +1 -1
  70. package/dist/cjs/scan.d.cts +1 -0
  71. package/dist/cjs/scan.d.ts +1 -0
  72. package/dist/cjs/scan.js +67 -32
  73. package/dist/cjs/scan.js.map +1 -1
  74. package/dist/cjs/search-error.js +4 -6
  75. package/dist/cjs/search-error.js.map +1 -1
  76. package/dist/cjs/traverse.d.cts +8 -0
  77. package/dist/cjs/traverse.d.ts +8 -0
  78. package/dist/cjs/traverse.js +97 -0
  79. package/dist/cjs/traverse.js.map +1 -0
  80. package/dist/esm/cli/check.js +7 -13
  81. package/dist/esm/cli/check.js.map +1 -1
  82. package/dist/esm/cli/download.d.ts +3 -0
  83. package/dist/esm/cli/download.js +27 -0
  84. package/dist/esm/cli/download.js.map +1 -0
  85. package/dist/esm/cli/index.d.ts +6 -3
  86. package/dist/esm/cli/index.js +13 -7
  87. package/dist/esm/cli/index.js.map +1 -1
  88. package/dist/esm/cli/init.js +4 -1
  89. package/dist/esm/cli/init.js.map +1 -1
  90. package/dist/esm/cli/map.js +3 -2
  91. package/dist/esm/cli/map.js.map +1 -1
  92. package/dist/esm/cli/named.js +10 -16
  93. package/dist/esm/cli/named.js.map +1 -1
  94. package/dist/esm/cli/path.d.ts +3 -0
  95. package/dist/esm/cli/path.js +51 -0
  96. package/dist/esm/cli/path.js.map +1 -0
  97. package/dist/esm/cli/peek.js +4 -2
  98. package/dist/esm/cli/peek.js.map +1 -1
  99. package/dist/esm/cli/related.d.ts +3 -0
  100. package/dist/esm/cli/related.js +26 -0
  101. package/dist/esm/cli/related.js.map +1 -0
  102. package/dist/esm/cli/search.js +2 -5
  103. package/dist/esm/cli/search.js.map +1 -1
  104. package/dist/esm/cli/shared.d.ts +3 -1
  105. package/dist/esm/cli/shared.js +28 -5
  106. package/dist/esm/cli/shared.js.map +1 -1
  107. package/dist/esm/cli/sql.d.ts +3 -0
  108. package/dist/esm/cli/{query.js → sql.js} +4 -4
  109. package/dist/esm/cli/sql.js.map +1 -0
  110. package/dist/esm/cli/status.js +8 -1
  111. package/dist/esm/cli/status.js.map +1 -1
  112. package/dist/esm/cli.js +13 -10
  113. package/dist/esm/cli.js.map +1 -1
  114. package/dist/esm/commands.d.ts +12 -4
  115. package/dist/esm/commands.js +111 -68
  116. package/dist/esm/commands.js.map +1 -1
  117. package/dist/esm/config.d.ts +6 -4
  118. package/dist/esm/config.js +126 -75
  119. package/dist/esm/config.js.map +1 -1
  120. package/dist/esm/db.js +16 -38
  121. package/dist/esm/db.js.map +1 -1
  122. package/dist/esm/errors.d.ts +1 -1
  123. package/dist/esm/errors.js.map +1 -1
  124. package/dist/esm/features/embed.d.ts +14 -0
  125. package/dist/esm/features/embed.js +110 -35
  126. package/dist/esm/features/embed.js.map +1 -1
  127. package/dist/esm/features/links.js +8 -17
  128. package/dist/esm/features/links.js.map +1 -1
  129. package/dist/esm/features/rank.js +2 -6
  130. package/dist/esm/features/rank.js.map +1 -1
  131. package/dist/esm/features/types.js.map +1 -1
  132. package/dist/esm/index.d.ts +1 -1
  133. package/dist/esm/index.js.map +1 -1
  134. package/dist/esm/output.js +13 -16
  135. package/dist/esm/output.js.map +1 -1
  136. package/dist/esm/progress.js +2 -4
  137. package/dist/esm/progress.js.map +1 -1
  138. package/dist/esm/scan.d.ts +1 -0
  139. package/dist/esm/scan.js +41 -34
  140. package/dist/esm/scan.js.map +1 -1
  141. package/dist/esm/search-error.js +4 -6
  142. package/dist/esm/search-error.js.map +1 -1
  143. package/dist/esm/traverse.d.ts +8 -0
  144. package/dist/esm/traverse.js +71 -0
  145. package/dist/esm/traverse.js.map +1 -0
  146. package/package.json +7 -3
  147. package/schema.json +12 -16
  148. package/skills/sense/EXAMPLES.md +8 -8
  149. package/skills/sense/SKILL.md +41 -19
  150. package/skills/sense-setup/EXAMPLES.md +24 -21
  151. package/skills/sense-setup/SKILL.md +9 -9
  152. package/dist/cjs/cli/query.d.cts +0 -3
  153. package/dist/cjs/cli/query.d.ts +0 -3
  154. package/dist/cjs/cli/query.js.map +0 -1
  155. package/dist/esm/cli/query.d.ts +0 -3
  156. package/dist/esm/cli/query.js.map +0 -1
@@ -1,6 +1,6 @@
1
1
  ---
2
2
  name: sense
3
- description: Query a markdown tree with the sense CLI: filter notes by frontmatter, full-text search the prose, follow wikilinks/backlinks, and read note outlines. Use when the user wants to query, filter, count, search, or report on a folder of markdown notes, when you need to find which notes discuss a topic before reading them, when you want a note's backlinks or structure, when a directory has a sense.config.json, or when asked to add a named query to one.
3
+ description: Query a markdown tree with the sense CLI: filter notes by frontmatter, full-text search the prose, follow wikilinks/backlinks, trace how notes connect (link path, k-hop neighborhood, similar-but-unlinked), and read note outlines. Use when the user wants to query, filter, count, search, or report on a folder of markdown notes, when you need to find which notes discuss a topic before reading them, when you want a note's backlinks or structure, when you want to know how two notes connect or what surrounds one in the link graph, when a directory has a sense.config.json, or when asked to add a saved entry to one.
4
4
  ---
5
5
 
6
6
  # sense
@@ -11,13 +11,15 @@ SQL over a markdown tree, kept fresh by a filesystem check on every query. Every
11
11
 
12
12
  Every result is a reference (path, metadata, excerpt), never file contents; prose enters context only when you Read it. Costs: `map` is fixed-size, a `search` row is tens of tokens, and a `peek` stays flat however large the note is. Which tool fits is a property of the question:
13
13
 
14
- - A deterministic, factual answer over known fields (counts, filters, "which notes have X") is SQL: `sense query`, a named query, or `search --where`. Enumerates every match; same result regardless of phrasing.
15
- - Locating notes about something is `search`, one text through every engine the scope has: word match (bare words AND-join, one absent word = zero lexical rows; write `a OR b OR c` for any-word), link-graph expansion, and vector similarity, fused into one ranked list. Read `via` per row: `match` rows contained your words; `vector`-only rows did not. A `vector`-only row means the search words don't appear in that note; it showed up because the model judged it semantically related. Vector rows are conceptual similarity, not typo-tolerance; false positives are expected, labeled, and bounded by `--k`. `--lexical` skips vectors for one command when word-presence is the question.
14
+ - A deterministic, factual answer over known fields (counts, filters, "which notes have X") is SQL: `sense sql`, a saved `{ sql }` entry, or `search --where`. Enumerates every match; same result regardless of phrasing.
15
+ - Locating notes about something is `search`, one text through every engine the scope has: word match (bare words AND-join, one absent word = zero lexical rows; write `a OR b OR c` for any-word), link-graph expansion, and vector similarity, fused into one ranked list. Read `via` per row: `match` rows contained your words; `vector`-only rows did not. A `vector`-only row means the search words don't appear in that note; it showed up because the model judged it semantically related. Vector rows are conceptual similarity, not typo-tolerance; false positives are expected, labeled, and bounded by `--k`. A scope searches with vectors when its preset has `semantic` on (the default) and the tree names an `embed` model; a `semantic: false` preset searches on words and links. A preset that asks for vectors when the model is not downloaded is an error naming `sense download`, not a quieter result: the same search must not answer differently before and after a download.
16
16
  - `map` answers "what is this tree" (fields, hub notes, recent changes) when the tree is unfamiliar.
17
- - `peek <path>` prices a file before you pay for it: outline with `[L143-162, ~380t]` ranges, links both ways. Every list shows its first 20 with the true total; the `sections` and `links` tables hold the rest, so a peek costs a few hundred tokens on any note, heading-dense monsters included.
17
+ - `peek <path>` prices a file before you pay for it: outline with `[L143-162, ~380t]` ranges and links both ways. Every list shows its first 20 with the true total; the `sections` and `links` tables hold the rest, so a peek costs a few hundred tokens on any note.
18
+ - `path <a> <b>` walks the link graph for a chain connecting two notes, or reports none within the depth bound: it answers how they connect, not just that both exist.
19
+ - `related <note>` ranks notes near in meaning to one note that it does not already link to: the links it is missing. It reads the meaning-vectors, so it needs vectors on for the scope and scans them, costing about what a semantic `search` does, not what a `peek` does. Vectors need the model on disk; nothing fetches it implicitly, so `sense download` is a one-time step per machine. Without it `search` still answers on words and links (it prints a note saying vectors are off), while `related` reports the missing model instead of an empty list.
18
20
  - When you know the file and need its contents, `Read` it. sense adds nothing there. On large files peek's ranges let you read just one section; small files are often cheaper whole.
19
21
 
20
- Output defaults to a table, built for humans; `--format json` returns the same rows machine-parseable. That also makes a saved query usable as a CI/hook gate with zero added mechanism: `[ "$(sense <query> --format json)" = "[]" ]` is true exactly when the query returned no rows.
22
+ Output defaults to a table, built for humans; `--format json` returns the same rows machine-parseable. That also makes a saved query usable as a CI/hook gate with zero added mechanism: `[ "$(sense <name> --format json)" = "[]" ]` is true exactly when it returned no rows.
21
23
 
22
24
  ## Commands
23
25
 
@@ -25,34 +27,54 @@ Output defaults to a table, built for humans; `--format json` returns the same r
25
27
  sense search "pricing OR billing OR invoicing" --where "f.status = 'active'" --k 10
26
28
  sense search "sourcing quotes" --preset raw # a named settings bundle from the config
27
29
  sense peek notes/pricing-model.md # a unique basename also works
30
+ sense path onboarding.md pricing-model.md # link chain between two notes, or none within the bound
31
+ sense related notes/pricing-model.md # notes similar by meaning it does not yet link to
28
32
  sense map
29
- sense query "<sql>" [params...] # ad-hoc SQL; ? binds positional args, count-checked
30
- sense <name> [params...] # named query or saved search from sense.config.json
31
- sense --list | status | rebuild | check
33
+ sense sql "<statement>" [params...] # ad-hoc SQL; ? binds positional args, count-checked
34
+ sense <name> [params...] # a saved query from sense.config.json
35
+ sense --list | status | rebuild | check | download
32
36
  ```
33
37
 
34
38
  - Terms pass verbatim to FTS5 MATCH. Bare words AND-join (one absent word means zero rows), so write `OR` yourself when you want any-word matching; double-quote punctuated terms (`"customer-facing"`, `"founder's"`); invalid syntax is an error, not a rewrite. The same rules apply to search commands you write into subagent briefs.
35
39
  - When a search misses, the recall levers are: OR-in synonyms and concrete instances (the index only knows the words in the files; a note about a specific tool rarely names its category), raise `--k` (a row costs tens of tokens), and widen the scope (`--preset`, or `--include` for an ad-hoc glob). Vector rows already cover the meaning-over-words gap by default. Each widening adds candidates and dilutes ranking, so the noise trade-off runs both ways.
36
40
  - A frontmatter query enumerates its matches deterministically; search ranks by term overlap, so results shift as phrasing shifts. Trade-off: a query needs a known field, search doesn't.
37
41
  - The `via` column says what produced each row: `match` (words hit), `link` (connected to notes that hit), `vector` (near in meaning), and combinations. The `lines` column, when set, points at the section that earned the row (the best-matching chunk on vector rows, the term cluster's section on large lexical notes) and is a direct `Read` range; null means the whole note is the reference.
38
- - Scope comes from presets: bare `search` uses the config's `default` preset; `--preset <name>` picks another (unknown names error, listing what's declared); `--include <glob>` is an ad-hoc scope that replaces the preset's globs for one command. `--where` takes any SQL condition against frontmatter alias `f`, not only field equality: `"f.status = 'active' AND has(f.tags, 'x')"`, `"datetime(f.created) >= datetime(?)"`, and filters within the scope. `sense status` shows every preset with its coverage.
42
+ - Scope is one vocabulary shared by `search`, `map`, `peek`, `path`, and `related`: bare command uses the config's `default` preset; `--preset <name>` picks another (unknown names error, listing what's declared); `--include <glob>` and `--exclude <glob>` are ad-hoc globs for one command, each overriding its own side of the preset, so one does not clear the other; `--no-exclude` drops the preset's `exclude` for one command, the only way to widen past it without editing config (it widens the query scope, not the index: a file no preset covers is never indexed). `--where` takes any SQL condition against frontmatter alias `f`, not only field equality: `"f.status = 'active' AND has(f.tags, 'x')"`, `"datetime(f.created) >= datetime(?)"`. There is no whole-index flag: a broad `default` preset, or a declared `all` preset (`include ["**/*"]`), is the whole tree. `sense status` shows every preset with its coverage.
39
43
  - `score` is a rank-fusion value: it ranks rows within one result set and is not comparable across queries, not a relevance magnitude. It encodes how many signals fired and at what rank, so a perfect lexical hit and a weak vector-only hit can read the same number. With vectors active, rows carry `similarity`: the cosine (-1 to 1) of the query against that file's best-matching chunk (the same chunk the `lines` range points at). It orders vector evidence within a result set; the range it spans depends on the corpus and the embedding model, and compresses on small trees, where even a nonsense query has a moderately near neighbour somewhere. Compare similarities within a result set rather than against a fixed cutoff carried between trees.
40
- - Absence evidence lives in the labels: `search --lexical` (or a semantic-off scope) returns 0 rows when the words are nowhere in the tree. Default `search` always returns up to `k` rows (nearest-neighbour search has a nearest neighbour for any input), so a result of only `via: vector` rows IS the absence signal for the words themselves; `similarity` and the snippet are the evidence for judging whether a vector row is a real conceptual hit.
41
- - Besides SQL strings, a config entry can save a whole search: `"hot": { "search": "pricing OR billing", "preset": "raw", "k": 20 }` runs as `sense hot`. The scenario's settings ride along with the name, so repeat runs need no flags. An invocation-level `--preset`, `--k`, `--where`, or `--lexical` overrides the saved value; `--list` marks these entries `(search)`.
42
- - `sense check` prepares every saved query and probes every saved search lexically with k=1, so a typo'd column, stale SQL, bad FTS5 syntax, or unknown preset fails at check time instead of silently mid-task. It reports row counts; whether an empty result is good or bad is the reader's judgment: a dead-link query returning rows means broken citations to fix, and the agent reads that directly.
44
+ - Absence evidence lives in the labels: a `semantic: false` preset (or a tree with no `embed` block) returns 0 rows when the words are nowhere in it. Default `search` always returns up to `k` rows (nearest-neighbour search has a nearest neighbour for any input), so a result of only `via: vector` rows IS the absence signal for the words themselves; `similarity` and the snippet are the evidence for judging whether a vector row is a real conceptual hit.
45
+ - A `queries` entry names the verb it runs, mirroring the two commands: `"dead-links": { "sql": "SELECT src, target FROM links WHERE dst IS NULL" }` runs as `sense dead-links`, and `"hot": { "search": "pricing OR billing", "preset": "raw", "k": 20 }` runs as `sense hot` with its settings baked in, so repeat runs need no flags. An invocation-level `--preset`, `--k`, or `--where` overrides a saved search's value; `--list` labels each entry `(sql)` or `(search)`.
46
+ - `sense check` prepares every `{ sql }` entry and probes every `{ search }` entry with k=1, so a typo'd column, stale SQL, bad FTS5 syntax, or unknown preset fails at check time instead of silently mid-task. The probe skips the vector half, so checking a config never embeds the tree or calls an api endpoint. It reports row counts; whether an empty result is good or bad is the reader's judgment: a dead-link query returning rows means broken citations to fix, and the agent reads that directly.
43
47
 
44
48
  ## SQL
45
49
 
46
50
  The commands are shorthands over those tables; anything they don't express, SQL does.
47
51
 
48
52
  ```
49
- sense query "SELECT name FROM pragma_table_info('frontmatter')" # what fields exist
50
- sense query "SELECT DISTINCT status FROM frontmatter" # what values a field takes
51
- sense query "SELECT src FROM links WHERE dst = ?" notes/pricing-model.md # backlinks
52
- sense query "SELECT src, target FROM links WHERE dst IS NULL" # dead links
53
- sense query "SELECT path FROM frontmatter WHERE path NOT IN (SELECT dst FROM links WHERE dst IS NOT NULL)" # orphans
54
- sense query "SELECT heading, start_line, tokens FROM sections WHERE path = ?" a.md # budget a read
55
- sense query "SELECT j.value, COUNT(*) n FROM frontmatter, json_each(frontmatter.tags) j GROUP BY j.value ORDER BY n DESC" # count per array member
53
+ sense sql "SELECT name FROM pragma_table_info('frontmatter')" # what fields exist
54
+ sense sql "SELECT DISTINCT status FROM frontmatter" # what values a field takes
55
+ sense sql "SELECT src FROM links WHERE dst = ?" notes/pricing-model.md # backlinks
56
+ sense sql "SELECT src, target FROM links WHERE dst IS NULL" # dead links
57
+ sense sql "SELECT path FROM frontmatter WHERE path NOT IN (SELECT dst FROM links WHERE dst IS NOT NULL) AND path NOT IN (SELECT src FROM links)" # linked neither way (fine if intentional; linking is optional)
58
+ sense sql "SELECT heading, start_line, tokens FROM sections WHERE path = ?" a.md # budget a read
59
+ sense sql "SELECT j.value, COUNT(*) n FROM frontmatter, json_each(frontmatter.tags) j GROUP BY j.value ORDER BY n DESC" # count per array member
60
+ ```
61
+
62
+ `path` covers the route between two notes, and `search`'s `via: link` rows are the ranked neighborhood around a query; a structural k-hop walk is a bounded `WITH RECURSIVE` over `links`. Bound the depth: an unbounded walk on a densely linked tree enumerates paths exponentially. Pass these through `sense sql "<sql>" <seed>` or save as `{ "sql": "..." }`.
63
+
64
+ ```
65
+ -- notes within 2 hops of a seed, links both ways (UNION dedups, so it terminates)
66
+ WITH RECURSIVE hop(path, d) AS (
67
+ SELECT ?, 0
68
+ UNION
69
+ SELECT CASE WHEN l.src = hop.path THEN l.dst ELSE l.src END, hop.d + 1
70
+ FROM hop JOIN links l ON (l.src = hop.path OR l.dst = hop.path) AND l.dst IS NOT NULL
71
+ WHERE hop.d < 2
72
+ )
73
+ SELECT DISTINCT path FROM hop WHERE d > 0;
74
+
75
+ -- notes cited alongside a seed: they share a note that links to both (co-citation)
76
+ SELECT DISTINCT b.dst FROM links a JOIN links b ON a.src = b.src
77
+ WHERE a.dst = ? AND b.dst IS NOT NULL AND b.dst <> a.dst;
56
78
  ```
57
79
 
58
80
  - `content MATCH` takes FTS5 syntax: `a OR b`, `"phrase"`, `pref*`, `NEAR(a b, 5)`, `summary: term`. Stemmed; markdown stripped at index time. Double-quote any term with punctuation. Bare `customer-facing` errors (`-` reads as a column filter), bare apostrophes are syntax errors: write `"customer-facing"`, `"founder's"`.
@@ -8,44 +8,46 @@ Four tree shapes, each with its config and the commands an agent actually runs.
8
8
 
9
9
  ```json
10
10
  {
11
- "version": 3,
11
+ "version": 4,
12
12
  "presets": {
13
13
  "default": { "include": ["wiki/**/*.md"], "k": 10 },
14
- "raw": { "include": ["raw/**/*.md"], "k": 5, "semantic": false }
14
+ "raw": { "include": ["raw/**/*.md"], "k": 5 }
15
15
  },
16
+ "embed": { "model": "minishlab/potion-retrieval-32M", "type": "static" },
16
17
  "queries": {
17
- "uncompiled": "SELECT path, _mtime FROM frontmatter WHERE path LIKE 'raw/%' AND path NOT IN (SELECT dst FROM links WHERE dst IS NOT NULL) ORDER BY _mtime DESC",
18
- "stubs": "SELECT path, _size FROM frontmatter WHERE path LIKE 'wiki/%' AND _size < 500 ORDER BY _size",
19
- "dead-links": "SELECT src, target FROM links WHERE dst IS NULL ORDER BY src"
18
+ "uncompiled": { "sql": "SELECT path, _mtime FROM frontmatter WHERE path LIKE 'raw/%' AND path NOT IN (SELECT dst FROM links WHERE dst IS NOT NULL) ORDER BY _mtime DESC" },
19
+ "stubs": { "sql": "SELECT path, _size FROM frontmatter WHERE path LIKE 'wiki/%' AND _size < 500 ORDER BY _size" },
20
+ "dead-links": { "sql": "SELECT src, target FROM links WHERE dst IS NULL ORDER BY src" }
20
21
  }
21
22
  }
22
23
  ```
23
24
 
24
25
  ```
25
26
  sense uncompiled # compile queue: raw files nothing cites yet
26
- sense search "how does attention scale" # wiki only (default preset), vectors on
27
- sense search "rotary embeddings" --preset raw # cite from sources; lexical, k=5
27
+ sense search "how does attention scale" # wiki only (default preset)
28
+ sense search "rotary embeddings" --preset raw # cite from sources, k=5
28
29
  sense dead-links # rows are broken citations to fix
29
30
  ```
30
31
 
31
- What the shape buys: bare search never ranks raw noise above compiled pages; raw pays no vector/link cost; the compile queue, stub list, and citation integrity are one saved query each. The maintenance loop is `uncompiled` → write the wiki page citing its sources → `dead-links` stays empty.
32
+ What the shape buys: bare search never ranks raw noise above compiled pages; the compile queue, stub list, and citation integrity are one saved query each. The maintenance loop is `uncompiled` → write the wiki page citing its sources → `dead-links` stays empty.
32
33
 
33
34
  ## B. Nightly agent memory, consolidated (the dreaming pattern)
34
35
 
35
- `memory/` accumulates small notes written at session end, each with `project`, `created`, and `kind` (observation / steer / decision) frontmatter. A consolidation agent runs periodically: prune, merge, surface contradictions for the human. Retired notes move to `archive/`: still queryable, no longer embedded or ranked.
36
+ `memory/` accumulates small notes written at session end, each with `project`, `created`, and `kind` (observation / steer / decision) frontmatter. A consolidation agent runs periodically: prune, merge, surface contradictions for the human. Retired notes move to `archive/`: still queryable under their own preset, out of the default scope.
36
37
 
37
38
  ```json
38
39
  {
39
- "version": 3,
40
+ "version": 4,
40
41
  "presets": {
41
42
  "default": { "include": ["memory/**/*.md"], "k": 10 },
42
- "archive": { "include": ["archive/**/*.md"], "k": 10, "semantic": false }
43
+ "archive": { "include": ["archive/**/*.md"], "k": 10 }
43
44
  },
45
+ "embed": { "model": "minishlab/potion-retrieval-32M", "type": "static" },
44
46
  "queries": {
45
- "project": "SELECT path, kind, created, title FROM frontmatter WHERE project = ? ORDER BY created DESC",
46
- "steers": "SELECT path, created, title FROM frontmatter WHERE kind = 'steer' AND project = ? ORDER BY created",
47
- "retirement": "SELECT path, project, created FROM frontmatter WHERE datetime(created) < datetime('now','-90 day') AND path NOT IN (SELECT dst FROM links WHERE dst IS NOT NULL)",
48
- "unfiled": "SELECT path FROM frontmatter WHERE project IS NULL"
47
+ "project": { "sql": "SELECT path, kind, created, title FROM frontmatter WHERE project = ? ORDER BY created DESC" },
48
+ "steers": { "sql": "SELECT path, created, title FROM frontmatter WHERE kind = 'steer' AND project = ? ORDER BY created" },
49
+ "retirement": { "sql": "SELECT path, project, created FROM frontmatter WHERE datetime(created) < datetime('now','-90 day') AND path NOT IN (SELECT dst FROM links WHERE dst IS NOT NULL)" },
50
+ "unfiled": { "sql": "SELECT path FROM frontmatter WHERE project IS NULL" }
49
51
  }
50
52
  }
51
53
  ```
@@ -68,15 +70,16 @@ Presets are structural (live vs archived); per-project filtering is metadata (`p
68
70
 
69
71
  ```json
70
72
  {
71
- "version": 3,
73
+ "version": 4,
72
74
  "presets": {
73
75
  "default": { "include": ["reviews/**/*.md", "notes/**/*.md"], "k": 10 },
74
- "source": { "include": ["sources/**/*.md"], "k": 5, "semantic": false }
76
+ "source": { "include": ["sources/**/*.md"], "k": 5 }
75
77
  },
78
+ "embed": { "model": "minishlab/potion-retrieval-32M", "type": "static" },
76
79
  "queries": {
77
- "unsupported": "SELECT path, title FROM frontmatter WHERE path LIKE 'reviews/%' AND path NOT IN (SELECT src FROM links WHERE dst LIKE 'notes/%')",
78
- "unread": "SELECT path FROM frontmatter WHERE path LIKE 'sources/%' AND path NOT IN (SELECT dst FROM links WHERE dst IS NOT NULL)",
79
- "by-topic": "SELECT path, title FROM frontmatter WHERE path NOT LIKE 'sources/%' AND has(topics, ?) ORDER BY path"
80
+ "unsupported": { "sql": "SELECT path, title FROM frontmatter WHERE path LIKE 'reviews/%' AND path NOT IN (SELECT src FROM links WHERE dst LIKE 'notes/%')" },
81
+ "unread": { "sql": "SELECT path FROM frontmatter WHERE path LIKE 'sources/%' AND path NOT IN (SELECT dst FROM links WHERE dst IS NOT NULL)" },
82
+ "by-topic": { "sql": "SELECT path, title FROM frontmatter WHERE path NOT LIKE 'sources/%' AND has(topics, ?) ORDER BY path" }
80
83
  }
81
84
  }
82
85
  ```
@@ -95,7 +98,7 @@ Someone else's Obsidian vault, heterogeneous with no structure worth declaring.
95
98
  ```
96
99
  sense map # fields in use, hub notes, recent changes
97
100
  sense search "dataview queries" # words + links + meaning, one ranked list
98
- sense query "SELECT j.value AS tag, COUNT(*) n FROM frontmatter, json_each(frontmatter.tags) j GROUP BY j.value ORDER BY n DESC LIMIT 20"
101
+ sense sql "SELECT j.value AS tag, COUNT(*) n FROM frontmatter, json_each(frontmatter.tags) j GROUP BY j.value ORDER BY n DESC LIMIT 20"
99
102
  sense peek "Plugins/dataview.md" # outline + links before reading
100
103
  ```
101
104
 
@@ -1,6 +1,6 @@
1
1
  ---
2
2
  name: sense-setup
3
- description: Set up the sense CLI on a markdown tree and make the tree-design decisions that shape it: sense init, presets (which files, which settings, vectors on or off), and the trade-offs of frontmatter conventions, summaries, folder layout, and note size. Use when creating or restructuring a markdown knowledge base, running sense init, editing sense.config.json, configuring search scope or vectors, or deciding how notes should be written for an agent to query later.
3
+ description: Set up the sense CLI on a markdown tree and make the tree-design decisions that shape it: sense init, presets (which files, which settings, whether the scope searches by meaning), the embed block that names the model, and the trade-offs of frontmatter conventions, summaries, folder layout, and note size. Use when creating or restructuring a markdown knowledge base, running sense init, editing sense.config.json, configuring search scope or vectors, or deciding how notes should be written for an agent to query later.
4
4
  ---
5
5
 
6
6
  # sense: setup and tree design
@@ -9,7 +9,7 @@ Querying an existing tree is the `sense` skill. This one covers making a tree: i
9
9
 
10
10
  ## Setup
11
11
 
12
- - `npm install -g sensemaking`, then `sense init` at the tree root writes `sense.config.json`: two presets (`default`, and `large` showing what a big vault tunes), everything on. Config discovery walks up from cwd; `--config <path>` overrides.
12
+ - `npm install -g sensemaking`, then `sense init` at the tree root writes `sense.config.json`: two presets (`default`, and `large` showing what a big tree tunes) and an `embed` block naming the model. `sense download` fetches that model once per machine; nothing fetches it implicitly. Config discovery walks up from cwd; `--config <path>` overrides.
13
13
  - Globs resolve relative to the config file, never the cwd.
14
14
  - `sense status` and `sense map` show each preset's coverage (files matched, embedded count), so what a config actually indexes is always visible in output. A config edit that changes coverage rebuilds the cache and names the preset that caused it on stderr.
15
15
 
@@ -21,29 +21,29 @@ A preset is a named, self-contained bundle of settings. `default` (required) is
21
21
  |---|---|---|
22
22
  | `include` / `exclude` | which files this preset covers (globs) | required |
23
23
  | `k` | how many results a search returns | 10 |
24
- | `semantic` | vectors for this preset's files and searches | on; only ever written as `false` |
24
+ | `semantic` | vectors for this preset's files and its searches | on; only ever written as `false` |
25
25
  | `where` | a standing SQL filter on frontmatter | none |
26
26
 
27
- **Indexing derives from presets.** A file is indexed if any preset includes it; it is embedded if any covering preset has semantic on. Consequences worth designing around:
27
+ **Indexing derives from presets.** A file is indexed if any preset includes it. Consequences worth designing around:
28
28
 
29
- - A layer of the tree covered only by a `semantic: false` preset (raw sources, archives, generated output) is fully searchable lexically and by SQL but costs no vector work. This is the main scale lever.
30
29
  - Files no preset includes are not indexed at all.
31
30
  - Presets may overlap; they are views, not partitions.
32
- - The first semantic search embeds everything covered (progress on stderr; minutes on tens of thousands of notes, seconds on small trees). Vectors use the built-in static model unless a top-level `"embed": { "model", "type": "static"|"api", "url", "key" }` block points at a Model2Vec model, local path, or OpenAI-compatible endpoint. `static` handles paraphrase and reworded concepts; tight domain jargon ("heart attack" for "myocardial infarction") is where an `api` transformer model tends to do better, measured in BENCHMARKING.md, "Retrieval quality".
33
31
  - Global `features` (`links`, `sections`, `rank`) still toggle tree-wide; most trees never touch them.
34
32
 
35
- **Large vaults**: everything except the vector build is measured linear to 100k notes with no tuning (BENCHMARKING.md). The knobs that matter are `k` (more, smaller results; rows carry `lines` section ranges, so agents read sections, not files) and `semantic: false` on the layers that don't earn vectors.
33
+ **Vectors take two decisions, in two places.** The top-level `"embed": { "model", "type": "static"|"api", "url", "key" }` block names the model and says whether the tree has vectors at all. A preset's `semantic` says whether that scope uses them: a layer searched for exact wording (ingested sources, archives, generated output) sets `semantic: false`, costs no embedding, and its searches run on words and links. That is the main scale lever, and it is the llm-wiki split: compiled pages searched by meaning, raw sources searched for the phrasing you are citing. `static` is the built-in pure-JS Model2Vec loader and handles paraphrase and reworded concepts; tight domain jargon ("heart attack" for "myocardial infarction") is where an `api` transformer model tends to do better, measured in BENCHMARKING.md, "Retrieval quality". Nothing downloads the model implicitly: `sense download` fetches it once per machine into the cache (`$XDG_CACHE_HOME/sensemaking/models`, else `~/.cache/...`), one directory per model, so several models coexist and switching between them rebuilds the index rather than mixing vector spaces. A `model` holding a path instead of a Hugging Face id points at a local directory, which `sense download` reports as nothing to fetch. The first search after that embeds the tree (progress on stderr; minutes on tens of thousands of notes, seconds on small trees).
34
+
35
+ **Large vaults**: everything except the vector build is measured linear to 100k notes with no tuning (BENCHMARKING.md). The knobs that matter are `k` (more, smaller results; rows carry `lines` section ranges, so agents read sections, not files) and `semantic: false` on the layers that do not earn vectors.
36
36
 
37
37
  ## Tree design decisions
38
38
 
39
39
  These belong to the tree's owner. sense works with any of them and reads no instruction files of its own; each choice only changes what queries can do.
40
40
 
41
- - **Frontmatter fields.** Columns are discovered per tree: whatever keys notes declare become queryable. Consistent fields across notes make SQL filters and named queries possible (`WHERE status = 'active'`). SQLite's compiled column limit (2,000; sqlite.org/limits.html) bounds distinct keys per tree. The crawl stops with an error naming the count and the levers. Reserved keys (dropped with a warning): `path`, `_mtime`, `_size`, `_rank`, `content`, `links`, `sections`. Values keep their YAML type: strings TEXT, whole numbers and booleans INTEGER (`true` is 1), fractions REAL, lists and maps JSON text; `map` prints the observed type per field.
41
+ - **Frontmatter fields.** Columns are discovered per tree: whatever keys notes declare become queryable. Consistent fields across notes make SQL filters and saved queries possible (`WHERE status = 'active'`). SQLite's compiled column limit (2,000; sqlite.org/limits.html) bounds distinct keys per tree. The crawl stops with an error naming the count and the levers. Reserved keys (dropped with a warning): `path`, `_mtime`, `_size`, `_rank`, `content`, `links`, `sections`. Values keep their YAML type: strings TEXT, whole numbers and booleans INTEGER (`true` is 1), fractions REAL, lists and maps JSON text; `map` prints the observed type per field.
42
42
  - **Presets are path-shaped; frontmatter is state-shaped.** A preset's coverage must be computable from the path alone (it decides indexing, baked into the cache). Volatile state (`status`, `project`, dates) lives in frontmatter and filters at query time (`where`, `has()`, `datetime()`). A state worth different *indexing* (retired memory, superseded sources) is a state worth moving the file: the archive-folder pattern in EXAMPLES.md.
43
43
  - **What a note omits is also a filter.** A layer that deliberately carries none of the fields the saved views filter on is excluded from all of them without any view naming the layer. Sparse fields cut both ways: less of the tree filters when you want breadth, and exactly this separation when layers differ in authority.
44
44
  - **Dates.** `datetime()` comparisons work for dates written as ISO 8601, the only format it parses. A tree that mixes date formats can store them, but can't compare them in SQL.
45
45
  - **Summaries.** A one-line `summary:` is optional and pays twice: it shows in every result row (often answering a question with no file read) and is a weighted search field ranked above body text. The cost is writing and maintaining the line as notes change.
46
46
  - **Folder shape.** Globs find the files, paths are queryable text, links resolve by basename at any depth, but presets make folders meaningful: a folder is the natural unit that gets its own coverage and settings.
47
47
  - **Note size.** Many small notes: precise search hits, whole-file reads stay cheap, more links to maintain. Fewer large notes: `sections`, `peek`, and the `lines` column carry the cost down to line-range reads. Both work.
48
- - **Recurring questions.** Save a scenario an agent will repeat: a SQL string for filters and reports, or a saved search (`"hot": { "search": "...", "preset": "raw", "k": 5 }`). Either runs as `sense <name>`, and `sense check` validates both kinds against the real tree, so a broken saved scenario fails at check time, not mid-task.
48
+ - **Recurring questions.** Save a scenario an agent will repeat under `queries`, naming the verb it runs: `{ "sql": "..." }` for filters and reports, or `{ "search": "...", "preset": "raw", "k": 5 }` for a ranked search. Either runs as `sense <name>`, and `sense check` validates both kinds against the real tree, so a broken saved scenario fails at check time, not mid-task.
49
49
  - **Where decisions live.** Choices that should outlive one conversation can be recorded in the agent's own instruction or skill files, or in a note in the tree itself; a one-off search over an existing corpus needs none of that.
@@ -1,3 +0,0 @@
1
- import type { Command } from './types.js';
2
- declare const query: Command;
3
- export default query;
@@ -1,3 +0,0 @@
1
- import type { Command } from './types.js';
2
- declare const query: Command;
3
- export default query;
@@ -1 +0,0 @@
1
- {"version":3,"sources":["/Users/kevin/Dev/OpenSource/ai/sensemaking/src/cli/query.ts"],"sourcesContent":["import { USAGE } from './index.ts';\nimport { CONFIG, FORMAT, formatOf, parse, runSql } from './shared.ts';\nimport type { Command } from './types.ts';\n\nconst query: Command = (ctx) => {\n const usage = `usage: ${ctx.name} ${USAGE.query}`;\n const { values, positionals } = parse(ctx.argv, usage, { ...FORMAT, ...CONFIG });\n const [sql, ...params] = positionals;\n if (!sql) ctx.usageError(usage);\n const format = formatOf(values);\n runSql(ctx.resolveConfig(values.config as string | undefined), sql, params, format, 'ad-hoc query');\n};\nexport default query;\n"],"names":["query","ctx","usage","USAGE","name","parse","argv","FORMAT","CONFIG","values","positionals","sql","params","usageError","format","formatOf","runSql","resolveConfig","config"],"mappings":";;;;+BAYA;;;eAAA;;;uBAZsB;wBACkC;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AAGxD,IAAMA,QAAiB,eAACC;IACtB,IAAMC,QAAQ,AAAC,UAAqBC,OAAZF,IAAIG,IAAI,EAAC,KAAe,OAAZD,cAAK,CAACH,KAAK;IAC/C,IAAgCK,SAAAA,IAAAA,eAAK,EAACJ,IAAIK,IAAI,EAAEJ,OAAO,mBAAKK,gBAAM,EAAKC,gBAAM,IAArEC,SAAwBJ,OAAxBI,QAAQC,cAAgBL,OAAhBK;IAChB,IAAyBA,yBAAAA,cAAlBC,MAAkBD,iBAAb,AAAGE,SAAUF,mBAAb;IACZ,IAAI,CAACC,KAAKV,IAAIY,UAAU,CAACX;IACzB,IAAMY,SAASC,IAAAA,kBAAQ,EAACN;IACxBO,IAAAA,gBAAM,EAACf,IAAIgB,aAAa,CAACR,OAAOS,MAAM,GAAyBP,KAAKC,QAAQE,QAAQ;AACtF;IACA,WAAed"}
@@ -1,3 +0,0 @@
1
- import type { Command } from './types.js';
2
- declare const query: Command;
3
- export default query;
@@ -1 +0,0 @@
1
- {"version":3,"sources":["/Users/kevin/Dev/OpenSource/ai/sensemaking/src/cli/query.ts"],"sourcesContent":["import { USAGE } from './index.ts';\nimport { CONFIG, FORMAT, formatOf, parse, runSql } from './shared.ts';\nimport type { Command } from './types.ts';\n\nconst query: Command = (ctx) => {\n const usage = `usage: ${ctx.name} ${USAGE.query}`;\n const { values, positionals } = parse(ctx.argv, usage, { ...FORMAT, ...CONFIG });\n const [sql, ...params] = positionals;\n if (!sql) ctx.usageError(usage);\n const format = formatOf(values);\n runSql(ctx.resolveConfig(values.config as string | undefined), sql, params, format, 'ad-hoc query');\n};\nexport default query;\n"],"names":["USAGE","CONFIG","FORMAT","formatOf","parse","runSql","query","ctx","usage","name","values","positionals","argv","sql","params","usageError","format","resolveConfig","config"],"mappings":"AAAA,SAASA,KAAK,QAAQ,aAAa;AACnC,SAASC,MAAM,EAAEC,MAAM,EAAEC,QAAQ,EAAEC,KAAK,EAAEC,MAAM,QAAQ,cAAc;AAGtE,MAAMC,QAAiB,CAACC;IACtB,MAAMC,QAAQ,CAAC,OAAO,EAAED,IAAIE,IAAI,CAAC,CAAC,EAAET,MAAMM,KAAK,EAAE;IACjD,MAAM,EAAEI,MAAM,EAAEC,WAAW,EAAE,GAAGP,MAAMG,IAAIK,IAAI,EAAEJ,OAAO;QAAE,GAAGN,MAAM;QAAE,GAAGD,MAAM;IAAC;IAC9E,MAAM,CAACY,KAAK,GAAGC,OAAO,GAAGH;IACzB,IAAI,CAACE,KAAKN,IAAIQ,UAAU,CAACP;IACzB,MAAMQ,SAASb,SAASO;IACxBL,OAAOE,IAAIU,aAAa,CAACP,OAAOQ,MAAM,GAAyBL,KAAKC,QAAQE,QAAQ;AACtF;AACA,eAAeV,MAAM"}