sensemaking 0.8.0 → 0.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (102) hide show
  1. package/README.md +48 -39
  2. package/dist/cjs/cli/check.js +24 -43
  3. package/dist/cjs/cli/check.js.map +1 -1
  4. package/dist/cjs/cli/index.js +2 -2
  5. package/dist/cjs/cli/index.js.map +1 -1
  6. package/dist/cjs/cli/named.js +18 -8
  7. package/dist/cjs/cli/named.js.map +1 -1
  8. package/dist/cjs/cli/search.d.cts +3 -0
  9. package/dist/cjs/cli/search.d.ts +3 -0
  10. package/dist/cjs/cli/{find.js → search.js} +7 -5
  11. package/dist/cjs/cli/search.js.map +1 -0
  12. package/dist/cjs/cli/shared.js.map +1 -1
  13. package/dist/cjs/cli/status.js +29 -12
  14. package/dist/cjs/cli/status.js.map +1 -1
  15. package/dist/cjs/cli/types.d.cts +3 -1
  16. package/dist/cjs/cli/types.d.ts +3 -1
  17. package/dist/cjs/cli.js +25 -5
  18. package/dist/cjs/cli.js.map +1 -1
  19. package/dist/cjs/commands.d.cts +11 -2
  20. package/dist/cjs/commands.d.ts +11 -2
  21. package/dist/cjs/commands.js +86 -21
  22. package/dist/cjs/commands.js.map +1 -1
  23. package/dist/cjs/config.d.cts +41 -14
  24. package/dist/cjs/config.d.ts +41 -14
  25. package/dist/cjs/config.js +448 -160
  26. package/dist/cjs/config.js.map +1 -1
  27. package/dist/cjs/db.d.cts +1 -1
  28. package/dist/cjs/db.d.ts +1 -1
  29. package/dist/cjs/db.js +208 -62
  30. package/dist/cjs/db.js.map +1 -1
  31. package/dist/cjs/errors.d.cts +1 -1
  32. package/dist/cjs/errors.d.ts +1 -1
  33. package/dist/cjs/errors.js.map +1 -1
  34. package/dist/cjs/features/embed.js +12 -2
  35. package/dist/cjs/features/embed.js.map +1 -1
  36. package/dist/cjs/features/types.d.cts +2 -1
  37. package/dist/cjs/features/types.d.ts +2 -1
  38. package/dist/cjs/index.d.cts +3 -3
  39. package/dist/cjs/index.d.ts +3 -3
  40. package/dist/cjs/index.js +6 -3
  41. package/dist/cjs/index.js.map +1 -1
  42. package/dist/cjs/output.d.cts +15 -0
  43. package/dist/cjs/output.d.ts +15 -0
  44. package/dist/cjs/output.js +26 -5
  45. package/dist/cjs/output.js.map +1 -1
  46. package/dist/cjs/scan.d.cts +3 -0
  47. package/dist/cjs/scan.d.ts +3 -0
  48. package/dist/cjs/scan.js +84 -4
  49. package/dist/cjs/scan.js.map +1 -1
  50. package/dist/esm/cli/check.js +20 -41
  51. package/dist/esm/cli/check.js.map +1 -1
  52. package/dist/esm/cli/index.js +1 -1
  53. package/dist/esm/cli/index.js.map +1 -1
  54. package/dist/esm/cli/named.js +17 -9
  55. package/dist/esm/cli/named.js.map +1 -1
  56. package/dist/esm/cli/search.d.ts +3 -0
  57. package/dist/esm/cli/search.js +16 -0
  58. package/dist/esm/cli/search.js.map +1 -0
  59. package/dist/esm/cli/shared.js +1 -1
  60. package/dist/esm/cli/shared.js.map +1 -1
  61. package/dist/esm/cli/status.js +10 -10
  62. package/dist/esm/cli/status.js.map +1 -1
  63. package/dist/esm/cli/types.d.ts +3 -1
  64. package/dist/esm/cli/types.js.map +1 -1
  65. package/dist/esm/cli.js +20 -4
  66. package/dist/esm/cli.js.map +1 -1
  67. package/dist/esm/commands.d.ts +11 -2
  68. package/dist/esm/commands.js +76 -24
  69. package/dist/esm/commands.js.map +1 -1
  70. package/dist/esm/config.d.ts +41 -14
  71. package/dist/esm/config.js +354 -114
  72. package/dist/esm/config.js.map +1 -1
  73. package/dist/esm/db.d.ts +1 -1
  74. package/dist/esm/db.js +53 -6
  75. package/dist/esm/db.js.map +1 -1
  76. package/dist/esm/errors.d.ts +1 -1
  77. package/dist/esm/errors.js.map +1 -1
  78. package/dist/esm/features/embed.js +12 -2
  79. package/dist/esm/features/embed.js.map +1 -1
  80. package/dist/esm/features/types.d.ts +2 -1
  81. package/dist/esm/features/types.js.map +1 -1
  82. package/dist/esm/index.d.ts +3 -3
  83. package/dist/esm/index.js +1 -1
  84. package/dist/esm/index.js.map +1 -1
  85. package/dist/esm/output.d.ts +15 -0
  86. package/dist/esm/output.js +19 -5
  87. package/dist/esm/output.js.map +1 -1
  88. package/dist/esm/scan.d.ts +3 -0
  89. package/dist/esm/scan.js +32 -4
  90. package/dist/esm/scan.js.map +1 -1
  91. package/package.json +9 -3
  92. package/schema.json +74 -54
  93. package/skills/sense/EXAMPLES.md +8 -6
  94. package/skills/sense/SKILL.md +51 -48
  95. package/skills/sense-setup/EXAMPLES.md +121 -0
  96. package/skills/sense-setup/SKILL.md +71 -66
  97. package/dist/cjs/cli/find.d.cts +0 -3
  98. package/dist/cjs/cli/find.d.ts +0 -3
  99. package/dist/cjs/cli/find.js.map +0 -1
  100. package/dist/esm/cli/find.d.ts +0 -3
  101. package/dist/esm/cli/find.js +0 -14
  102. package/dist/esm/cli/find.js.map +0 -1
package/schema.json CHANGED
@@ -4,7 +4,7 @@
4
4
  "description": "Config for sense -- SQL over the frontmatter, content, links, and structure of a markdown tree.",
5
5
  "type": "object",
6
6
  "additionalProperties": false,
7
- "required": ["scan", "queries"],
7
+ "required": ["presets", "queries"],
8
8
  "properties": {
9
9
  "$schema": {
10
10
  "type": "string",
@@ -12,31 +12,54 @@
12
12
  },
13
13
  "version": {
14
14
  "type": "integer",
15
- "enum": [1, 2],
15
+ "enum": [1, 2, 3],
16
16
  "description": "Config format version. Older versions are auto-migrated (and the file rewritten) on load; a version newer than this sense build supports makes it exit with an error rather than misinterpret the file. Omit to default to 1."
17
17
  },
18
- "scan": {
18
+ "presets": {
19
19
  "type": "object",
20
- "additionalProperties": false,
21
- "required": ["include"],
22
- "description": "Which files become rows in the `frontmatter` table.",
23
- "properties": {
24
- "include": {
25
- "type": "array",
26
- "items": { "type": "string" },
27
- "minItems": 1,
28
- "description": "Glob patterns (globby syntax), resolved relative to this config file's own directory -- never the invocation cwd."
20
+ "description": "File selection, index-time, and named search defaults. A preset is self-contained: no inheritance between presets, and no ordering -- presets may overlap freely (they are views, not partitions), and a preset whose globs match nothing is valid. A file is indexed iff any preset's include/exclude covers it (union across presets); it is embedded iff any covering preset has semantic on. `default` is used when a command names no preset, and must be declared. `sense init` generates `default` plus one `large` example.",
21
+ "minProperties": 1,
22
+ "additionalProperties": {
23
+ "type": "object",
24
+ "additionalProperties": false,
25
+ "required": ["include"],
26
+ "properties": {
27
+ "include": {
28
+ "type": "array",
29
+ "items": { "type": "string" },
30
+ "minItems": 1,
31
+ "description": "Glob patterns (globby syntax), resolved relative to this config file's own directory -- never the invocation cwd. Required for every preset, including `default`."
32
+ },
33
+ "exclude": {
34
+ "type": "array",
35
+ "items": { "type": "string" },
36
+ "minItems": 1,
37
+ "description": "Glob patterns excluded from this preset's `include`, tsconfig-standard semantics."
38
+ },
39
+ "k": {
40
+ "type": "integer",
41
+ "minimum": 1,
42
+ "description": "Default result count for `search` when this preset is in effect (named, or `default` when none is). A --k on the invocation, or a saved search's own `k`, overrides it. Defaults to 10 when the preset omits it too."
43
+ },
44
+ "semantic": {
45
+ "type": "boolean",
46
+ "description": "Vector participation for files this preset covers. Absent or true means on (the default); only ever meaningfully written as false. A file covered only by semantic-false presets is indexed but never embedded, so it searches lexically automatically -- no error, just fewer signals."
47
+ },
48
+ "where": {
49
+ "type": "string",
50
+ "description": "Standing SQL condition against frontmatter alias `f`, applied when `search` runs under this preset without an explicit --where. An explicit --where (or a saved search's own `where`) replaces it rather than ANDing with it, so a caller can always widen back to the preset's whole scope."
51
+ }
29
52
  }
30
53
  }
31
54
  },
32
55
  "features": {
33
56
  "type": "object",
34
57
  "additionalProperties": false,
35
- "description": "Independently toggleable features. Absent block or key means enabled; set false to disable. Toggling a feature rebuilds the cache. Commands degrade rather than fail: `find` without links is BM25-only, `map` without rank omits hubs, `peek` without sections omits the outline.",
58
+ "description": "Global feature defaults. Absent block or key means enabled. `embed` is not a member of this block -- vector participation is controlled per preset by `semantic` (see \"presets\" and the top-level \"embed\" block for provider settings). Toggling a feature rebuilds the cache. Commands degrade rather than fail: `search` without links is BM25-only, `map` without rank omits hubs, `peek` without sections omits the outline.",
36
59
  "properties": {
37
60
  "links": {
38
61
  "type": "boolean",
39
- "description": "Extract wikilinks and relative markdown links into the `links` table (src, target as written, dst resolved path or NULL for dead links; an ambiguous basename resolves to the lexicographically first match). Powers backlinks, graph expansion in `find`, and `peek`'s link lists."
62
+ "description": "Extract wikilinks and relative markdown links into the `links` table (src, target as written, dst resolved path or NULL for dead links; an ambiguous basename resolves to the lexicographically first match). Powers backlinks, graph expansion in `search`, and `peek`'s link lists."
40
63
  },
41
64
  "sections": {
42
65
  "type": "boolean",
@@ -45,63 +68,60 @@
45
68
  "rank": {
46
69
  "type": "boolean",
47
70
  "description": "PageRank over resolved links into `frontmatter._rank` at reconcile -- a static importance prior. Powers `map`'s hub list; usable in any ORDER BY. Requires `links`."
48
- },
49
- "embed": {
50
- "description": "Semantic vectors, opt-in (absent = off; most trees don't need them). When on, vectors are computed and kept fresh at reconcile; expansion runs only on `find --semantic` invocations -- default `find` results are unchanged. `true` = defaults (static type, minishlab/potion-retrieval-32M, downloaded to ~/.cache/sensemaking on first use). Object form: `model` (Hugging Face id or local path), `type` (`static` = built-in pure-JS Model2Vec loader; `api` = OpenAI-compatible POST /embeddings), `url` (api base, e.g. http://localhost:11434/v1), `key` (name of the env var holding the bearer token). Changing model or type rebuilds the cache like a feature toggle.",
51
- "oneOf": [
52
- { "type": "boolean" },
53
- {
54
- "type": "object",
55
- "additionalProperties": false,
56
- "properties": {
57
- "model": { "type": "string" },
58
- "type": { "type": "string", "enum": ["static", "api"] },
59
- "url": { "type": "string" },
60
- "key": { "type": "string" }
61
- }
62
- }
63
- ]
64
71
  }
65
72
  }
66
73
  },
67
- "defaults": {
74
+ "embed": {
68
75
  "type": "object",
69
76
  "additionalProperties": false,
70
- "description": "Tree-declared defaults for the commands.",
77
+ "description": "Embed provider settings only -- presence alone does not enable embedding. Vector participation is decided per preset by that preset's `semantic` field; this block only says which model/provider to use when at least one preset wants vectors. `model` (Hugging Face id or local path, default minishlab/potion-retrieval-32M), `type` (`static` = built-in pure-JS Model2Vec loader, default; `api` = OpenAI-compatible POST /embeddings), `url` (api base, e.g. http://localhost:11434/v1), `key` (name of the env var holding the bearer token). Changing model or type rebuilds the cache like a feature toggle.",
71
78
  "properties": {
72
- "find": {
73
- "type": "object",
74
- "additionalProperties": false,
75
- "description": "Default scope for `find`. Use when a tree holds a layer that should stay queryable but out of ordinary search -- generated output, or unverified source extractions that must not be cited as conclusions.",
76
- "properties": {
77
- "where": {
78
- "type": "string",
79
- "description": "SQL condition against frontmatter alias `f`, e.g. \"f.type != 'raw'\". Applied when `find` runs without --where; an explicit --where replaces it (so `--where \"1=1\"` searches the whole tree). Reported by `sense status`."
80
- }
81
- }
82
- }
79
+ "model": { "type": "string" },
80
+ "type": { "type": "string", "enum": ["static", "api"] },
81
+ "url": { "type": "string" },
82
+ "key": { "type": "string" }
83
83
  }
84
84
  },
85
- "checks": {
86
- "type": "object",
87
- "description": "Assertions over saved queries, for trees whose queries encode invariants: `\"checks\": { \"dead-links\": \"empty\" }` makes `sense check` fail (exit 1) when that query returns rows. Without an entry, `check` reports zero-row queries as `ok, but returns 0 rows` -- informational, since an ordinary query returning nothing is sometimes a bug and sometimes just an empty result. Keys must name saved queries.",
88
- "additionalProperties": { "type": "string", "enum": ["empty"] }
89
- },
90
85
  "queries": {
91
86
  "type": "object",
92
- "description": "Named queries runnable as `sense <name> [params...]`, either a SQL string or a saved find. SQL string: `?` placeholders bind to CLI positional args in order. Tables: `frontmatter` (one row per file, one column per discovered frontmatter key, plus `path`/`_mtime`/`_size`/`_rank`), `content` (FTS5: `title`, `summary`, `text`, `path`), `links` (`src`, `target`, `dst`), and `sections` (`path`, `idx`, `level`, `heading`, `start_line`, `end_line`, `tokens`). `has(field, value)`: array membership on a JSON-array field, substring match on a string (so has(f.status, 'active') also matches 'inactive'), false on NULL. Exact matches: `=` for scalars, `EXISTS (SELECT 1 FROM json_each(f.tags) WHERE value = ?)` for array members. Canonical query: `SELECT f.path, content.title, content.summary, CASE WHEN length(content.text) <= 16384 THEN snippet(content, -1, '«', '»', '…', 10) END AS hit FROM frontmatter f JOIN content ON content.path = f.path WHERE content MATCH ? ORDER BY bm25(content, 10.0, 5.0, 1.0) LIMIT 10`. The CASE bounds snippet(), which re-tokenizes each matched document and costs seconds per query once a tree holds a megabyte-scale note; `find` applies the same bound internally. Saved find object: `sense <name>` behaves like `sense find <find> [--k] [--where] [--semantic]` with zero flags; it takes no positional parameters. Reserved frontmatter keys: `path`, `_mtime`, `_size`, `_rank`, `content`, `links`, `sections`. Reserved query names (unreachable as subcommands, either form): `init`, `query`, `find`, `map`, `peek`, `watch`, `status`, `rebuild`, `check`.",
87
+ "description": "Named queries runnable as `sense <name> [params...]`, one of three shapes: a raw SQL string, `{ sql }`, or a saved search. SQL (string or `{ sql }`): `?` placeholders bind to CLI positional args in order; deterministic and enumerating, including raw FTS5 `MATCH` for word search under your own SQL -- \"0 rows = not in the tree\" lives here. Tables: `frontmatter` (one row per file, one column per discovered frontmatter key, plus `path`/`_mtime`/`_size`/`_rank`), `content` (FTS5: `title`, `summary`, `text`, `path`), `links` (`src`, `target`, `dst`), `sections` (`path`, `idx`, `level`, `heading`, `start_line`, `end_line`, `tokens`), and `preset_files` (`preset`, `path`) -- which presets cover which files. `has(field, value)`: array membership on a JSON-array field, substring match on a string (so has(f.status, 'active') also matches 'inactive'), false on NULL. Exact matches: `=` for scalars, `EXISTS (SELECT 1 FROM json_each(f.tags) WHERE value = ?)` for array members. Canonical query: `SELECT f.path, content.title, content.summary, CASE WHEN length(content.text) <= 16384 THEN snippet(content, -1, '«', '»', '…', 10) END AS hit FROM frontmatter f JOIN content ON content.path = f.path WHERE content MATCH ? ORDER BY bm25(content, 10.0, 5.0, 1.0) LIMIT 10`. The CASE bounds snippet(), which re-tokenizes each matched document and costs seconds per query once a tree holds a megabyte-scale note; `search` applies the same bound internally. Saved search object: one text driving every engine the scoped preset has -- word match, links, vectors -- fused into one ranked list, `via` labeling which engine produced each row; `search` text must be non-empty (a saved query saves a question -- a scope without one is just flags). `preset` names one declared preset (defaults to `default`); `include` is an ad hoc glob scope that replaces the preset's include/exclude entirely, same as `search --include`; `where` and `k` behave like the `search` command's flags. `semantic: false` opts a query down to lexical only -- the rare precision opt-out, not a gate (semantic participation otherwise follows the scoped preset's `semantic` automatically). `sense <name>` behaves like `sense search <search> [--preset] [--include] [--where] [--k] [--lexical]` with zero flags; it takes no positional parameters. `sense check` probes every saved query/search (lexically, k=1 for searches) so a broken one fails loudly at check time -- it makes no assertion about the result itself; a returned row set is the reader's judgment. Reserved frontmatter keys: `path`, `_mtime`, `_size`, `_rank`, `content`, `links`, `sections`. Reserved query names (unreachable as subcommands, any shape): `init`, `query`, `search`, `find`, `map`, `peek`, `watch`, `status`, `rebuild`, `check`.",
93
88
  "additionalProperties": {
94
89
  "oneOf": [
95
90
  { "type": "string" },
96
91
  {
97
92
  "type": "object",
98
93
  "additionalProperties": false,
99
- "required": ["find"],
94
+ "required": ["sql"],
100
95
  "properties": {
101
- "find": { "type": "string", "description": "FTS5 MATCH terms, verbatim -- same syntax as `sense find \"<terms>\"`." },
102
- "k": { "type": "integer", "minimum": 1, "description": "Result count, same as `find --k`. Defaults to 10. A --k on the invocation overrides this." },
103
- "where": { "type": "string", "description": "SQL condition against frontmatter alias `f`, same as `find --where`. A --where on the invocation replaces this rather than ANDing with it." },
104
- "semantic": { "type": "boolean", "description": "Adds vector expansion, same as `find --semantic`. Requires features.embed; `sense check` reports a saved find with semantic true as failed when embed is off, without running it." }
96
+ "sql": { "type": "string", "description": "Raw SQL, same as the string shape -- `?` placeholders bind to CLI positional args in order." }
97
+ }
98
+ },
99
+ {
100
+ "type": "object",
101
+ "additionalProperties": false,
102
+ "required": ["search"],
103
+ "properties": {
104
+ "search": {
105
+ "type": "string",
106
+ "minLength": 1,
107
+ "description": "One text driving every engine the scoped preset has -- word match, links, vectors -- fused into one ranked list. Same syntax as `sense search \"<text>\"`. Must be non-empty."
108
+ },
109
+ "preset": {
110
+ "type": "string",
111
+ "description": "Preset name to search under. Defaults to `default`. An unknown name is an error listing declared presets."
112
+ },
113
+ "include": {
114
+ "type": "array",
115
+ "items": { "type": "string" },
116
+ "minItems": 1,
117
+ "description": "Ad hoc glob scope, same as `search --include`. Replaces the preset's include/exclude entirely rather than layering on top of it."
118
+ },
119
+ "k": { "type": "integer", "minimum": 1, "description": "Result count, same as `search --k`. Defaults to the preset's own `k`, then 10. A --k on the invocation overrides this." },
120
+ "where": { "type": "string", "description": "SQL condition against frontmatter alias `f`, same as `search --where`. A --where on the invocation replaces this rather than ANDing with it." },
121
+ "semantic": {
122
+ "type": "boolean",
123
+ "description": "false opts this query down to lexical-only, same as `search --lexical` -- the rare precision opt-out. Semantic participation otherwise follows the scoped preset's `semantic` automatically, so true is redundant with the default."
124
+ }
105
125
  }
106
126
  }
107
127
  ]
@@ -6,7 +6,7 @@ the filesystem, on the paths that earned it.
6
6
  ## A. "Do the notes say anything about X?"
7
7
 
8
8
  ```
9
- sense find "pricing OR billing OR invoicing" --k 10 --format json
9
+ sense search "pricing OR billing OR invoicing" --k 10 --format json
10
10
  ```
11
11
 
12
12
  ```json
@@ -79,7 +79,7 @@ sense query "SELECT DISTINCT type FROM frontmatter" # what a field's values a
79
79
  ## F. "The notes say it in different words"
80
80
 
81
81
  ```
82
- sense find "children dying from poor nutrition" --semantic --k 3 --format json
82
+ sense search "children dying from poor nutrition" --k 3 --format json
83
83
  ```
84
84
 
85
85
  ```json
@@ -90,7 +90,9 @@ sense find "children dying from poor nutrition" --semantic --k 3 --format json
90
90
  ```
91
91
 
92
92
  A `via: "vector"` row never contained the terms — it is semantically near them; `similarity` is
93
- the cosine against the chunk `lines` names, a direct `Read` range. Only on trees whose config enables `features.embed`.
93
+ the cosine against the chunk `lines` names, a direct `Read` range. Vector rows appear whenever
94
+ the scope's preset has semantic on (the default); a result of only vector rows means the words
95
+ themselves are nowhere in the scope.
94
96
 
95
97
  ## Consequences
96
98
 
@@ -98,9 +100,9 @@ the cosine against the chunk `lines` names, a direct `Read` range. Only on trees
98
100
  |---|---|---|
99
101
  | `SELECT text FROM content` | returns the tree's entire prose | `snippet(content, -1, '«', '»', '…', 10)` excerpts the match |
100
102
  | `snippet()` on a tree holding a megabyte-scale note | re-tokenizes the whole document per matched row: seconds per query | bound it: `CASE WHEN length(content.text) <= 16384 THEN snippet(...) END`, or select `summary` instead |
101
- | `sense find "pricing"` | matches only that word's stem | OR-in synonyms and instances: `"pricing OR billing OR invoicing"` |
102
- | `sense find "pricing model details"` | bare words AND-join; one absent word = zero rows | OR the words, or quote an exact phrase |
103
- | `sense find "customer-facing OR on-site"` | bare punctuation is FTS5 syntax (`-` reads as a column filter) | double-quote: `"customer-facing" OR "on-site"` |
103
+ | `sense search "pricing"` | lexically matches only that word's stem (vector rows still widen by meaning) | OR-in synonyms and instances: `"pricing OR billing OR invoicing"` |
104
+ | `sense search "pricing model details"` | bare words AND-join; one absent word = zero lexical rows | OR the words, or quote an exact phrase |
105
+ | `sense search "customer-facing OR on-site"` | bare punctuation is FTS5 syntax (`-` reads as a column filter) | double-quote: `"customer-facing" OR "on-site"` |
104
106
  | `Read` of a large file for one section | costs the whole file | `peek`, then `Read` the line range |
105
107
  | row queries without `LIMIT` | unbounded output (aggregates are already bounded) | `LIMIT n` |
106
108
  | saving one-off queries to config | config churn | ad-hoc `sense query`; save reusable views |
@@ -14,20 +14,21 @@ storage; `map` and `status` report which are on.
14
14
  ## What each tool is for
15
15
 
16
16
  Every result is a reference (path, metadata, excerpt), never file contents; prose enters
17
- context only when you Read it. Costs: `map` is fixed-size, a `find` row is tens of tokens,
17
+ context only when you Read it. Costs: `map` is fixed-size, a `search` row is tens of tokens,
18
18
  and a `peek` stays flat however large the note is. Which tool fits is a property of the
19
19
  question:
20
20
 
21
21
  - A deterministic, factual answer over known fields — counts, filters, "which notes have
22
- X" — is SQL: `sense query`, a named query, or `find --where`. Enumerates every match;
22
+ X" — is SQL: `sense query`, a named query, or `search --where`. Enumerates every match;
23
23
  same result regardless of phrasing.
24
- - Locating notes by words in their prose is `find` — ranked lexical match. Results shift as
25
- phrasing shifts, and bare words AND-join (one absent word = zero rows): write
26
- `a OR b OR c` for any-word matching.
27
- - A conceptual question the notes phrase in different words is `find --semantic` (exists only
28
- on trees whose config enables `embed`): adds meaning-based candidates labeled `via: vector`.
29
- Conceptual similarity, not typo-tolerance; false positives are expected, labeled, and
30
- bounded by `--k`.
24
+ - Locating notes about something is `search` — one text through every engine the scope
25
+ has: word match (bare words AND-join — one absent word = zero lexical rows; write
26
+ `a OR b OR c` for any-word), link-graph expansion, and vector similarity, fused into one
27
+ ranked list. Read `via` per row: `match` rows contained your words; `vector`-only rows
28
+ did not — they are the "these words aren't in the tree; this is what's near in meaning"
29
+ signal. Vector rows are conceptual similarity, not typo-tolerance; false positives are
30
+ expected, labeled, and bounded by `--k`. `--lexical` skips vectors for one command when
31
+ word-presence is the question.
31
32
  - `map` answers "what is this tree" — fields, hub notes, recent changes — when the tree is
32
33
  unfamiliar.
33
34
  - `peek <path>` prices a file before you pay for it: outline with `[L143-162, ~380t]`
@@ -44,11 +45,12 @@ machine-parseable.
44
45
  ## Commands
45
46
 
46
47
  ```
47
- sense find "pricing OR billing OR invoicing" --where "f.status = 'active'" --k 10
48
+ sense search "pricing OR billing OR invoicing" --where "f.status = 'active'" --k 10
49
+ sense search "sourcing quotes" --preset raw # a named settings bundle from the config
48
50
  sense peek notes/pricing-model.md # a unique basename also works
49
51
  sense map
50
52
  sense query "<sql>" [params...] # ad-hoc SQL; ? binds positional args, count-checked
51
- sense <name> [params...] # named query or saved find from sense.config.json
53
+ sense <name> [params...] # named query or saved search from sense.config.json
52
54
  sense --list | status | rebuild | check
53
55
  ```
54
56
 
@@ -58,48 +60,48 @@ sense --list | status | rebuild | check
58
60
  rules apply to search commands you write into subagent briefs.
59
61
  - When a search misses, the recall levers are: OR-in synonyms and concrete instances (the
60
62
  index only knows the words in the files — a note about a specific tool rarely names its
61
- category), raise `--k` (a row costs tens of tokens), and on embed-enabled trees `--semantic`
62
- (matches meaning where term overlap fails). Each widening adds candidates and dilutes
63
- ranking, so the noise trade-off runs both ways.
63
+ category), raise `--k` (a row costs tens of tokens), and widen the scope (`--preset`, or
64
+ `--include` for an ad-hoc glob). Vector rows already cover the meaning-over-words gap by
65
+ default. Each widening adds candidates and dilutes ranking, so the noise trade-off runs
66
+ both ways.
64
67
  - A frontmatter query enumerates its matches deterministically; search ranks by term overlap,
65
68
  so results shift as phrasing shifts. Trade-off: a query needs a known field, search doesn't.
66
- - `find` fuses BM25 with link-graph expansion; the `via` column says what produced each row —
67
- `match` (terms hit), `link` (connected to notes that hit), `match+link` (both). With
68
- `--semantic`, `vector` joins the composition. The `lines` column, when set, points at the
69
- section that earned the row — the best-matching chunk on vector rows, the term cluster's
70
- section on large lexical notes — and is a direct `Read` range; null means the whole note
71
- is the reference.
72
- - `--where` takes any SQL condition against frontmatter alias `f` — not only field equality:
73
- `"f.status = 'active' AND has(f.tags, 'x')"`, `"f.path NOT LIKE 'generated/%'"`,
74
- `"f.created >= datetime(?)"`. A tree can declare a default scope in `sense.config.json`
75
- (`defaults.find.where`); an explicit `--where` replaces it, so `--where "1=1"` searches
76
- everything. `sense status` prints the active default.
69
+ - The `via` column says what produced each row — `match` (words hit), `link` (connected to
70
+ notes that hit), `vector` (near in meaning), and combinations. The `lines` column, when
71
+ set, points at the section that earned the row — the best-matching chunk on vector rows,
72
+ the term cluster's section on large lexical notes — and is a direct `Read` range; null
73
+ means the whole note is the reference.
74
+ - Scope comes from presets: bare `search` uses the config's `default` preset; `--preset
75
+ <name>` picks another (unknown names error, listing what's declared); `--include <glob>`
76
+ is an ad-hoc scope that replaces the preset's globs for one command. `--where` takes any
77
+ SQL condition against frontmatter alias `f` — not only field equality:
78
+ `"f.status = 'active' AND has(f.tags, 'x')"`, `"datetime(f.created) >= datetime(?)"` —
79
+ and filters within the scope. `sense status` shows every preset with its coverage.
77
80
  - `score` is a rank-fusion value: it ranks rows within one result set and is not comparable
78
81
  across queries, not a relevance magnitude — it encodes how many signals fired and at what
79
82
  rank, so a perfect lexical hit and a weak vector-only hit can read the same number. With
80
- `--semantic`, rows carry `similarity`: the cosine (-1 to 1) of the query against that
83
+ vectors active, rows carry `similarity`: the cosine (-1 to 1) of the query against that
81
84
  file's best-matching chunk — the same chunk the `lines` range points at. It orders vector
82
85
  evidence within a result set; the range it spans depends on the corpus and the embedding
83
86
  model, and compresses on small trees, where even a nonsense query has a moderately near
84
87
  neighbour somewhere. Compare similarities within a result set rather than against a fixed
85
88
  cutoff carried between trees.
86
- - Lexical `find` returns 0 rows when nothing matches, so it answers "is this in the tree at
87
- all". `--semantic` always returns up to `k` rows — nearest-neighbour search has a nearest
88
- neighbour for any input — so absence is a lexical question; `similarity` and the snippet
89
- are the evidence for judging whether a vector row is a real hit.
90
- - Besides SQL strings, a config entry can save a whole `find` invocation:
91
- `"hot": { "find": "pricing OR billing", "k": 20, "where": "...", "semantic": true }` runs
92
- as `sense hot` — the scenario's settings ride along with the name, so repeat runs need no
93
- flags. An invocation-level `--k`, `--where`, or `--semantic` overrides the saved value;
94
- `--list` marks these entries `(find)`.
95
- - `sense check` prepares every saved query (catching syntax and unknown-column errors), runs
96
- the ones taking no parameters, and prints row counts: a saved query returning 0 rows looks
97
- the same as a true empty result until something distinguishes them. Saved finds are probed
98
- lexically with k=1 (a bad `where` column or FTS5 syntax fails here, not mid-task); their
99
- semantic pass is never run by `check`, only checked against `features.embed`. For queries
100
- that encode invariants (a dead-link list, an unsupported-claims list — rows are
101
- violations), `checks: { "<name>": "empty" }` in the config inverts the meaning: `check`
102
- fails when the query returns rows, making it usable as a test suite rather than a linter.
89
+ - Absence evidence lives in the labels: `search --lexical` (or a semantic-off scope)
90
+ returns 0 rows when the words are nowhere in the tree. Default `search` always returns
91
+ up to `k` rows — nearest-neighbour search has a nearest neighbour for any input — so a
92
+ result of only `via: vector` rows IS the absence signal for the words themselves;
93
+ `similarity` and the snippet are the evidence for judging whether a vector row is a real
94
+ conceptual hit.
95
+ - Besides SQL strings, a config entry can save a whole search:
96
+ `"hot": { "search": "pricing OR billing", "preset": "raw", "k": 20 }` runs as
97
+ `sense hot` — the scenario's settings ride along with the name, so repeat runs need no
98
+ flags. An invocation-level `--preset`, `--k`, `--where`, or `--lexical` overrides the
99
+ saved value; `--list` marks these entries `(search)`.
100
+ - `sense check` prepares every saved query and probes every saved search lexically with
101
+ k=1, so a typo'd column, stale SQL, bad FTS5 syntax, or unknown preset fails at check
102
+ time instead of silently mid-task. It reports row counts; whether an empty result is
103
+ good or bad is the reader's judgment — a dead-link query returning rows means broken
104
+ citations to fix, and the agent reads that directly.
103
105
 
104
106
  ## SQL
105
107
 
@@ -122,7 +124,7 @@ sense query "SELECT j.value, COUNT(*) n FROM frontmatter, json_each(frontmatter.
122
124
  - Rank with `ORDER BY bm25(content, 10.0, 5.0, 1.0)` (title > summary > body); excerpt with
123
125
  `snippet(content, -1, '«', '»', '…', 10)`. snippet() re-tokenizes each matched doc and its
124
126
  cost grows superlinearly with doc size — measured ~10 s per query on a tree holding one
125
- 1 MB note. `find` bounds this itself (docs past 16 KB get an equivalent excerpt another
127
+ 1 MB note. `search` bounds this itself (docs past 16 KB get an equivalent excerpt another
126
128
  way); in hand-written SQL, guard it: `CASE WHEN length(text) <= 16384 THEN snippet(...)
127
129
  END`, or select `title`/`summary` instead of an excerpt.
128
130
  - Select `content.title`/`content.summary` (always exist, empty when absent) rather than
@@ -152,10 +154,11 @@ Worked traces: [EXAMPLES.md](EXAMPLES.md).
152
154
 
153
155
  - Missing CLI: `npm install -g sensemaking`. Missing config: `sense init` at the tree root.
154
156
  Discovery walks up from cwd; `--config <path>` overrides. Setting up or restructuring a
155
- tree (features, frontmatter conventions, note design) is the `sense-setup` skill.
156
- - `map` and `status` report feature state (`features: links, sections, rank · off: embed
157
- (features.embed)`). Invoking a capability whose feature is off is an error naming the
158
- config key to enable — nothing silently falls back.
157
+ tree (presets, frontmatter conventions, note design) is the `sense-setup` skill.
158
+ - `map` and `status` report each preset's coverage (files matched, embedded count) —
159
+ indexing derives from presets, so the coverage numbers are how you see what a config
160
+ actually indexes and embeds. A scope with fewer signals just uses fewer (a semantic-off
161
+ preset searches lexically); a saved search naming an unknown preset errors at `check`.
159
162
  - Save a query into `sense.config.json` only when it will be reused; run ad-hoc otherwise.
160
163
  - A one-line `summary:` per note is optional and pays twice: it appears in result rows and is a
161
164
  weighted search field. Date comparisons work for dates written as ISO 8601 (`2026-08-12`, or
@@ -0,0 +1,121 @@
1
+ # sense-setup: worked configurations
2
+
3
+ Four tree shapes, each with its config and the commands an agent actually runs. Field
4
+ names and folder names are illustrative — your tree defines its own.
5
+
6
+ ## A. Compiled wiki over immutable sources (the llm-wiki pattern)
7
+
8
+ `raw/` holds ingested sources — big, never hand-edited. `wiki/` holds agent-compiled
9
+ pages — linked, curated. The human drops sources and asks questions; the agent compiles
10
+ and cites.
11
+
12
+ ```json
13
+ {
14
+ "version": 3,
15
+ "presets": {
16
+ "default": { "include": ["wiki/**/*.md"], "k": 10 },
17
+ "raw": { "include": ["raw/**/*.md"], "k": 5, "semantic": false }
18
+ },
19
+ "queries": {
20
+ "uncompiled": "SELECT path, _mtime FROM frontmatter WHERE path LIKE 'raw/%' AND path NOT IN (SELECT dst FROM links WHERE dst IS NOT NULL) ORDER BY _mtime DESC",
21
+ "stubs": "SELECT path, _size FROM frontmatter WHERE path LIKE 'wiki/%' AND _size < 500 ORDER BY _size",
22
+ "dead-links": "SELECT src, target FROM links WHERE dst IS NULL ORDER BY src"
23
+ }
24
+ }
25
+ ```
26
+
27
+ ```
28
+ sense uncompiled # compile queue: raw files nothing cites yet
29
+ sense search "how does attention scale" # wiki only (default preset), vectors on
30
+ sense search "rotary embeddings" --preset raw # cite from sources; lexical, k=5
31
+ sense dead-links # rows are broken citations to fix
32
+ ```
33
+
34
+ What the shape buys: bare search never ranks raw noise above compiled pages; raw pays no
35
+ vector/link cost; the compile queue, stub list, and citation integrity are one saved
36
+ query each. The maintenance loop is `uncompiled` → write the wiki page citing its sources
37
+ → `dead-links` stays empty.
38
+
39
+ ## B. Nightly agent memory, consolidated (the dreaming pattern)
40
+
41
+ `memory/` accumulates small notes written at session end, each with `project`,
42
+ `created`, and `kind` (observation / steer / decision) frontmatter. A consolidation agent
43
+ runs periodically: prune, merge, surface contradictions for the human. Retired notes move
44
+ to `archive/` — still queryable, no longer embedded or ranked.
45
+
46
+ ```json
47
+ {
48
+ "version": 3,
49
+ "presets": {
50
+ "default": { "include": ["memory/**/*.md"], "k": 10 },
51
+ "archive": { "include": ["archive/**/*.md"], "k": 10, "semantic": false }
52
+ },
53
+ "queries": {
54
+ "project": "SELECT path, kind, created, title FROM frontmatter WHERE project = ? ORDER BY created DESC",
55
+ "steers": "SELECT path, created, title FROM frontmatter WHERE kind = 'steer' AND project = ? ORDER BY created",
56
+ "retirement": "SELECT path, project, created FROM frontmatter WHERE datetime(created) < datetime('now','-90 day') AND path NOT IN (SELECT dst FROM links WHERE dst IS NOT NULL)",
57
+ "unfiled": "SELECT path FROM frontmatter WHERE project IS NULL"
58
+ }
59
+ }
60
+ ```
61
+
62
+ ```
63
+ sense project acme-app # one project's notes, newest first
64
+ sense search "prefers terse commit messages" --k 5
65
+ → memory/acme-app/2026-08-02-commits.md via: match
66
+ → memory/acme-app/2026-06-11-style.md via: vector similarity: 0.71 # near-duplicate → merge candidate
67
+ sense steers acme-app # oldest first: does a new steer override an old one?
68
+ sense retirement # old + uncited → move to archive/
69
+ sense unfiled # rows are notes missing a project — file them
70
+ ```
71
+
72
+ Presets are structural (live vs archived); per-project filtering is metadata (`project = ?`)
73
+ — one tree serves every project. Semantic search over the memory preset is the
74
+ near-duplicate detector: search a new note's own summary and read `similarity` within the
75
+ results. Whether an old steer was overridden is a question for the human — found by
76
+ search, never decided by it.
77
+
78
+ ## C. Evidence corpus: claims trace to sources
79
+
80
+ `sources/` (immutable imports), `notes/` (one reading note per source), `reviews/`
81
+ (synthesis whose claims must cite notes). The same shape fits incident reports and
82
+ postmortems, user research and findings, due diligence and memos.
83
+
84
+ ```json
85
+ {
86
+ "version": 3,
87
+ "presets": {
88
+ "default": { "include": ["reviews/**/*.md", "notes/**/*.md"], "k": 10 },
89
+ "source": { "include": ["sources/**/*.md"], "k": 5, "semantic": false }
90
+ },
91
+ "queries": {
92
+ "unsupported": "SELECT path, title FROM frontmatter WHERE path LIKE 'reviews/%' AND path NOT IN (SELECT src FROM links WHERE dst LIKE 'notes/%')",
93
+ "unread": "SELECT path FROM frontmatter WHERE path LIKE 'sources/%' AND path NOT IN (SELECT dst FROM links WHERE dst IS NOT NULL)",
94
+ "by-topic": "SELECT path, title FROM frontmatter WHERE path NOT LIKE 'sources/%' AND has(topics, ?) ORDER BY path"
95
+ }
96
+ }
97
+ ```
98
+
99
+ ```
100
+ sense search "replication failures in priming studies" # reviews + notes; sources never dilute
101
+ sense search "the claim's exact phrasing" --preset source # citation pull on demand
102
+ sense unsupported # rows are synthesis claims with no note behind them
103
+ sense unread # the reading queue: sources no note cites
104
+ ```
105
+
106
+ ## D. A plain vault: zero configuration
107
+
108
+ Someone else's Obsidian vault, heterogeneous, no structure worth declaring — the
109
+ `sense init` starter untouched. The workflow is discovery:
110
+
111
+ ```
112
+ sense map # fields in use, hub notes, recent changes
113
+ sense search "dataview queries" # words + links + meaning, one ranked list
114
+ sense query "SELECT j.value AS tag, COUNT(*) n FROM frontmatter, json_each(frontmatter.tags) j GROUP BY j.value ORDER BY n DESC LIMIT 20"
115
+ sense peek "Plugins/dataview.md" # outline + links before reading
116
+ ```
117
+
118
+ Presets earn their place only when a tree has parts deserving different treatment; a tree
119
+ that is one kind of thing needs none of the vocabulary above. On a big vault, raise
120
+ `default`'s `k` and read `lines` ranges instead of whole files — or start from the
121
+ starter's `large` preset.