sensemaking 0.8.0 → 0.9.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +48 -39
- package/dist/cjs/cli/check.js +24 -43
- package/dist/cjs/cli/check.js.map +1 -1
- package/dist/cjs/cli/index.js +2 -2
- package/dist/cjs/cli/index.js.map +1 -1
- package/dist/cjs/cli/named.js +18 -8
- package/dist/cjs/cli/named.js.map +1 -1
- package/dist/cjs/cli/search.d.cts +3 -0
- package/dist/cjs/cli/search.d.ts +3 -0
- package/dist/cjs/cli/{find.js → search.js} +7 -5
- package/dist/cjs/cli/search.js.map +1 -0
- package/dist/cjs/cli/shared.js.map +1 -1
- package/dist/cjs/cli/status.js +29 -12
- package/dist/cjs/cli/status.js.map +1 -1
- package/dist/cjs/cli/types.d.cts +3 -1
- package/dist/cjs/cli/types.d.ts +3 -1
- package/dist/cjs/cli.js +25 -5
- package/dist/cjs/cli.js.map +1 -1
- package/dist/cjs/commands.d.cts +11 -2
- package/dist/cjs/commands.d.ts +11 -2
- package/dist/cjs/commands.js +86 -21
- package/dist/cjs/commands.js.map +1 -1
- package/dist/cjs/config.d.cts +41 -14
- package/dist/cjs/config.d.ts +41 -14
- package/dist/cjs/config.js +448 -160
- package/dist/cjs/config.js.map +1 -1
- package/dist/cjs/db.d.cts +1 -1
- package/dist/cjs/db.d.ts +1 -1
- package/dist/cjs/db.js +208 -62
- package/dist/cjs/db.js.map +1 -1
- package/dist/cjs/errors.d.cts +1 -1
- package/dist/cjs/errors.d.ts +1 -1
- package/dist/cjs/errors.js.map +1 -1
- package/dist/cjs/features/embed.js +12 -2
- package/dist/cjs/features/embed.js.map +1 -1
- package/dist/cjs/features/types.d.cts +2 -1
- package/dist/cjs/features/types.d.ts +2 -1
- package/dist/cjs/index.d.cts +3 -3
- package/dist/cjs/index.d.ts +3 -3
- package/dist/cjs/index.js +6 -3
- package/dist/cjs/index.js.map +1 -1
- package/dist/cjs/output.d.cts +15 -0
- package/dist/cjs/output.d.ts +15 -0
- package/dist/cjs/output.js +26 -5
- package/dist/cjs/output.js.map +1 -1
- package/dist/cjs/scan.d.cts +3 -0
- package/dist/cjs/scan.d.ts +3 -0
- package/dist/cjs/scan.js +84 -4
- package/dist/cjs/scan.js.map +1 -1
- package/dist/esm/cli/check.js +20 -41
- package/dist/esm/cli/check.js.map +1 -1
- package/dist/esm/cli/index.js +1 -1
- package/dist/esm/cli/index.js.map +1 -1
- package/dist/esm/cli/named.js +17 -9
- package/dist/esm/cli/named.js.map +1 -1
- package/dist/esm/cli/search.d.ts +3 -0
- package/dist/esm/cli/search.js +16 -0
- package/dist/esm/cli/search.js.map +1 -0
- package/dist/esm/cli/shared.js +1 -1
- package/dist/esm/cli/shared.js.map +1 -1
- package/dist/esm/cli/status.js +10 -10
- package/dist/esm/cli/status.js.map +1 -1
- package/dist/esm/cli/types.d.ts +3 -1
- package/dist/esm/cli/types.js.map +1 -1
- package/dist/esm/cli.js +20 -4
- package/dist/esm/cli.js.map +1 -1
- package/dist/esm/commands.d.ts +11 -2
- package/dist/esm/commands.js +76 -24
- package/dist/esm/commands.js.map +1 -1
- package/dist/esm/config.d.ts +41 -14
- package/dist/esm/config.js +354 -114
- package/dist/esm/config.js.map +1 -1
- package/dist/esm/db.d.ts +1 -1
- package/dist/esm/db.js +53 -6
- package/dist/esm/db.js.map +1 -1
- package/dist/esm/errors.d.ts +1 -1
- package/dist/esm/errors.js.map +1 -1
- package/dist/esm/features/embed.js +12 -2
- package/dist/esm/features/embed.js.map +1 -1
- package/dist/esm/features/types.d.ts +2 -1
- package/dist/esm/features/types.js.map +1 -1
- package/dist/esm/index.d.ts +3 -3
- package/dist/esm/index.js +1 -1
- package/dist/esm/index.js.map +1 -1
- package/dist/esm/output.d.ts +15 -0
- package/dist/esm/output.js +19 -5
- package/dist/esm/output.js.map +1 -1
- package/dist/esm/scan.d.ts +3 -0
- package/dist/esm/scan.js +32 -4
- package/dist/esm/scan.js.map +1 -1
- package/package.json +9 -3
- package/schema.json +74 -54
- package/skills/sense/EXAMPLES.md +8 -6
- package/skills/sense/SKILL.md +51 -48
- package/skills/sense-setup/EXAMPLES.md +121 -0
- package/skills/sense-setup/SKILL.md +71 -66
- package/dist/cjs/cli/find.d.cts +0 -3
- package/dist/cjs/cli/find.d.ts +0 -3
- package/dist/cjs/cli/find.js.map +0 -1
- package/dist/esm/cli/find.d.ts +0 -3
- package/dist/esm/cli/find.js +0 -14
- package/dist/esm/cli/find.js.map +0 -1
package/schema.json
CHANGED
|
@@ -4,7 +4,7 @@
|
|
|
4
4
|
"description": "Config for sense -- SQL over the frontmatter, content, links, and structure of a markdown tree.",
|
|
5
5
|
"type": "object",
|
|
6
6
|
"additionalProperties": false,
|
|
7
|
-
"required": ["
|
|
7
|
+
"required": ["presets", "queries"],
|
|
8
8
|
"properties": {
|
|
9
9
|
"$schema": {
|
|
10
10
|
"type": "string",
|
|
@@ -12,31 +12,54 @@
|
|
|
12
12
|
},
|
|
13
13
|
"version": {
|
|
14
14
|
"type": "integer",
|
|
15
|
-
"enum": [1, 2],
|
|
15
|
+
"enum": [1, 2, 3],
|
|
16
16
|
"description": "Config format version. Older versions are auto-migrated (and the file rewritten) on load; a version newer than this sense build supports makes it exit with an error rather than misinterpret the file. Omit to default to 1."
|
|
17
17
|
},
|
|
18
|
-
"
|
|
18
|
+
"presets": {
|
|
19
19
|
"type": "object",
|
|
20
|
-
"
|
|
21
|
-
"
|
|
22
|
-
"
|
|
23
|
-
|
|
24
|
-
"
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
"
|
|
28
|
-
|
|
20
|
+
"description": "File selection, index-time, and named search defaults. A preset is self-contained: no inheritance between presets, and no ordering -- presets may overlap freely (they are views, not partitions), and a preset whose globs match nothing is valid. A file is indexed iff any preset's include/exclude covers it (union across presets); it is embedded iff any covering preset has semantic on. `default` is used when a command names no preset, and must be declared. `sense init` generates `default` plus one `large` example.",
|
|
21
|
+
"minProperties": 1,
|
|
22
|
+
"additionalProperties": {
|
|
23
|
+
"type": "object",
|
|
24
|
+
"additionalProperties": false,
|
|
25
|
+
"required": ["include"],
|
|
26
|
+
"properties": {
|
|
27
|
+
"include": {
|
|
28
|
+
"type": "array",
|
|
29
|
+
"items": { "type": "string" },
|
|
30
|
+
"minItems": 1,
|
|
31
|
+
"description": "Glob patterns (globby syntax), resolved relative to this config file's own directory -- never the invocation cwd. Required for every preset, including `default`."
|
|
32
|
+
},
|
|
33
|
+
"exclude": {
|
|
34
|
+
"type": "array",
|
|
35
|
+
"items": { "type": "string" },
|
|
36
|
+
"minItems": 1,
|
|
37
|
+
"description": "Glob patterns excluded from this preset's `include`, tsconfig-standard semantics."
|
|
38
|
+
},
|
|
39
|
+
"k": {
|
|
40
|
+
"type": "integer",
|
|
41
|
+
"minimum": 1,
|
|
42
|
+
"description": "Default result count for `search` when this preset is in effect (named, or `default` when none is). A --k on the invocation, or a saved search's own `k`, overrides it. Defaults to 10 when the preset omits it too."
|
|
43
|
+
},
|
|
44
|
+
"semantic": {
|
|
45
|
+
"type": "boolean",
|
|
46
|
+
"description": "Vector participation for files this preset covers. Absent or true means on (the default); only ever meaningfully written as false. A file covered only by semantic-false presets is indexed but never embedded, so it searches lexically automatically -- no error, just fewer signals."
|
|
47
|
+
},
|
|
48
|
+
"where": {
|
|
49
|
+
"type": "string",
|
|
50
|
+
"description": "Standing SQL condition against frontmatter alias `f`, applied when `search` runs under this preset without an explicit --where. An explicit --where (or a saved search's own `where`) replaces it rather than ANDing with it, so a caller can always widen back to the preset's whole scope."
|
|
51
|
+
}
|
|
29
52
|
}
|
|
30
53
|
}
|
|
31
54
|
},
|
|
32
55
|
"features": {
|
|
33
56
|
"type": "object",
|
|
34
57
|
"additionalProperties": false,
|
|
35
|
-
"description": "
|
|
58
|
+
"description": "Global feature defaults. Absent block or key means enabled. `embed` is not a member of this block -- vector participation is controlled per preset by `semantic` (see \"presets\" and the top-level \"embed\" block for provider settings). Toggling a feature rebuilds the cache. Commands degrade rather than fail: `search` without links is BM25-only, `map` without rank omits hubs, `peek` without sections omits the outline.",
|
|
36
59
|
"properties": {
|
|
37
60
|
"links": {
|
|
38
61
|
"type": "boolean",
|
|
39
|
-
"description": "Extract wikilinks and relative markdown links into the `links` table (src, target as written, dst resolved path or NULL for dead links; an ambiguous basename resolves to the lexicographically first match). Powers backlinks, graph expansion in `
|
|
62
|
+
"description": "Extract wikilinks and relative markdown links into the `links` table (src, target as written, dst resolved path or NULL for dead links; an ambiguous basename resolves to the lexicographically first match). Powers backlinks, graph expansion in `search`, and `peek`'s link lists."
|
|
40
63
|
},
|
|
41
64
|
"sections": {
|
|
42
65
|
"type": "boolean",
|
|
@@ -45,63 +68,60 @@
|
|
|
45
68
|
"rank": {
|
|
46
69
|
"type": "boolean",
|
|
47
70
|
"description": "PageRank over resolved links into `frontmatter._rank` at reconcile -- a static importance prior. Powers `map`'s hub list; usable in any ORDER BY. Requires `links`."
|
|
48
|
-
},
|
|
49
|
-
"embed": {
|
|
50
|
-
"description": "Semantic vectors, opt-in (absent = off; most trees don't need them). When on, vectors are computed and kept fresh at reconcile; expansion runs only on `find --semantic` invocations -- default `find` results are unchanged. `true` = defaults (static type, minishlab/potion-retrieval-32M, downloaded to ~/.cache/sensemaking on first use). Object form: `model` (Hugging Face id or local path), `type` (`static` = built-in pure-JS Model2Vec loader; `api` = OpenAI-compatible POST /embeddings), `url` (api base, e.g. http://localhost:11434/v1), `key` (name of the env var holding the bearer token). Changing model or type rebuilds the cache like a feature toggle.",
|
|
51
|
-
"oneOf": [
|
|
52
|
-
{ "type": "boolean" },
|
|
53
|
-
{
|
|
54
|
-
"type": "object",
|
|
55
|
-
"additionalProperties": false,
|
|
56
|
-
"properties": {
|
|
57
|
-
"model": { "type": "string" },
|
|
58
|
-
"type": { "type": "string", "enum": ["static", "api"] },
|
|
59
|
-
"url": { "type": "string" },
|
|
60
|
-
"key": { "type": "string" }
|
|
61
|
-
}
|
|
62
|
-
}
|
|
63
|
-
]
|
|
64
71
|
}
|
|
65
72
|
}
|
|
66
73
|
},
|
|
67
|
-
"
|
|
74
|
+
"embed": {
|
|
68
75
|
"type": "object",
|
|
69
76
|
"additionalProperties": false,
|
|
70
|
-
"description": "
|
|
77
|
+
"description": "Embed provider settings only -- presence alone does not enable embedding. Vector participation is decided per preset by that preset's `semantic` field; this block only says which model/provider to use when at least one preset wants vectors. `model` (Hugging Face id or local path, default minishlab/potion-retrieval-32M), `type` (`static` = built-in pure-JS Model2Vec loader, default; `api` = OpenAI-compatible POST /embeddings), `url` (api base, e.g. http://localhost:11434/v1), `key` (name of the env var holding the bearer token). Changing model or type rebuilds the cache like a feature toggle.",
|
|
71
78
|
"properties": {
|
|
72
|
-
"
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
"properties": {
|
|
77
|
-
"where": {
|
|
78
|
-
"type": "string",
|
|
79
|
-
"description": "SQL condition against frontmatter alias `f`, e.g. \"f.type != 'raw'\". Applied when `find` runs without --where; an explicit --where replaces it (so `--where \"1=1\"` searches the whole tree). Reported by `sense status`."
|
|
80
|
-
}
|
|
81
|
-
}
|
|
82
|
-
}
|
|
79
|
+
"model": { "type": "string" },
|
|
80
|
+
"type": { "type": "string", "enum": ["static", "api"] },
|
|
81
|
+
"url": { "type": "string" },
|
|
82
|
+
"key": { "type": "string" }
|
|
83
83
|
}
|
|
84
84
|
},
|
|
85
|
-
"checks": {
|
|
86
|
-
"type": "object",
|
|
87
|
-
"description": "Assertions over saved queries, for trees whose queries encode invariants: `\"checks\": { \"dead-links\": \"empty\" }` makes `sense check` fail (exit 1) when that query returns rows. Without an entry, `check` reports zero-row queries as `ok, but returns 0 rows` -- informational, since an ordinary query returning nothing is sometimes a bug and sometimes just an empty result. Keys must name saved queries.",
|
|
88
|
-
"additionalProperties": { "type": "string", "enum": ["empty"] }
|
|
89
|
-
},
|
|
90
85
|
"queries": {
|
|
91
86
|
"type": "object",
|
|
92
|
-
"description": "Named queries runnable as `sense <name> [params...]`,
|
|
87
|
+
"description": "Named queries runnable as `sense <name> [params...]`, one of three shapes: a raw SQL string, `{ sql }`, or a saved search. SQL (string or `{ sql }`): `?` placeholders bind to CLI positional args in order; deterministic and enumerating, including raw FTS5 `MATCH` for word search under your own SQL -- \"0 rows = not in the tree\" lives here. Tables: `frontmatter` (one row per file, one column per discovered frontmatter key, plus `path`/`_mtime`/`_size`/`_rank`), `content` (FTS5: `title`, `summary`, `text`, `path`), `links` (`src`, `target`, `dst`), `sections` (`path`, `idx`, `level`, `heading`, `start_line`, `end_line`, `tokens`), and `preset_files` (`preset`, `path`) -- which presets cover which files. `has(field, value)`: array membership on a JSON-array field, substring match on a string (so has(f.status, 'active') also matches 'inactive'), false on NULL. Exact matches: `=` for scalars, `EXISTS (SELECT 1 FROM json_each(f.tags) WHERE value = ?)` for array members. Canonical query: `SELECT f.path, content.title, content.summary, CASE WHEN length(content.text) <= 16384 THEN snippet(content, -1, '«', '»', '…', 10) END AS hit FROM frontmatter f JOIN content ON content.path = f.path WHERE content MATCH ? ORDER BY bm25(content, 10.0, 5.0, 1.0) LIMIT 10`. The CASE bounds snippet(), which re-tokenizes each matched document and costs seconds per query once a tree holds a megabyte-scale note; `search` applies the same bound internally. Saved search object: one text driving every engine the scoped preset has -- word match, links, vectors -- fused into one ranked list, `via` labeling which engine produced each row; `search` text must be non-empty (a saved query saves a question -- a scope without one is just flags). `preset` names one declared preset (defaults to `default`); `include` is an ad hoc glob scope that replaces the preset's include/exclude entirely, same as `search --include`; `where` and `k` behave like the `search` command's flags. `semantic: false` opts a query down to lexical only -- the rare precision opt-out, not a gate (semantic participation otherwise follows the scoped preset's `semantic` automatically). `sense <name>` behaves like `sense search <search> [--preset] [--include] [--where] [--k] [--lexical]` with zero flags; it takes no positional parameters. `sense check` probes every saved query/search (lexically, k=1 for searches) so a broken one fails loudly at check time -- it makes no assertion about the result itself; a returned row set is the reader's judgment. Reserved frontmatter keys: `path`, `_mtime`, `_size`, `_rank`, `content`, `links`, `sections`. Reserved query names (unreachable as subcommands, any shape): `init`, `query`, `search`, `find`, `map`, `peek`, `watch`, `status`, `rebuild`, `check`.",
|
|
93
88
|
"additionalProperties": {
|
|
94
89
|
"oneOf": [
|
|
95
90
|
{ "type": "string" },
|
|
96
91
|
{
|
|
97
92
|
"type": "object",
|
|
98
93
|
"additionalProperties": false,
|
|
99
|
-
"required": ["
|
|
94
|
+
"required": ["sql"],
|
|
100
95
|
"properties": {
|
|
101
|
-
"
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
96
|
+
"sql": { "type": "string", "description": "Raw SQL, same as the string shape -- `?` placeholders bind to CLI positional args in order." }
|
|
97
|
+
}
|
|
98
|
+
},
|
|
99
|
+
{
|
|
100
|
+
"type": "object",
|
|
101
|
+
"additionalProperties": false,
|
|
102
|
+
"required": ["search"],
|
|
103
|
+
"properties": {
|
|
104
|
+
"search": {
|
|
105
|
+
"type": "string",
|
|
106
|
+
"minLength": 1,
|
|
107
|
+
"description": "One text driving every engine the scoped preset has -- word match, links, vectors -- fused into one ranked list. Same syntax as `sense search \"<text>\"`. Must be non-empty."
|
|
108
|
+
},
|
|
109
|
+
"preset": {
|
|
110
|
+
"type": "string",
|
|
111
|
+
"description": "Preset name to search under. Defaults to `default`. An unknown name is an error listing declared presets."
|
|
112
|
+
},
|
|
113
|
+
"include": {
|
|
114
|
+
"type": "array",
|
|
115
|
+
"items": { "type": "string" },
|
|
116
|
+
"minItems": 1,
|
|
117
|
+
"description": "Ad hoc glob scope, same as `search --include`. Replaces the preset's include/exclude entirely rather than layering on top of it."
|
|
118
|
+
},
|
|
119
|
+
"k": { "type": "integer", "minimum": 1, "description": "Result count, same as `search --k`. Defaults to the preset's own `k`, then 10. A --k on the invocation overrides this." },
|
|
120
|
+
"where": { "type": "string", "description": "SQL condition against frontmatter alias `f`, same as `search --where`. A --where on the invocation replaces this rather than ANDing with it." },
|
|
121
|
+
"semantic": {
|
|
122
|
+
"type": "boolean",
|
|
123
|
+
"description": "false opts this query down to lexical-only, same as `search --lexical` -- the rare precision opt-out. Semantic participation otherwise follows the scoped preset's `semantic` automatically, so true is redundant with the default."
|
|
124
|
+
}
|
|
105
125
|
}
|
|
106
126
|
}
|
|
107
127
|
]
|
package/skills/sense/EXAMPLES.md
CHANGED
|
@@ -6,7 +6,7 @@ the filesystem, on the paths that earned it.
|
|
|
6
6
|
## A. "Do the notes say anything about X?"
|
|
7
7
|
|
|
8
8
|
```
|
|
9
|
-
sense
|
|
9
|
+
sense search "pricing OR billing OR invoicing" --k 10 --format json
|
|
10
10
|
```
|
|
11
11
|
|
|
12
12
|
```json
|
|
@@ -79,7 +79,7 @@ sense query "SELECT DISTINCT type FROM frontmatter" # what a field's values a
|
|
|
79
79
|
## F. "The notes say it in different words"
|
|
80
80
|
|
|
81
81
|
```
|
|
82
|
-
sense
|
|
82
|
+
sense search "children dying from poor nutrition" --k 3 --format json
|
|
83
83
|
```
|
|
84
84
|
|
|
85
85
|
```json
|
|
@@ -90,7 +90,9 @@ sense find "children dying from poor nutrition" --semantic --k 3 --format json
|
|
|
90
90
|
```
|
|
91
91
|
|
|
92
92
|
A `via: "vector"` row never contained the terms — it is semantically near them; `similarity` is
|
|
93
|
-
the cosine against the chunk `lines` names, a direct `Read` range.
|
|
93
|
+
the cosine against the chunk `lines` names, a direct `Read` range. Vector rows appear whenever
|
|
94
|
+
the scope's preset has semantic on (the default); a result of only vector rows means the words
|
|
95
|
+
themselves are nowhere in the scope.
|
|
94
96
|
|
|
95
97
|
## Consequences
|
|
96
98
|
|
|
@@ -98,9 +100,9 @@ the cosine against the chunk `lines` names, a direct `Read` range. Only on trees
|
|
|
98
100
|
|---|---|---|
|
|
99
101
|
| `SELECT text FROM content` | returns the tree's entire prose | `snippet(content, -1, '«', '»', '…', 10)` excerpts the match |
|
|
100
102
|
| `snippet()` on a tree holding a megabyte-scale note | re-tokenizes the whole document per matched row: seconds per query | bound it: `CASE WHEN length(content.text) <= 16384 THEN snippet(...) END`, or select `summary` instead |
|
|
101
|
-
| `sense
|
|
102
|
-
| `sense
|
|
103
|
-
| `sense
|
|
103
|
+
| `sense search "pricing"` | lexically matches only that word's stem (vector rows still widen by meaning) | OR-in synonyms and instances: `"pricing OR billing OR invoicing"` |
|
|
104
|
+
| `sense search "pricing model details"` | bare words AND-join; one absent word = zero lexical rows | OR the words, or quote an exact phrase |
|
|
105
|
+
| `sense search "customer-facing OR on-site"` | bare punctuation is FTS5 syntax (`-` reads as a column filter) | double-quote: `"customer-facing" OR "on-site"` |
|
|
104
106
|
| `Read` of a large file for one section | costs the whole file | `peek`, then `Read` the line range |
|
|
105
107
|
| row queries without `LIMIT` | unbounded output (aggregates are already bounded) | `LIMIT n` |
|
|
106
108
|
| saving one-off queries to config | config churn | ad-hoc `sense query`; save reusable views |
|
package/skills/sense/SKILL.md
CHANGED
|
@@ -14,20 +14,21 @@ storage; `map` and `status` report which are on.
|
|
|
14
14
|
## What each tool is for
|
|
15
15
|
|
|
16
16
|
Every result is a reference (path, metadata, excerpt), never file contents; prose enters
|
|
17
|
-
context only when you Read it. Costs: `map` is fixed-size, a `
|
|
17
|
+
context only when you Read it. Costs: `map` is fixed-size, a `search` row is tens of tokens,
|
|
18
18
|
and a `peek` stays flat however large the note is. Which tool fits is a property of the
|
|
19
19
|
question:
|
|
20
20
|
|
|
21
21
|
- A deterministic, factual answer over known fields — counts, filters, "which notes have
|
|
22
|
-
X" — is SQL: `sense query`, a named query, or `
|
|
22
|
+
X" — is SQL: `sense query`, a named query, or `search --where`. Enumerates every match;
|
|
23
23
|
same result regardless of phrasing.
|
|
24
|
-
- Locating notes
|
|
25
|
-
|
|
26
|
-
`a OR b OR c` for any-word
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
bounded by `--k`.
|
|
24
|
+
- Locating notes about something is `search` — one text through every engine the scope
|
|
25
|
+
has: word match (bare words AND-join — one absent word = zero lexical rows; write
|
|
26
|
+
`a OR b OR c` for any-word), link-graph expansion, and vector similarity, fused into one
|
|
27
|
+
ranked list. Read `via` per row: `match` rows contained your words; `vector`-only rows
|
|
28
|
+
did not — they are the "these words aren't in the tree; this is what's near in meaning"
|
|
29
|
+
signal. Vector rows are conceptual similarity, not typo-tolerance; false positives are
|
|
30
|
+
expected, labeled, and bounded by `--k`. `--lexical` skips vectors for one command when
|
|
31
|
+
word-presence is the question.
|
|
31
32
|
- `map` answers "what is this tree" — fields, hub notes, recent changes — when the tree is
|
|
32
33
|
unfamiliar.
|
|
33
34
|
- `peek <path>` prices a file before you pay for it: outline with `[L143-162, ~380t]`
|
|
@@ -44,11 +45,12 @@ machine-parseable.
|
|
|
44
45
|
## Commands
|
|
45
46
|
|
|
46
47
|
```
|
|
47
|
-
sense
|
|
48
|
+
sense search "pricing OR billing OR invoicing" --where "f.status = 'active'" --k 10
|
|
49
|
+
sense search "sourcing quotes" --preset raw # a named settings bundle from the config
|
|
48
50
|
sense peek notes/pricing-model.md # a unique basename also works
|
|
49
51
|
sense map
|
|
50
52
|
sense query "<sql>" [params...] # ad-hoc SQL; ? binds positional args, count-checked
|
|
51
|
-
sense <name> [params...] # named query or saved
|
|
53
|
+
sense <name> [params...] # named query or saved search from sense.config.json
|
|
52
54
|
sense --list | status | rebuild | check
|
|
53
55
|
```
|
|
54
56
|
|
|
@@ -58,48 +60,48 @@ sense --list | status | rebuild | check
|
|
|
58
60
|
rules apply to search commands you write into subagent briefs.
|
|
59
61
|
- When a search misses, the recall levers are: OR-in synonyms and concrete instances (the
|
|
60
62
|
index only knows the words in the files — a note about a specific tool rarely names its
|
|
61
|
-
category), raise `--k` (a row costs tens of tokens), and
|
|
62
|
-
|
|
63
|
-
ranking, so the noise trade-off runs
|
|
63
|
+
category), raise `--k` (a row costs tens of tokens), and widen the scope (`--preset`, or
|
|
64
|
+
`--include` for an ad-hoc glob). Vector rows already cover the meaning-over-words gap by
|
|
65
|
+
default. Each widening adds candidates and dilutes ranking, so the noise trade-off runs
|
|
66
|
+
both ways.
|
|
64
67
|
- A frontmatter query enumerates its matches deterministically; search ranks by term overlap,
|
|
65
68
|
so results shift as phrasing shifts. Trade-off: a query needs a known field, search doesn't.
|
|
66
|
-
-
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
section
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
69
|
+
- The `via` column says what produced each row — `match` (words hit), `link` (connected to
|
|
70
|
+
notes that hit), `vector` (near in meaning), and combinations. The `lines` column, when
|
|
71
|
+
set, points at the section that earned the row — the best-matching chunk on vector rows,
|
|
72
|
+
the term cluster's section on large lexical notes — and is a direct `Read` range; null
|
|
73
|
+
means the whole note is the reference.
|
|
74
|
+
- Scope comes from presets: bare `search` uses the config's `default` preset; `--preset
|
|
75
|
+
<name>` picks another (unknown names error, listing what's declared); `--include <glob>`
|
|
76
|
+
is an ad-hoc scope that replaces the preset's globs for one command. `--where` takes any
|
|
77
|
+
SQL condition against frontmatter alias `f` — not only field equality:
|
|
78
|
+
`"f.status = 'active' AND has(f.tags, 'x')"`, `"datetime(f.created) >= datetime(?)"` —
|
|
79
|
+
and filters within the scope. `sense status` shows every preset with its coverage.
|
|
77
80
|
- `score` is a rank-fusion value: it ranks rows within one result set and is not comparable
|
|
78
81
|
across queries, not a relevance magnitude — it encodes how many signals fired and at what
|
|
79
82
|
rank, so a perfect lexical hit and a weak vector-only hit can read the same number. With
|
|
80
|
-
|
|
83
|
+
vectors active, rows carry `similarity`: the cosine (-1 to 1) of the query against that
|
|
81
84
|
file's best-matching chunk — the same chunk the `lines` range points at. It orders vector
|
|
82
85
|
evidence within a result set; the range it spans depends on the corpus and the embedding
|
|
83
86
|
model, and compresses on small trees, where even a nonsense query has a moderately near
|
|
84
87
|
neighbour somewhere. Compare similarities within a result set rather than against a fixed
|
|
85
88
|
cutoff carried between trees.
|
|
86
|
-
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
fails when the query returns rows, making it usable as a test suite rather than a linter.
|
|
89
|
+
- Absence evidence lives in the labels: `search --lexical` (or a semantic-off scope)
|
|
90
|
+
returns 0 rows when the words are nowhere in the tree. Default `search` always returns
|
|
91
|
+
up to `k` rows — nearest-neighbour search has a nearest neighbour for any input — so a
|
|
92
|
+
result of only `via: vector` rows IS the absence signal for the words themselves;
|
|
93
|
+
`similarity` and the snippet are the evidence for judging whether a vector row is a real
|
|
94
|
+
conceptual hit.
|
|
95
|
+
- Besides SQL strings, a config entry can save a whole search:
|
|
96
|
+
`"hot": { "search": "pricing OR billing", "preset": "raw", "k": 20 }` runs as
|
|
97
|
+
`sense hot` — the scenario's settings ride along with the name, so repeat runs need no
|
|
98
|
+
flags. An invocation-level `--preset`, `--k`, `--where`, or `--lexical` overrides the
|
|
99
|
+
saved value; `--list` marks these entries `(search)`.
|
|
100
|
+
- `sense check` prepares every saved query and probes every saved search lexically with
|
|
101
|
+
k=1, so a typo'd column, stale SQL, bad FTS5 syntax, or unknown preset fails at check
|
|
102
|
+
time instead of silently mid-task. It reports row counts; whether an empty result is
|
|
103
|
+
good or bad is the reader's judgment — a dead-link query returning rows means broken
|
|
104
|
+
citations to fix, and the agent reads that directly.
|
|
103
105
|
|
|
104
106
|
## SQL
|
|
105
107
|
|
|
@@ -122,7 +124,7 @@ sense query "SELECT j.value, COUNT(*) n FROM frontmatter, json_each(frontmatter.
|
|
|
122
124
|
- Rank with `ORDER BY bm25(content, 10.0, 5.0, 1.0)` (title > summary > body); excerpt with
|
|
123
125
|
`snippet(content, -1, '«', '»', '…', 10)`. snippet() re-tokenizes each matched doc and its
|
|
124
126
|
cost grows superlinearly with doc size — measured ~10 s per query on a tree holding one
|
|
125
|
-
1 MB note. `
|
|
127
|
+
1 MB note. `search` bounds this itself (docs past 16 KB get an equivalent excerpt another
|
|
126
128
|
way); in hand-written SQL, guard it: `CASE WHEN length(text) <= 16384 THEN snippet(...)
|
|
127
129
|
END`, or select `title`/`summary` instead of an excerpt.
|
|
128
130
|
- Select `content.title`/`content.summary` (always exist, empty when absent) rather than
|
|
@@ -152,10 +154,11 @@ Worked traces: [EXAMPLES.md](EXAMPLES.md).
|
|
|
152
154
|
|
|
153
155
|
- Missing CLI: `npm install -g sensemaking`. Missing config: `sense init` at the tree root.
|
|
154
156
|
Discovery walks up from cwd; `--config <path>` overrides. Setting up or restructuring a
|
|
155
|
-
tree (
|
|
156
|
-
- `map` and `status` report
|
|
157
|
-
|
|
158
|
-
|
|
157
|
+
tree (presets, frontmatter conventions, note design) is the `sense-setup` skill.
|
|
158
|
+
- `map` and `status` report each preset's coverage (files matched, embedded count) —
|
|
159
|
+
indexing derives from presets, so the coverage numbers are how you see what a config
|
|
160
|
+
actually indexes and embeds. A scope with fewer signals just uses fewer (a semantic-off
|
|
161
|
+
preset searches lexically); a saved search naming an unknown preset errors at `check`.
|
|
159
162
|
- Save a query into `sense.config.json` only when it will be reused; run ad-hoc otherwise.
|
|
160
163
|
- A one-line `summary:` per note is optional and pays twice: it appears in result rows and is a
|
|
161
164
|
weighted search field. Date comparisons work for dates written as ISO 8601 (`2026-08-12`, or
|
|
@@ -0,0 +1,121 @@
|
|
|
1
|
+
# sense-setup: worked configurations
|
|
2
|
+
|
|
3
|
+
Four tree shapes, each with its config and the commands an agent actually runs. Field
|
|
4
|
+
names and folder names are illustrative — your tree defines its own.
|
|
5
|
+
|
|
6
|
+
## A. Compiled wiki over immutable sources (the llm-wiki pattern)
|
|
7
|
+
|
|
8
|
+
`raw/` holds ingested sources — big, never hand-edited. `wiki/` holds agent-compiled
|
|
9
|
+
pages — linked, curated. The human drops sources and asks questions; the agent compiles
|
|
10
|
+
and cites.
|
|
11
|
+
|
|
12
|
+
```json
|
|
13
|
+
{
|
|
14
|
+
"version": 3,
|
|
15
|
+
"presets": {
|
|
16
|
+
"default": { "include": ["wiki/**/*.md"], "k": 10 },
|
|
17
|
+
"raw": { "include": ["raw/**/*.md"], "k": 5, "semantic": false }
|
|
18
|
+
},
|
|
19
|
+
"queries": {
|
|
20
|
+
"uncompiled": "SELECT path, _mtime FROM frontmatter WHERE path LIKE 'raw/%' AND path NOT IN (SELECT dst FROM links WHERE dst IS NOT NULL) ORDER BY _mtime DESC",
|
|
21
|
+
"stubs": "SELECT path, _size FROM frontmatter WHERE path LIKE 'wiki/%' AND _size < 500 ORDER BY _size",
|
|
22
|
+
"dead-links": "SELECT src, target FROM links WHERE dst IS NULL ORDER BY src"
|
|
23
|
+
}
|
|
24
|
+
}
|
|
25
|
+
```
|
|
26
|
+
|
|
27
|
+
```
|
|
28
|
+
sense uncompiled # compile queue: raw files nothing cites yet
|
|
29
|
+
sense search "how does attention scale" # wiki only (default preset), vectors on
|
|
30
|
+
sense search "rotary embeddings" --preset raw # cite from sources; lexical, k=5
|
|
31
|
+
sense dead-links # rows are broken citations to fix
|
|
32
|
+
```
|
|
33
|
+
|
|
34
|
+
What the shape buys: bare search never ranks raw noise above compiled pages; raw pays no
|
|
35
|
+
vector/link cost; the compile queue, stub list, and citation integrity are one saved
|
|
36
|
+
query each. The maintenance loop is `uncompiled` → write the wiki page citing its sources
|
|
37
|
+
→ `dead-links` stays empty.
|
|
38
|
+
|
|
39
|
+
## B. Nightly agent memory, consolidated (the dreaming pattern)
|
|
40
|
+
|
|
41
|
+
`memory/` accumulates small notes written at session end, each with `project`,
|
|
42
|
+
`created`, and `kind` (observation / steer / decision) frontmatter. A consolidation agent
|
|
43
|
+
runs periodically: prune, merge, surface contradictions for the human. Retired notes move
|
|
44
|
+
to `archive/` — still queryable, no longer embedded or ranked.
|
|
45
|
+
|
|
46
|
+
```json
|
|
47
|
+
{
|
|
48
|
+
"version": 3,
|
|
49
|
+
"presets": {
|
|
50
|
+
"default": { "include": ["memory/**/*.md"], "k": 10 },
|
|
51
|
+
"archive": { "include": ["archive/**/*.md"], "k": 10, "semantic": false }
|
|
52
|
+
},
|
|
53
|
+
"queries": {
|
|
54
|
+
"project": "SELECT path, kind, created, title FROM frontmatter WHERE project = ? ORDER BY created DESC",
|
|
55
|
+
"steers": "SELECT path, created, title FROM frontmatter WHERE kind = 'steer' AND project = ? ORDER BY created",
|
|
56
|
+
"retirement": "SELECT path, project, created FROM frontmatter WHERE datetime(created) < datetime('now','-90 day') AND path NOT IN (SELECT dst FROM links WHERE dst IS NOT NULL)",
|
|
57
|
+
"unfiled": "SELECT path FROM frontmatter WHERE project IS NULL"
|
|
58
|
+
}
|
|
59
|
+
}
|
|
60
|
+
```
|
|
61
|
+
|
|
62
|
+
```
|
|
63
|
+
sense project acme-app # one project's notes, newest first
|
|
64
|
+
sense search "prefers terse commit messages" --k 5
|
|
65
|
+
→ memory/acme-app/2026-08-02-commits.md via: match
|
|
66
|
+
→ memory/acme-app/2026-06-11-style.md via: vector similarity: 0.71 # near-duplicate → merge candidate
|
|
67
|
+
sense steers acme-app # oldest first: does a new steer override an old one?
|
|
68
|
+
sense retirement # old + uncited → move to archive/
|
|
69
|
+
sense unfiled # rows are notes missing a project — file them
|
|
70
|
+
```
|
|
71
|
+
|
|
72
|
+
Presets are structural (live vs archived); per-project filtering is metadata (`project = ?`)
|
|
73
|
+
— one tree serves every project. Semantic search over the memory preset is the
|
|
74
|
+
near-duplicate detector: search a new note's own summary and read `similarity` within the
|
|
75
|
+
results. Whether an old steer was overridden is a question for the human — found by
|
|
76
|
+
search, never decided by it.
|
|
77
|
+
|
|
78
|
+
## C. Evidence corpus: claims trace to sources
|
|
79
|
+
|
|
80
|
+
`sources/` (immutable imports), `notes/` (one reading note per source), `reviews/`
|
|
81
|
+
(synthesis whose claims must cite notes). The same shape fits incident reports and
|
|
82
|
+
postmortems, user research and findings, due diligence and memos.
|
|
83
|
+
|
|
84
|
+
```json
|
|
85
|
+
{
|
|
86
|
+
"version": 3,
|
|
87
|
+
"presets": {
|
|
88
|
+
"default": { "include": ["reviews/**/*.md", "notes/**/*.md"], "k": 10 },
|
|
89
|
+
"source": { "include": ["sources/**/*.md"], "k": 5, "semantic": false }
|
|
90
|
+
},
|
|
91
|
+
"queries": {
|
|
92
|
+
"unsupported": "SELECT path, title FROM frontmatter WHERE path LIKE 'reviews/%' AND path NOT IN (SELECT src FROM links WHERE dst LIKE 'notes/%')",
|
|
93
|
+
"unread": "SELECT path FROM frontmatter WHERE path LIKE 'sources/%' AND path NOT IN (SELECT dst FROM links WHERE dst IS NOT NULL)",
|
|
94
|
+
"by-topic": "SELECT path, title FROM frontmatter WHERE path NOT LIKE 'sources/%' AND has(topics, ?) ORDER BY path"
|
|
95
|
+
}
|
|
96
|
+
}
|
|
97
|
+
```
|
|
98
|
+
|
|
99
|
+
```
|
|
100
|
+
sense search "replication failures in priming studies" # reviews + notes; sources never dilute
|
|
101
|
+
sense search "the claim's exact phrasing" --preset source # citation pull on demand
|
|
102
|
+
sense unsupported # rows are synthesis claims with no note behind them
|
|
103
|
+
sense unread # the reading queue: sources no note cites
|
|
104
|
+
```
|
|
105
|
+
|
|
106
|
+
## D. A plain vault: zero configuration
|
|
107
|
+
|
|
108
|
+
Someone else's Obsidian vault, heterogeneous, no structure worth declaring — the
|
|
109
|
+
`sense init` starter untouched. The workflow is discovery:
|
|
110
|
+
|
|
111
|
+
```
|
|
112
|
+
sense map # fields in use, hub notes, recent changes
|
|
113
|
+
sense search "dataview queries" # words + links + meaning, one ranked list
|
|
114
|
+
sense query "SELECT j.value AS tag, COUNT(*) n FROM frontmatter, json_each(frontmatter.tags) j GROUP BY j.value ORDER BY n DESC LIMIT 20"
|
|
115
|
+
sense peek "Plugins/dataview.md" # outline + links before reading
|
|
116
|
+
```
|
|
117
|
+
|
|
118
|
+
Presets earn their place only when a tree has parts deserving different treatment; a tree
|
|
119
|
+
that is one kind of thing needs none of the vocabulary above. On a big vault, raise
|
|
120
|
+
`default`'s `k` and read `lines` ranges instead of whole files — or start from the
|
|
121
|
+
starter's `large` preset.
|