sensemaking 0.9.4 → 0.10.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +50 -100
- package/dist/cjs/cli/check.js +7 -13
- package/dist/cjs/cli/check.js.map +1 -1
- package/dist/cjs/cli/download.d.cts +3 -0
- package/dist/cjs/cli/download.d.ts +3 -0
- package/dist/cjs/cli/download.js +214 -0
- package/dist/cjs/cli/download.js.map +1 -0
- package/dist/cjs/cli/index.d.cts +6 -3
- package/dist/cjs/cli/index.d.ts +6 -3
- package/dist/cjs/cli/index.js +23 -5
- package/dist/cjs/cli/index.js.map +1 -1
- package/dist/cjs/cli/init.js +4 -1
- package/dist/cjs/cli/init.js.map +1 -1
- package/dist/cjs/cli/map.js +2 -2
- package/dist/cjs/cli/map.js.map +1 -1
- package/dist/cjs/cli/named.js +10 -18
- package/dist/cjs/cli/named.js.map +1 -1
- package/dist/cjs/cli/path.d.cts +3 -0
- package/dist/cjs/cli/path.d.ts +3 -0
- package/dist/cjs/cli/path.js +161 -0
- package/dist/cjs/cli/path.js.map +1 -0
- package/dist/cjs/cli/peek.js +3 -2
- package/dist/cjs/cli/peek.js.map +1 -1
- package/dist/cjs/cli/related.d.cts +3 -0
- package/dist/cjs/cli/related.d.ts +3 -0
- package/dist/cjs/cli/related.js +279 -0
- package/dist/cjs/cli/related.js.map +1 -0
- package/dist/cjs/cli/search.js +3 -7
- package/dist/cjs/cli/search.js.map +1 -1
- package/dist/cjs/cli/shared.d.cts +3 -1
- package/dist/cjs/cli/shared.d.ts +3 -1
- package/dist/cjs/cli/shared.js +28 -5
- package/dist/cjs/cli/shared.js.map +1 -1
- package/dist/cjs/cli/sql.d.cts +3 -0
- package/dist/cjs/cli/sql.d.ts +3 -0
- package/dist/cjs/cli/{query.js → sql.js} +4 -4
- package/dist/cjs/cli/sql.js.map +1 -0
- package/dist/cjs/cli/status.js +9 -2
- package/dist/cjs/cli/status.js.map +1 -1
- package/dist/cjs/cli.js +14 -15
- package/dist/cjs/cli.js.map +1 -1
- package/dist/cjs/commands.d.cts +12 -4
- package/dist/cjs/commands.d.ts +12 -4
- package/dist/cjs/commands.js +138 -58
- package/dist/cjs/commands.js.map +1 -1
- package/dist/cjs/config.d.cts +6 -4
- package/dist/cjs/config.d.ts +6 -4
- package/dist/cjs/config.js +137 -58
- package/dist/cjs/config.js.map +1 -1
- package/dist/cjs/db.js +14 -34
- package/dist/cjs/db.js.map +1 -1
- package/dist/cjs/errors.d.cts +1 -1
- package/dist/cjs/errors.d.ts +1 -1
- package/dist/cjs/errors.js.map +1 -1
- package/dist/cjs/features/embed.d.cts +14 -0
- package/dist/cjs/features/embed.d.ts +14 -0
- package/dist/cjs/features/embed.js +248 -54
- package/dist/cjs/features/embed.js.map +1 -1
- package/dist/cjs/features/links.js +7 -16
- package/dist/cjs/features/links.js.map +1 -1
- package/dist/cjs/features/rank.js +2 -6
- package/dist/cjs/features/rank.js.map +1 -1
- package/dist/cjs/index.d.cts +1 -1
- package/dist/cjs/index.d.ts +1 -1
- package/dist/cjs/index.js.map +1 -1
- package/dist/cjs/output.js +11 -14
- package/dist/cjs/output.js.map +1 -1
- package/dist/cjs/progress.js +2 -4
- package/dist/cjs/progress.js.map +1 -1
- package/dist/cjs/scan.d.cts +1 -0
- package/dist/cjs/scan.d.ts +1 -0
- package/dist/cjs/scan.js +67 -32
- package/dist/cjs/scan.js.map +1 -1
- package/dist/cjs/search-error.js +4 -6
- package/dist/cjs/search-error.js.map +1 -1
- package/dist/cjs/traverse.d.cts +8 -0
- package/dist/cjs/traverse.d.ts +8 -0
- package/dist/cjs/traverse.js +97 -0
- package/dist/cjs/traverse.js.map +1 -0
- package/dist/esm/cli/check.js +7 -13
- package/dist/esm/cli/check.js.map +1 -1
- package/dist/esm/cli/download.d.ts +3 -0
- package/dist/esm/cli/download.js +27 -0
- package/dist/esm/cli/download.js.map +1 -0
- package/dist/esm/cli/index.d.ts +6 -3
- package/dist/esm/cli/index.js +13 -7
- package/dist/esm/cli/index.js.map +1 -1
- package/dist/esm/cli/init.js +4 -1
- package/dist/esm/cli/init.js.map +1 -1
- package/dist/esm/cli/map.js +3 -2
- package/dist/esm/cli/map.js.map +1 -1
- package/dist/esm/cli/named.js +10 -16
- package/dist/esm/cli/named.js.map +1 -1
- package/dist/esm/cli/path.d.ts +3 -0
- package/dist/esm/cli/path.js +51 -0
- package/dist/esm/cli/path.js.map +1 -0
- package/dist/esm/cli/peek.js +4 -2
- package/dist/esm/cli/peek.js.map +1 -1
- package/dist/esm/cli/related.d.ts +3 -0
- package/dist/esm/cli/related.js +26 -0
- package/dist/esm/cli/related.js.map +1 -0
- package/dist/esm/cli/search.js +2 -5
- package/dist/esm/cli/search.js.map +1 -1
- package/dist/esm/cli/shared.d.ts +3 -1
- package/dist/esm/cli/shared.js +28 -5
- package/dist/esm/cli/shared.js.map +1 -1
- package/dist/esm/cli/sql.d.ts +3 -0
- package/dist/esm/cli/{query.js → sql.js} +4 -4
- package/dist/esm/cli/sql.js.map +1 -0
- package/dist/esm/cli/status.js +8 -1
- package/dist/esm/cli/status.js.map +1 -1
- package/dist/esm/cli.js +13 -10
- package/dist/esm/cli.js.map +1 -1
- package/dist/esm/commands.d.ts +12 -4
- package/dist/esm/commands.js +111 -68
- package/dist/esm/commands.js.map +1 -1
- package/dist/esm/config.d.ts +6 -4
- package/dist/esm/config.js +126 -75
- package/dist/esm/config.js.map +1 -1
- package/dist/esm/db.js +16 -38
- package/dist/esm/db.js.map +1 -1
- package/dist/esm/errors.d.ts +1 -1
- package/dist/esm/errors.js.map +1 -1
- package/dist/esm/features/embed.d.ts +14 -0
- package/dist/esm/features/embed.js +110 -35
- package/dist/esm/features/embed.js.map +1 -1
- package/dist/esm/features/links.js +8 -17
- package/dist/esm/features/links.js.map +1 -1
- package/dist/esm/features/rank.js +2 -6
- package/dist/esm/features/rank.js.map +1 -1
- package/dist/esm/features/types.js.map +1 -1
- package/dist/esm/index.d.ts +1 -1
- package/dist/esm/index.js.map +1 -1
- package/dist/esm/output.js +13 -16
- package/dist/esm/output.js.map +1 -1
- package/dist/esm/progress.js +2 -4
- package/dist/esm/progress.js.map +1 -1
- package/dist/esm/scan.d.ts +1 -0
- package/dist/esm/scan.js +41 -34
- package/dist/esm/scan.js.map +1 -1
- package/dist/esm/search-error.js +4 -6
- package/dist/esm/search-error.js.map +1 -1
- package/dist/esm/traverse.d.ts +8 -0
- package/dist/esm/traverse.js +71 -0
- package/dist/esm/traverse.js.map +1 -0
- package/package.json +18 -5
- package/schema.json +12 -16
- package/skills/sense/EXAMPLES.md +13 -22
- package/skills/sense/SKILL.md +65 -140
- package/skills/sense-setup/EXAMPLES.md +30 -46
- package/skills/sense-setup/SKILL.md +22 -83
- package/dist/cjs/cli/query.d.cts +0 -3
- package/dist/cjs/cli/query.d.ts +0 -3
- package/dist/cjs/cli/query.js.map +0 -1
- package/dist/esm/cli/query.d.ts +0 -3
- package/dist/esm/cli/query.js.map +0 -1
package/skills/sense/SKILL.md
CHANGED
|
@@ -1,48 +1,25 @@
|
|
|
1
1
|
---
|
|
2
2
|
name: sense
|
|
3
|
-
description: Query a markdown tree with the sense CLI
|
|
3
|
+
description: Query a markdown tree with the sense CLI: filter notes by frontmatter, full-text search the prose, follow wikilinks/backlinks, trace how notes connect (link path, k-hop neighborhood, similar-but-unlinked), and read note outlines. Use when the user wants to query, filter, count, search, or report on a folder of markdown notes, when you need to find which notes discuss a topic before reading them, when you want a note's backlinks or structure, when you want to know how two notes connect or what surrounds one in the link graph, when a directory has a sense.config.json, or when asked to add a saved entry to one.
|
|
4
4
|
---
|
|
5
5
|
|
|
6
6
|
# sense
|
|
7
7
|
|
|
8
|
-
SQL over a markdown tree, kept fresh by a filesystem check on every query. Every file becomes
|
|
9
|
-
rows in `frontmatter` (one column per key, plus `path`/`_mtime`/`_size`/`_rank`), `content`
|
|
10
|
-
(FTS5: `title`, `summary`, `text`), `links` (`src`, `target`, `dst` — `NULL` dst = dead link),
|
|
11
|
-
and `sections` (heading outline with line ranges and token estimates). Features add their own
|
|
12
|
-
storage; `map` and `status` report which are on.
|
|
8
|
+
SQL over a markdown tree, kept fresh by a filesystem check on every query. Every file becomes rows in `frontmatter` (one column per key, plus `path`/`_mtime`/`_size`/`_rank`), `content` (FTS5: `title`, `summary`, `text`), `links` (`src`, `target`, `dst`; `NULL` dst = dead link), and `sections` (heading outline with line ranges and token estimates). Features add their own storage; `map` and `status` report which are on.
|
|
13
9
|
|
|
14
10
|
## What each tool is for
|
|
15
11
|
|
|
16
|
-
Every result is a reference (path, metadata, excerpt), never file contents; prose enters
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
-
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
-
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
ranked list. Read `via` per row: `match` rows contained your words; `vector`-only rows
|
|
28
|
-
did not — they are the "these words aren't in the tree; this is what's near in meaning"
|
|
29
|
-
signal. Vector rows are conceptual similarity, not typo-tolerance; false positives are
|
|
30
|
-
expected, labeled, and bounded by `--k`. `--lexical` skips vectors for one command when
|
|
31
|
-
word-presence is the question.
|
|
32
|
-
- `map` answers "what is this tree" — fields, hub notes, recent changes — when the tree is
|
|
33
|
-
unfamiliar.
|
|
34
|
-
- `peek <path>` prices a file before you pay for it: outline with `[L143-162, ~380t]`
|
|
35
|
-
ranges, links both ways. Every list shows its first 20 with the true total; the
|
|
36
|
-
`sections` and `links` tables hold the rest, so a peek costs a few hundred tokens on any
|
|
37
|
-
note — heading-dense monsters included.
|
|
38
|
-
- When you know the file and need its contents, `Read` it — sense adds nothing there. On
|
|
39
|
-
large files peek's ranges let you read just one section; small files are often cheaper
|
|
40
|
-
whole.
|
|
41
|
-
|
|
42
|
-
Output defaults to a table, built for humans; `--format json` returns the same rows
|
|
43
|
-
machine-parseable. That also makes a saved query usable as a CI/hook gate with zero added
|
|
44
|
-
mechanism: `[ "$(sense <query> --format json)" = "[]" ]` is true exactly when the query
|
|
45
|
-
returned no rows.
|
|
12
|
+
Every result is a reference (path, metadata, excerpt), never file contents; prose enters context only when you Read it. Costs: `map` is fixed-size, a `search` row is tens of tokens, and a `peek` stays flat however large the note is. Which tool fits is a property of the question:
|
|
13
|
+
|
|
14
|
+
- A deterministic, factual answer over known fields (counts, filters, "which notes have X") is SQL: `sense sql`, a saved `{ sql }` entry, or `search --where`. Enumerates every match; same result regardless of phrasing.
|
|
15
|
+
- Locating notes about something is `search`, one text through every engine the scope has: word match (bare words AND-join, one absent word = zero lexical rows; write `a OR b OR c` for any-word), link-graph expansion, and vector similarity, fused into one ranked list. Read `via` per row: `match` rows contained your words; `vector`-only rows did not. A `vector`-only row means the search words don't appear in that note; it showed up because the model judged it semantically related. Vector rows are conceptual similarity, not typo-tolerance; false positives are expected, labeled, and bounded by `--k`. A scope searches with vectors when its preset has `semantic` on (the default) and the tree names an `embed` model; a `semantic: false` preset searches on words and links. A preset that asks for vectors when the model is not downloaded is an error naming `sense download`, not a quieter result: the same search must not answer differently before and after a download.
|
|
16
|
+
- `map` answers "what is this tree" (fields, hub notes, recent changes) when the tree is unfamiliar.
|
|
17
|
+
- `peek <path>` prices a file before you pay for it: outline with `[L143-162, ~380t]` ranges and links both ways. Every list shows its first 20 with the true total; the `sections` and `links` tables hold the rest, so a peek costs a few hundred tokens on any note.
|
|
18
|
+
- `path <a> <b>` walks the link graph for a chain connecting two notes, or reports none within the depth bound: it answers how they connect, not just that both exist.
|
|
19
|
+
- `related <note>` ranks notes near in meaning to one note that it does not already link to: the links it is missing. It reads the meaning-vectors, so it needs vectors on for the scope and scans them, costing about what a semantic `search` does, not what a `peek` does. Vectors need the model on disk; nothing fetches it implicitly, so `sense download` is a one-time step per machine. Without it `search` still answers on words and links (it prints a note saying vectors are off), while `related` reports the missing model instead of an empty list.
|
|
20
|
+
- When you know the file and need its contents, `Read` it. sense adds nothing there. On large files peek's ranges let you read just one section; small files are often cheaper whole.
|
|
21
|
+
|
|
22
|
+
Output defaults to a table, built for humans; `--format json` returns the same rows machine-parseable. That also makes a saved query usable as a CI/hook gate with zero added mechanism: `[ "$(sense <name> --format json)" = "[]" ]` is true exactly when it returned no rows.
|
|
46
23
|
|
|
47
24
|
## Commands
|
|
48
25
|
|
|
@@ -50,124 +27,72 @@ returned no rows.
|
|
|
50
27
|
sense search "pricing OR billing OR invoicing" --where "f.status = 'active'" --k 10
|
|
51
28
|
sense search "sourcing quotes" --preset raw # a named settings bundle from the config
|
|
52
29
|
sense peek notes/pricing-model.md # a unique basename also works
|
|
30
|
+
sense path onboarding.md pricing-model.md # link chain between two notes, or none within the bound
|
|
31
|
+
sense related notes/pricing-model.md # notes similar by meaning it does not yet link to
|
|
53
32
|
sense map
|
|
54
|
-
sense
|
|
55
|
-
sense <name> [params...] #
|
|
56
|
-
sense --list | status | rebuild | check
|
|
33
|
+
sense sql "<statement>" [params...] # ad-hoc SQL; ? binds positional args, count-checked
|
|
34
|
+
sense <name> [params...] # a saved query from sense.config.json
|
|
35
|
+
sense --list | status | rebuild | check | download
|
|
57
36
|
```
|
|
58
37
|
|
|
59
|
-
- Terms pass verbatim to FTS5 MATCH. Bare words AND-join
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
-
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
both ways.
|
|
69
|
-
- A frontmatter query enumerates its matches deterministically; search ranks by term overlap,
|
|
70
|
-
so results shift as phrasing shifts. Trade-off: a query needs a known field, search doesn't.
|
|
71
|
-
- The `via` column says what produced each row — `match` (words hit), `link` (connected to
|
|
72
|
-
notes that hit), `vector` (near in meaning), and combinations. The `lines` column, when
|
|
73
|
-
set, points at the section that earned the row — the best-matching chunk on vector rows,
|
|
74
|
-
the term cluster's section on large lexical notes — and is a direct `Read` range; null
|
|
75
|
-
means the whole note is the reference.
|
|
76
|
-
- Scope comes from presets: bare `search` uses the config's `default` preset; `--preset
|
|
77
|
-
<name>` picks another (unknown names error, listing what's declared); `--include <glob>`
|
|
78
|
-
is an ad-hoc scope that replaces the preset's globs for one command. `--where` takes any
|
|
79
|
-
SQL condition against frontmatter alias `f` — not only field equality:
|
|
80
|
-
`"f.status = 'active' AND has(f.tags, 'x')"`, `"datetime(f.created) >= datetime(?)"` —
|
|
81
|
-
and filters within the scope. `sense status` shows every preset with its coverage.
|
|
82
|
-
- `score` is a rank-fusion value: it ranks rows within one result set and is not comparable
|
|
83
|
-
across queries, not a relevance magnitude — it encodes how many signals fired and at what
|
|
84
|
-
rank, so a perfect lexical hit and a weak vector-only hit can read the same number. With
|
|
85
|
-
vectors active, rows carry `similarity`: the cosine (-1 to 1) of the query against that
|
|
86
|
-
file's best-matching chunk — the same chunk the `lines` range points at. It orders vector
|
|
87
|
-
evidence within a result set; the range it spans depends on the corpus and the embedding
|
|
88
|
-
model, and compresses on small trees, where even a nonsense query has a moderately near
|
|
89
|
-
neighbour somewhere. Compare similarities within a result set rather than against a fixed
|
|
90
|
-
cutoff carried between trees.
|
|
91
|
-
- Absence evidence lives in the labels: `search --lexical` (or a semantic-off scope)
|
|
92
|
-
returns 0 rows when the words are nowhere in the tree. Default `search` always returns
|
|
93
|
-
up to `k` rows — nearest-neighbour search has a nearest neighbour for any input — so a
|
|
94
|
-
result of only `via: vector` rows IS the absence signal for the words themselves;
|
|
95
|
-
`similarity` and the snippet are the evidence for judging whether a vector row is a real
|
|
96
|
-
conceptual hit.
|
|
97
|
-
- Besides SQL strings, a config entry can save a whole search:
|
|
98
|
-
`"hot": { "search": "pricing OR billing", "preset": "raw", "k": 20 }` runs as
|
|
99
|
-
`sense hot` — the scenario's settings ride along with the name, so repeat runs need no
|
|
100
|
-
flags. An invocation-level `--preset`, `--k`, `--where`, or `--lexical` overrides the
|
|
101
|
-
saved value; `--list` marks these entries `(search)`.
|
|
102
|
-
- `sense check` prepares every saved query and probes every saved search lexically with
|
|
103
|
-
k=1, so a typo'd column, stale SQL, bad FTS5 syntax, or unknown preset fails at check
|
|
104
|
-
time instead of silently mid-task. It reports row counts; whether an empty result is
|
|
105
|
-
good or bad is the reader's judgment — a dead-link query returning rows means broken
|
|
106
|
-
citations to fix, and the agent reads that directly.
|
|
38
|
+
- Terms pass verbatim to FTS5 MATCH. Bare words AND-join (one absent word means zero rows), so write `OR` yourself when you want any-word matching; double-quote punctuated terms (`"customer-facing"`, `"founder's"`); invalid syntax is an error, not a rewrite. The same rules apply to search commands you write into subagent briefs.
|
|
39
|
+
- When a search misses, the recall levers are: OR-in synonyms and concrete instances (the index only knows the words in the files; a note about a specific tool rarely names its category), raise `--k` (a row costs tens of tokens), and widen the scope (`--preset`, or `--include` for an ad-hoc glob). Vector rows already cover the meaning-over-words gap by default. Each widening adds candidates and dilutes ranking, so the noise trade-off runs both ways.
|
|
40
|
+
- A frontmatter query enumerates its matches deterministically; search ranks by term overlap, so results shift as phrasing shifts. Trade-off: a query needs a known field, search doesn't.
|
|
41
|
+
- The `via` column says what produced each row: `match` (words hit), `link` (connected to notes that hit), `vector` (near in meaning), and combinations. The `lines` column, when set, points at the section that earned the row (the best-matching chunk on vector rows, the term cluster's section on large lexical notes) and is a direct `Read` range; null means the whole note is the reference.
|
|
42
|
+
- Scope is one vocabulary shared by `search`, `map`, `peek`, `path`, and `related`: bare command uses the config's `default` preset; `--preset <name>` picks another (unknown names error, listing what's declared); `--include <glob>` and `--exclude <glob>` are ad-hoc globs for one command, each overriding its own side of the preset, so one does not clear the other; `--no-exclude` drops the preset's `exclude` for one command, the only way to widen past it without editing config (it widens the query scope, not the index: a file no preset covers is never indexed). `--where` takes any SQL condition against frontmatter alias `f`, not only field equality: `"f.status = 'active' AND has(f.tags, 'x')"`, `"datetime(f.created) >= datetime(?)"`. There is no whole-index flag: a broad `default` preset, or a declared `all` preset (`include ["**/*"]`), is the whole tree. `sense status` shows every preset with its coverage.
|
|
43
|
+
- `score` is a rank-fusion value: it ranks rows within one result set and is not comparable across queries, not a relevance magnitude. It encodes how many signals fired and at what rank, so a perfect lexical hit and a weak vector-only hit can read the same number. With vectors active, rows carry `similarity`: the cosine (-1 to 1) of the query against that file's best-matching chunk (the same chunk the `lines` range points at). It orders vector evidence within a result set; the range it spans depends on the corpus and the embedding model, and compresses on small trees, where even a nonsense query has a moderately near neighbour somewhere. Compare similarities within a result set rather than against a fixed cutoff carried between trees.
|
|
44
|
+
- Absence evidence lives in the labels: a `semantic: false` preset (or a tree with no `embed` block) returns 0 rows when the words are nowhere in it. Default `search` always returns up to `k` rows (nearest-neighbour search has a nearest neighbour for any input), so a result of only `via: vector` rows IS the absence signal for the words themselves; `similarity` and the snippet are the evidence for judging whether a vector row is a real conceptual hit.
|
|
45
|
+
- A `queries` entry names the verb it runs, mirroring the two commands: `"dead-links": { "sql": "SELECT src, target FROM links WHERE dst IS NULL" }` runs as `sense dead-links`, and `"hot": { "search": "pricing OR billing", "preset": "raw", "k": 20 }` runs as `sense hot` with its settings baked in, so repeat runs need no flags. An invocation-level `--preset`, `--k`, or `--where` overrides a saved search's value; `--list` labels each entry `(sql)` or `(search)`.
|
|
46
|
+
- `sense check` prepares every `{ sql }` entry and probes every `{ search }` entry with k=1, so a typo'd column, stale SQL, bad FTS5 syntax, or unknown preset fails at check time instead of silently mid-task. The probe skips the vector half, so checking a config never embeds the tree or calls an api endpoint. It reports row counts; whether an empty result is good or bad is the reader's judgment: a dead-link query returning rows means broken citations to fix, and the agent reads that directly.
|
|
107
47
|
|
|
108
48
|
## SQL
|
|
109
49
|
|
|
110
50
|
The commands are shorthands over those tables; anything they don't express, SQL does.
|
|
111
51
|
|
|
112
52
|
```
|
|
113
|
-
sense
|
|
114
|
-
sense
|
|
115
|
-
sense
|
|
116
|
-
sense
|
|
117
|
-
sense
|
|
118
|
-
sense
|
|
119
|
-
sense
|
|
53
|
+
sense sql "SELECT name FROM pragma_table_info('frontmatter')" # what fields exist
|
|
54
|
+
sense sql "SELECT DISTINCT status FROM frontmatter" # what values a field takes
|
|
55
|
+
sense sql "SELECT src FROM links WHERE dst = ?" notes/pricing-model.md # backlinks
|
|
56
|
+
sense sql "SELECT src, target FROM links WHERE dst IS NULL" # dead links
|
|
57
|
+
sense sql "SELECT path FROM frontmatter WHERE path NOT IN (SELECT dst FROM links WHERE dst IS NOT NULL) AND path NOT IN (SELECT src FROM links)" # linked neither way (fine if intentional; linking is optional)
|
|
58
|
+
sense sql "SELECT heading, start_line, tokens FROM sections WHERE path = ?" a.md # budget a read
|
|
59
|
+
sense sql "SELECT j.value, COUNT(*) n FROM frontmatter, json_each(frontmatter.tags) j GROUP BY j.value ORDER BY n DESC" # count per array member
|
|
60
|
+
```
|
|
61
|
+
|
|
62
|
+
`path` covers the route between two notes, and `search`'s `via: link` rows are the ranked neighborhood around a query; a structural k-hop walk is a bounded `WITH RECURSIVE` over `links`. Bound the depth: an unbounded walk on a densely linked tree enumerates paths exponentially. Pass these through `sense sql "<sql>" <seed>` or save as `{ "sql": "..." }`.
|
|
63
|
+
|
|
64
|
+
```
|
|
65
|
+
-- notes within 2 hops of a seed, links both ways (UNION dedups, so it terminates)
|
|
66
|
+
WITH RECURSIVE hop(path, d) AS (
|
|
67
|
+
SELECT ?, 0
|
|
68
|
+
UNION
|
|
69
|
+
SELECT CASE WHEN l.src = hop.path THEN l.dst ELSE l.src END, hop.d + 1
|
|
70
|
+
FROM hop JOIN links l ON (l.src = hop.path OR l.dst = hop.path) AND l.dst IS NOT NULL
|
|
71
|
+
WHERE hop.d < 2
|
|
72
|
+
)
|
|
73
|
+
SELECT DISTINCT path FROM hop WHERE d > 0;
|
|
74
|
+
|
|
75
|
+
-- notes cited alongside a seed: they share a note that links to both (co-citation)
|
|
76
|
+
SELECT DISTINCT b.dst FROM links a JOIN links b ON a.src = b.src
|
|
77
|
+
WHERE a.dst = ? AND b.dst IS NOT NULL AND b.dst <> a.dst;
|
|
120
78
|
```
|
|
121
79
|
|
|
122
|
-
- `content MATCH` takes FTS5 syntax: `a OR b`, `"phrase"`, `pref*`, `NEAR(a b, 5)`,
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
-
|
|
127
|
-
|
|
128
|
-
|
|
129
|
-
1 MB note. `search` bounds this itself (docs past 16 KB get an equivalent excerpt another
|
|
130
|
-
way); in hand-written SQL, guard it: `CASE WHEN length(text) <= 16384 THEN snippet(...)
|
|
131
|
-
END`, or select `title`/`summary` instead of an excerpt.
|
|
132
|
-
- Select `content.title`/`content.summary` (always exist, empty when absent) rather than
|
|
133
|
-
`f.title`/`f.summary` (discovered columns — error on trees that never declare them).
|
|
134
|
-
- Frontmatter values keep their YAML type: strings are TEXT, whole numbers and booleans are
|
|
135
|
-
INTEGER (`true` stores as 1, so `WHERE flag = 1` matches and `WHERE flag = 'true'` matches
|
|
136
|
-
nothing), fractions are REAL, lists and maps are JSON text. `map` prints the observed type
|
|
137
|
-
per field, and a field showing two types (`integer,text`) has drifted across notes.
|
|
138
|
-
- `has(field, value)`: array membership on JSON-array fields, substring on strings, false on NULL
|
|
139
|
-
— the `includes()` convention. Substring means `has(f.status, 'active')` also matches
|
|
140
|
-
`inactive`; exact scalar match is `f.status = ?`, deliberate substring is `LIKE`, exact array
|
|
141
|
-
membership is `EXISTS (SELECT 1 FROM json_each(f.tags) WHERE value = ?)`.
|
|
142
|
-
To aggregate per member instead, use `json_each(frontmatter.<field>)` (above) -- GROUP BY on the
|
|
143
|
-
raw column splits `["a","b"]` and `["b","a"]` into separate buckets.
|
|
144
|
-
- Date fields are stored as written. Compare through `datetime()`, which normalizes ISO 8601
|
|
145
|
-
timezone offsets to UTC: `WHERE datetime(created) >= datetime(?)`. Bare string comparison
|
|
146
|
-
is only safe when every note uses the same offset.
|
|
147
|
-
- To bound what a query puts into context: `snippet()` excerpts just the matching text,
|
|
148
|
-
`LIMIT` caps row counts, and selecting `path`/`title`/`summary` keeps rows small.
|
|
149
|
-
`SELECT text FROM content` returns the tree's entire prose (sense warns past 50 KB).
|
|
150
|
-
Aggregates (`COUNT`, `GROUP BY`) are already bounded. `SELECT * FROM frontmatter` is always
|
|
151
|
-
safe — prose is not a frontmatter column.
|
|
80
|
+
- `content MATCH` takes FTS5 syntax: `a OR b`, `"phrase"`, `pref*`, `NEAR(a b, 5)`, `summary: term`. Stemmed; markdown stripped at index time. Double-quote any term with punctuation. Bare `customer-facing` errors (`-` reads as a column filter), bare apostrophes are syntax errors: write `"customer-facing"`, `"founder's"`.
|
|
81
|
+
- Rank with `ORDER BY bm25(content, 10.0, 5.0, 1.0)` (title > summary > body); excerpt with `snippet(content, -1, '«', '»', '…', 10)`. snippet() re-tokenizes each matched doc and its cost grows superlinearly with doc size, measured ~10 s per query on a tree holding one 1 MB note. `search` bounds this itself (docs past 16 KB get an equivalent excerpt another way); in hand-written SQL, guard it: `CASE WHEN length(text) <= 16384 THEN snippet(...) END`, or select `title`/`summary` instead of an excerpt.
|
|
82
|
+
- Select `content.title`/`content.summary` (always exist, empty when absent) rather than `f.title`/`f.summary` (discovered columns; error on trees that never declare them).
|
|
83
|
+
- Frontmatter values keep their YAML type: strings are TEXT, whole numbers and booleans are INTEGER (`true` stores as 1, so `WHERE flag = 1` matches and `WHERE flag = 'true'` matches nothing), fractions are REAL, lists and maps are JSON text. `map` prints the observed type per field, and a field showing two types (`integer,text`) has drifted across notes.
|
|
84
|
+
- `has(field, value)`: array membership on JSON-array fields, substring on strings, false on NULL. This is the `includes()` convention. Substring means `has(f.status, 'active')` also matches `inactive`; exact scalar match is `f.status = ?`, deliberate substring is `LIKE`, exact array membership is `EXISTS (SELECT 1 FROM json_each(f.tags) WHERE value = ?)`. To aggregate per member instead, use `json_each(frontmatter.<field>)` (above) -- GROUP BY on the raw column splits `["a","b"]` and `["b","a"]` into separate buckets.
|
|
85
|
+
- Date fields are stored as written. Compare through `datetime()`, which normalizes ISO 8601 timezone offsets to UTC: `WHERE datetime(created) >= datetime(?)`. Bare string comparison is only safe when every note uses the same offset.
|
|
86
|
+
- To bound what a query puts into context: `snippet()` excerpts just the matching text, `LIMIT` caps row counts, and selecting `path`/`title`/`summary` keeps rows small. `SELECT text FROM content` returns the tree's entire prose (sense warns past 50 KB). Aggregates (`COUNT`, `GROUP BY`) are already bounded. `SELECT * FROM frontmatter` is always safe: prose is not a frontmatter column.
|
|
152
87
|
|
|
153
88
|
Worked traces: [EXAMPLES.md](EXAMPLES.md).
|
|
154
89
|
|
|
155
90
|
## Setup and upkeep
|
|
156
91
|
|
|
157
|
-
- Missing CLI: `npm install -g sensemaking`. Missing config: `sense init` at the tree root.
|
|
158
|
-
|
|
159
|
-
tree (presets, frontmatter conventions, note design) is the `sense-setup` skill.
|
|
160
|
-
- `map` and `status` report each preset's coverage (files matched, embedded count) —
|
|
161
|
-
indexing derives from presets, so the coverage numbers are how you see what a config
|
|
162
|
-
actually indexes and embeds. A scope with fewer signals just uses fewer (a semantic-off
|
|
163
|
-
preset searches lexically); a saved search naming an unknown preset errors at `check`.
|
|
92
|
+
- Missing CLI: `npm install -g sensemaking`. Missing config: `sense init` at the tree root. Discovery walks up from cwd; `--config <path>` overrides. Setting up or restructuring a tree (presets, frontmatter conventions, note design) is the `sense-setup` skill.
|
|
93
|
+
- `map` and `status` report each preset's coverage (files matched, embedded count). Indexing derives from presets, so the coverage numbers are how you see what a config actually indexes and embeds. A scope with fewer signals just uses fewer (a semantic-off preset searches lexically); a saved search naming an unknown preset errors at `check`.
|
|
164
94
|
- Save a query into `sense.config.json` only when it will be reused; run ad-hoc otherwise.
|
|
165
|
-
- A one-line `summary:` per note is optional and pays twice: it appears in result rows and is a
|
|
166
|
-
|
|
167
|
-
|
|
168
|
-
|
|
169
|
-
- Reserved frontmatter keys (dropped with a warning): `path`, `_mtime`, `_size`, `_rank`,
|
|
170
|
-
`content`, `links`, `sections`.
|
|
171
|
-
- Exit codes: `0` ok, `1` error (SQLite message verbatim), `2` usage (unknown query, wrong
|
|
172
|
-
param count).
|
|
173
|
-
- Doubted cache: `sense rebuild`. Rarely needed — every query reconciles first.
|
|
95
|
+
- A one-line `summary:` per note is optional and pays twice: it appears in result rows and is a weighted search field. Date comparisons work for dates written as ISO 8601 (`2026-08-12`, or with time and offset), the only format `datetime()` parses. Field names in examples (`status`, `tags`, `created`) are illustrative; your tree defines its own.
|
|
96
|
+
- Reserved frontmatter keys (dropped with a warning): `path`, `_mtime`, `_size`, `_rank`, `content`, `links`, `sections`.
|
|
97
|
+
- Exit codes: `0` ok, `1` error (SQLite message verbatim), `2` usage (unknown query, wrong param count).
|
|
98
|
+
- Doubted cache: `sense rebuild`. Rarely needed; every query reconciles first.
|
|
@@ -1,60 +1,53 @@
|
|
|
1
1
|
# sense-setup: worked configurations
|
|
2
2
|
|
|
3
|
-
Four tree shapes, each with its config and the commands an agent actually runs. Field
|
|
4
|
-
names and folder names are illustrative — your tree defines its own.
|
|
3
|
+
Four tree shapes, each with its config and the commands an agent actually runs. Field names and folder names are illustrative; your tree defines its own.
|
|
5
4
|
|
|
6
5
|
## A. Compiled wiki over immutable sources (the llm-wiki pattern)
|
|
7
6
|
|
|
8
|
-
`raw/` holds ingested sources
|
|
9
|
-
pages — linked, curated. The human drops sources and asks questions; the agent compiles
|
|
10
|
-
and cites.
|
|
7
|
+
`raw/` holds ingested sources: big, never hand-edited. `wiki/` holds agent-compiled pages: linked, curated. The human drops sources and asks questions; the agent compiles and cites.
|
|
11
8
|
|
|
12
9
|
```json
|
|
13
10
|
{
|
|
14
|
-
"version":
|
|
11
|
+
"version": 4,
|
|
15
12
|
"presets": {
|
|
16
13
|
"default": { "include": ["wiki/**/*.md"], "k": 10 },
|
|
17
|
-
"raw": { "include": ["raw/**/*.md"], "k": 5
|
|
14
|
+
"raw": { "include": ["raw/**/*.md"], "k": 5 }
|
|
18
15
|
},
|
|
16
|
+
"embed": { "model": "minishlab/potion-retrieval-32M", "type": "static" },
|
|
19
17
|
"queries": {
|
|
20
|
-
"uncompiled": "SELECT path, _mtime FROM frontmatter WHERE path LIKE 'raw/%' AND path NOT IN (SELECT dst FROM links WHERE dst IS NOT NULL) ORDER BY _mtime DESC",
|
|
21
|
-
"stubs": "SELECT path, _size FROM frontmatter WHERE path LIKE 'wiki/%' AND _size < 500 ORDER BY _size",
|
|
22
|
-
"dead-links": "SELECT src, target FROM links WHERE dst IS NULL ORDER BY src"
|
|
18
|
+
"uncompiled": { "sql": "SELECT path, _mtime FROM frontmatter WHERE path LIKE 'raw/%' AND path NOT IN (SELECT dst FROM links WHERE dst IS NOT NULL) ORDER BY _mtime DESC" },
|
|
19
|
+
"stubs": { "sql": "SELECT path, _size FROM frontmatter WHERE path LIKE 'wiki/%' AND _size < 500 ORDER BY _size" },
|
|
20
|
+
"dead-links": { "sql": "SELECT src, target FROM links WHERE dst IS NULL ORDER BY src" }
|
|
23
21
|
}
|
|
24
22
|
}
|
|
25
23
|
```
|
|
26
24
|
|
|
27
25
|
```
|
|
28
26
|
sense uncompiled # compile queue: raw files nothing cites yet
|
|
29
|
-
sense search "how does attention scale" # wiki only (default preset)
|
|
30
|
-
sense search "rotary embeddings" --preset raw # cite from sources
|
|
27
|
+
sense search "how does attention scale" # wiki only (default preset)
|
|
28
|
+
sense search "rotary embeddings" --preset raw # cite from sources, k=5
|
|
31
29
|
sense dead-links # rows are broken citations to fix
|
|
32
30
|
```
|
|
33
31
|
|
|
34
|
-
What the shape buys: bare search never ranks raw noise above compiled pages;
|
|
35
|
-
vector/link cost; the compile queue, stub list, and citation integrity are one saved
|
|
36
|
-
query each. The maintenance loop is `uncompiled` → write the wiki page citing its sources
|
|
37
|
-
→ `dead-links` stays empty.
|
|
32
|
+
What the shape buys: bare search never ranks raw noise above compiled pages; the compile queue, stub list, and citation integrity are one saved query each. The maintenance loop is `uncompiled` → write the wiki page citing its sources → `dead-links` stays empty.
|
|
38
33
|
|
|
39
34
|
## B. Nightly agent memory, consolidated (the dreaming pattern)
|
|
40
35
|
|
|
41
|
-
`memory/` accumulates small notes written at session end, each with `project`,
|
|
42
|
-
`created`, and `kind` (observation / steer / decision) frontmatter. A consolidation agent
|
|
43
|
-
runs periodically: prune, merge, surface contradictions for the human. Retired notes move
|
|
44
|
-
to `archive/` — still queryable, no longer embedded or ranked.
|
|
36
|
+
`memory/` accumulates small notes written at session end, each with `project`, `created`, and `kind` (observation / steer / decision) frontmatter. A consolidation agent runs periodically: prune, merge, surface contradictions for the human. Retired notes move to `archive/`: still queryable under their own preset, out of the default scope.
|
|
45
37
|
|
|
46
38
|
```json
|
|
47
39
|
{
|
|
48
|
-
"version":
|
|
40
|
+
"version": 4,
|
|
49
41
|
"presets": {
|
|
50
42
|
"default": { "include": ["memory/**/*.md"], "k": 10 },
|
|
51
|
-
"archive": { "include": ["archive/**/*.md"], "k": 10
|
|
43
|
+
"archive": { "include": ["archive/**/*.md"], "k": 10 }
|
|
52
44
|
},
|
|
45
|
+
"embed": { "model": "minishlab/potion-retrieval-32M", "type": "static" },
|
|
53
46
|
"queries": {
|
|
54
|
-
"project": "SELECT path, kind, created, title FROM frontmatter WHERE project = ? ORDER BY created DESC",
|
|
55
|
-
"steers": "SELECT path, created, title FROM frontmatter WHERE kind = 'steer' AND project = ? ORDER BY created",
|
|
56
|
-
"retirement": "SELECT path, project, created FROM frontmatter WHERE datetime(created) < datetime('now','-90 day') AND path NOT IN (SELECT dst FROM links WHERE dst IS NOT NULL)",
|
|
57
|
-
"unfiled": "SELECT path FROM frontmatter WHERE project IS NULL"
|
|
47
|
+
"project": { "sql": "SELECT path, kind, created, title FROM frontmatter WHERE project = ? ORDER BY created DESC" },
|
|
48
|
+
"steers": { "sql": "SELECT path, created, title FROM frontmatter WHERE kind = 'steer' AND project = ? ORDER BY created" },
|
|
49
|
+
"retirement": { "sql": "SELECT path, project, created FROM frontmatter WHERE datetime(created) < datetime('now','-90 day') AND path NOT IN (SELECT dst FROM links WHERE dst IS NOT NULL)" },
|
|
50
|
+
"unfiled": { "sql": "SELECT path FROM frontmatter WHERE project IS NULL" }
|
|
58
51
|
}
|
|
59
52
|
}
|
|
60
53
|
```
|
|
@@ -69,29 +62,24 @@ sense retirement # old + uncited → move to arch
|
|
|
69
62
|
sense unfiled # rows are notes missing a project — file them
|
|
70
63
|
```
|
|
71
64
|
|
|
72
|
-
Presets are structural (live vs archived); per-project filtering is metadata (`project = ?`)
|
|
73
|
-
— one tree serves every project. Semantic search over the memory preset is the
|
|
74
|
-
near-duplicate detector: search a new note's own summary and read `similarity` within the
|
|
75
|
-
results. Whether an old steer was overridden is a question for the human — found by
|
|
76
|
-
search, never decided by it.
|
|
65
|
+
Presets are structural (live vs archived); per-project filtering is metadata (`project = ?`). One tree serves every project. Semantic search over the memory preset is the near-duplicate detector: search a new note's own summary and read `similarity` within the results. Whether an old steer was overridden is a question for the human, found by search, never decided by it.
|
|
77
66
|
|
|
78
67
|
## C. Evidence corpus: claims trace to sources
|
|
79
68
|
|
|
80
|
-
`sources/` (immutable imports), `notes/` (one reading note per source), `reviews/`
|
|
81
|
-
(synthesis whose claims must cite notes). The same shape fits incident reports and
|
|
82
|
-
postmortems, user research and findings, due diligence and memos.
|
|
69
|
+
`sources/` (immutable imports), `notes/` (one reading note per source), `reviews/` (synthesis whose claims must cite notes). The same shape fits incident reports and postmortems, user research and findings, due diligence and memos.
|
|
83
70
|
|
|
84
71
|
```json
|
|
85
72
|
{
|
|
86
|
-
"version":
|
|
73
|
+
"version": 4,
|
|
87
74
|
"presets": {
|
|
88
75
|
"default": { "include": ["reviews/**/*.md", "notes/**/*.md"], "k": 10 },
|
|
89
|
-
"source": { "include": ["sources/**/*.md"], "k": 5
|
|
76
|
+
"source": { "include": ["sources/**/*.md"], "k": 5 }
|
|
90
77
|
},
|
|
78
|
+
"embed": { "model": "minishlab/potion-retrieval-32M", "type": "static" },
|
|
91
79
|
"queries": {
|
|
92
|
-
"unsupported": "SELECT path, title FROM frontmatter WHERE path LIKE 'reviews/%' AND path NOT IN (SELECT src FROM links WHERE dst LIKE 'notes/%')",
|
|
93
|
-
"unread": "SELECT path FROM frontmatter WHERE path LIKE 'sources/%' AND path NOT IN (SELECT dst FROM links WHERE dst IS NOT NULL)",
|
|
94
|
-
"by-topic": "SELECT path, title FROM frontmatter WHERE path NOT LIKE 'sources/%' AND has(topics, ?) ORDER BY path"
|
|
80
|
+
"unsupported": { "sql": "SELECT path, title FROM frontmatter WHERE path LIKE 'reviews/%' AND path NOT IN (SELECT src FROM links WHERE dst LIKE 'notes/%')" },
|
|
81
|
+
"unread": { "sql": "SELECT path FROM frontmatter WHERE path LIKE 'sources/%' AND path NOT IN (SELECT dst FROM links WHERE dst IS NOT NULL)" },
|
|
82
|
+
"by-topic": { "sql": "SELECT path, title FROM frontmatter WHERE path NOT LIKE 'sources/%' AND has(topics, ?) ORDER BY path" }
|
|
95
83
|
}
|
|
96
84
|
}
|
|
97
85
|
```
|
|
@@ -105,17 +93,13 @@ sense unread # the reading queue: sources no
|
|
|
105
93
|
|
|
106
94
|
## D. A plain vault: zero configuration
|
|
107
95
|
|
|
108
|
-
Someone else's Obsidian vault, heterogeneous
|
|
109
|
-
`sense init` starter untouched. The workflow is discovery:
|
|
96
|
+
Someone else's Obsidian vault, heterogeneous with no structure worth declaring. The `sense init` starter is left untouched. The workflow is discovery:
|
|
110
97
|
|
|
111
98
|
```
|
|
112
99
|
sense map # fields in use, hub notes, recent changes
|
|
113
100
|
sense search "dataview queries" # words + links + meaning, one ranked list
|
|
114
|
-
sense
|
|
101
|
+
sense sql "SELECT j.value AS tag, COUNT(*) n FROM frontmatter, json_each(frontmatter.tags) j GROUP BY j.value ORDER BY n DESC LIMIT 20"
|
|
115
102
|
sense peek "Plugins/dataview.md" # outline + links before reading
|
|
116
103
|
```
|
|
117
104
|
|
|
118
|
-
Presets earn their place only when a tree has parts deserving different treatment; a tree
|
|
119
|
-
that is one kind of thing needs none of the vocabulary above. On a big vault, raise
|
|
120
|
-
`default`'s `k` and read `lines` ranges instead of whole files — or start from the
|
|
121
|
-
starter's `large` preset.
|
|
105
|
+
Presets earn their place only when a tree has parts deserving different treatment; a tree that is one kind of thing needs none of the vocabulary above. On a big vault, raise `default`'s `k` and read `lines` ranges instead of whole files, or start from the starter's `large` preset.
|
|
@@ -1,110 +1,49 @@
|
|
|
1
1
|
---
|
|
2
2
|
name: sense-setup
|
|
3
|
-
description: Set up the sense CLI on a markdown tree and make the tree-design decisions that shape it
|
|
3
|
+
description: Set up the sense CLI on a markdown tree and make the tree-design decisions that shape it: sense init, presets (which files, which settings, whether the scope searches by meaning), the embed block that names the model, and the trade-offs of frontmatter conventions, summaries, folder layout, and note size. Use when creating or restructuring a markdown knowledge base, running sense init, editing sense.config.json, configuring search scope or vectors, or deciding how notes should be written for an agent to query later.
|
|
4
4
|
---
|
|
5
5
|
|
|
6
6
|
# sense: setup and tree design
|
|
7
7
|
|
|
8
|
-
Querying an existing tree is the `sense` skill. This one covers making a tree:
|
|
9
|
-
installing, writing presets, and the design decisions a tree owner faces.
|
|
10
|
-
Worked configurations for common tree shapes: [EXAMPLES.md](EXAMPLES.md).
|
|
8
|
+
Querying an existing tree is the `sense` skill. This one covers making a tree: installing, writing presets, and the design decisions a tree owner faces. Worked configurations for common tree shapes: [EXAMPLES.md](EXAMPLES.md).
|
|
11
9
|
|
|
12
10
|
## Setup
|
|
13
11
|
|
|
14
|
-
- `npm install -g sensemaking`, then `sense init` at the tree root writes
|
|
15
|
-
`sense.config.json` — two presets (`default`, and `large` showing what a big
|
|
16
|
-
vault tunes), everything on. Config discovery walks up from cwd;
|
|
17
|
-
`--config <path>` overrides.
|
|
12
|
+
- `npm install -g sensemaking`, then `sense init` at the tree root writes `sense.config.json`: two presets (`default`, and `large` showing what a big tree tunes) and an `embed` block naming the model. `sense download` fetches that model once per machine; nothing fetches it implicitly. Config discovery walks up from cwd; `--config <path>` overrides.
|
|
18
13
|
- Globs resolve relative to the config file, never the cwd.
|
|
19
|
-
- `sense status` and `sense map` show each preset's coverage (files matched,
|
|
20
|
-
embedded count), so what a config actually indexes is always visible in
|
|
21
|
-
output. A config edit that changes coverage rebuilds the cache and names the
|
|
22
|
-
preset that caused it on stderr.
|
|
14
|
+
- `sense status` and `sense map` show each preset's coverage (files matched, embedded count), so what a config actually indexes is always visible in output. A config edit that changes coverage rebuilds the cache and names the preset that caused it on stderr.
|
|
23
15
|
|
|
24
16
|
## Presets
|
|
25
17
|
|
|
26
|
-
A preset is a named, self-contained bundle of settings. `default` (required) is
|
|
27
|
-
what bare commands use; every other preset is addressed by name
|
|
28
|
-
(`sense search "..." --preset raw`, or `"preset": "raw"` in a saved search).
|
|
29
|
-
No inheritance: what a preset states is all it does.
|
|
18
|
+
A preset is a named, self-contained bundle of settings. `default` (required) is what bare commands use; every other preset is addressed by name (`sense search "..." --preset raw`, or `"preset": "raw"` in a saved search). No inheritance: what a preset states is all it does.
|
|
30
19
|
|
|
31
20
|
| field | means | default |
|
|
32
21
|
|---|---|---|
|
|
33
22
|
| `include` / `exclude` | which files this preset covers (globs) | required |
|
|
34
23
|
| `k` | how many results a search returns | 10 |
|
|
35
|
-
| `semantic` | vectors for this preset's files and searches | on; only ever written as `false` |
|
|
24
|
+
| `semantic` | vectors for this preset's files and its searches | on; only ever written as `false` |
|
|
36
25
|
| `where` | a standing SQL filter on frontmatter | none |
|
|
37
26
|
|
|
38
|
-
**Indexing derives from presets.** A file is indexed if any preset includes it
|
|
39
|
-
it is embedded if any covering preset has semantic on. Consequences worth
|
|
40
|
-
designing around:
|
|
27
|
+
**Indexing derives from presets.** A file is indexed if any preset includes it. Consequences worth designing around:
|
|
41
28
|
|
|
42
|
-
- A layer of the tree covered only by a `semantic: false` preset (raw sources,
|
|
43
|
-
archives, generated output) is fully searchable lexically and by SQL but
|
|
44
|
-
costs no vector work — the main scale lever.
|
|
45
29
|
- Files no preset includes are not indexed at all.
|
|
46
30
|
- Presets may overlap; they are views, not partitions.
|
|
47
|
-
-
|
|
48
|
-
minutes on tens of thousands of notes, seconds on small trees). Vectors use
|
|
49
|
-
the built-in static model unless a top-level
|
|
50
|
-
`"embed": { "model", "type": "static"|"api", "url", "key" }` block points at
|
|
51
|
-
a Model2Vec model, local path, or OpenAI-compatible endpoint. `static`
|
|
52
|
-
handles paraphrase and reworded concepts; tight domain jargon ("heart
|
|
53
|
-
attack" for "myocardial infarction") is where an `api` transformer model
|
|
54
|
-
tends to do better — measured in BENCHMARKING.md, "Retrieval quality".
|
|
55
|
-
- Global `features` (`links`, `sections`, `rank`) still toggle tree-wide;
|
|
56
|
-
most trees never touch them.
|
|
31
|
+
- Global `features` (`links`, `sections`, `rank`) still toggle tree-wide; most trees never touch them.
|
|
57
32
|
|
|
58
|
-
**
|
|
59
|
-
100k notes with no tuning (BENCHMARKING.md). The knobs that matter are `k`
|
|
60
|
-
(more, smaller results — rows carry `lines` section ranges, so agents read
|
|
61
|
-
sections, not files) and `semantic: false` on the layers that don't earn
|
|
62
|
-
vectors.
|
|
33
|
+
**Vectors take two decisions, in two places.** The top-level `"embed": { "model", "type": "static"|"api", "url", "key" }` block names the model and says whether the tree has vectors at all. A preset's `semantic` says whether that scope uses them: a layer searched for exact wording (ingested sources, archives, generated output) sets `semantic: false`, costs no embedding, and its searches run on words and links. That is the main scale lever, and it is the llm-wiki split: compiled pages searched by meaning, raw sources searched for the phrasing you are citing. `static` is the built-in pure-JS Model2Vec loader and handles paraphrase and reworded concepts; tight domain jargon ("heart attack" for "myocardial infarction") is where an `api` transformer model tends to do better, measured in BENCHMARKING.md, "Retrieval quality". Nothing downloads the model implicitly: `sense download` fetches it once per machine into the cache (`$XDG_CACHE_HOME/sensemaking/models`, else `~/.cache/...`), one directory per model, so several models coexist and switching between them rebuilds the index rather than mixing vector spaces. A `model` holding a path instead of a Hugging Face id points at a local directory, which `sense download` reports as nothing to fetch. The first search after that embeds the tree (progress on stderr; minutes on tens of thousands of notes, seconds on small trees).
|
|
63
34
|
|
|
64
|
-
|
|
35
|
+
**Large vaults**: everything except the vector build is measured linear to 100k notes with no tuning (BENCHMARKING.md). The knobs that matter are `k` (more, smaller results; rows carry `lines` section ranges, so agents read sections, not files) and `semantic: false` on the layers that do not earn vectors.
|
|
65
36
|
|
|
66
|
-
|
|
67
|
-
instruction files of its own; each choice only changes what queries can do.
|
|
37
|
+
## Tree design decisions
|
|
68
38
|
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
- **
|
|
79
|
-
|
|
80
|
-
cache). Volatile state (`status`, `project`, dates) lives in frontmatter and
|
|
81
|
-
filters at query time (`where`, `has()`, `datetime()`). A state worth
|
|
82
|
-
different *indexing* (retired memory, superseded sources) is a state worth
|
|
83
|
-
moving the file — the archive-folder pattern in EXAMPLES.md.
|
|
84
|
-
- **What a note omits is also a filter.** A layer that deliberately carries none
|
|
85
|
-
of the fields the saved views filter on is excluded from all of them without
|
|
86
|
-
any view naming the layer. Sparse fields cut both ways: less of the tree
|
|
87
|
-
filters when you want breadth, and exactly this separation when layers
|
|
88
|
-
differ in authority.
|
|
89
|
-
- **Dates.** `datetime()` comparisons work for dates written as ISO 8601 —
|
|
90
|
-
the only format it parses. A tree that mixes date formats can store them,
|
|
91
|
-
but can't compare them in SQL.
|
|
92
|
-
- **Summaries.** A one-line `summary:` is optional and pays twice: it shows in
|
|
93
|
-
every result row (often answering a question with no file read) and is a
|
|
94
|
-
weighted search field ranked above body text. The cost is writing and
|
|
95
|
-
maintaining the line as notes change.
|
|
96
|
-
- **Folder shape.** Globs find the files, paths are queryable text, links
|
|
97
|
-
resolve by basename at any depth — but presets make folders meaningful:
|
|
98
|
-
a folder is the natural unit that gets its own coverage and settings.
|
|
99
|
-
- **Note size.** Many small notes: precise search hits, whole-file reads stay
|
|
100
|
-
cheap, more links to maintain. Fewer large notes: `sections`, `peek`, and
|
|
101
|
-
the `lines` column carry the cost down to line-range reads. Both work.
|
|
102
|
-
- **Recurring questions.** Save a scenario an agent will repeat: a SQL string
|
|
103
|
-
for filters and reports, or a saved search
|
|
104
|
-
(`"hot": { "search": "...", "preset": "raw", "k": 5 }`). Either runs as
|
|
105
|
-
`sense <name>`, and `sense check` validates both kinds against the real
|
|
106
|
-
tree, so a broken saved scenario fails at check time, not mid-task.
|
|
107
|
-
- **Where decisions live.** Choices that should outlive one conversation can
|
|
108
|
-
be recorded in the agent's own instruction or skill files, or in a note in
|
|
109
|
-
the tree itself; a one-off search over an existing corpus needs none of
|
|
110
|
-
that.
|
|
39
|
+
These belong to the tree's owner. sense works with any of them and reads no instruction files of its own; each choice only changes what queries can do.
|
|
40
|
+
|
|
41
|
+
- **Frontmatter fields.** Columns are discovered per tree: whatever keys notes declare become queryable. Consistent fields across notes make SQL filters and saved queries possible (`WHERE status = 'active'`). SQLite's compiled column limit (2,000; sqlite.org/limits.html) bounds distinct keys per tree. The crawl stops with an error naming the count and the levers. Reserved keys (dropped with a warning): `path`, `_mtime`, `_size`, `_rank`, `content`, `links`, `sections`. Values keep their YAML type: strings TEXT, whole numbers and booleans INTEGER (`true` is 1), fractions REAL, lists and maps JSON text; `map` prints the observed type per field.
|
|
42
|
+
- **Presets are path-shaped; frontmatter is state-shaped.** A preset's coverage must be computable from the path alone (it decides indexing, baked into the cache). Volatile state (`status`, `project`, dates) lives in frontmatter and filters at query time (`where`, `has()`, `datetime()`). A state worth different *indexing* (retired memory, superseded sources) is a state worth moving the file: the archive-folder pattern in EXAMPLES.md.
|
|
43
|
+
- **What a note omits is also a filter.** A layer that deliberately carries none of the fields the saved views filter on is excluded from all of them without any view naming the layer. Sparse fields cut both ways: less of the tree filters when you want breadth, and exactly this separation when layers differ in authority.
|
|
44
|
+
- **Dates.** `datetime()` comparisons work for dates written as ISO 8601, the only format it parses. A tree that mixes date formats can store them, but can't compare them in SQL.
|
|
45
|
+
- **Summaries.** A one-line `summary:` is optional and pays twice: it shows in every result row (often answering a question with no file read) and is a weighted search field ranked above body text. The cost is writing and maintaining the line as notes change.
|
|
46
|
+
- **Folder shape.** Globs find the files, paths are queryable text, links resolve by basename at any depth, but presets make folders meaningful: a folder is the natural unit that gets its own coverage and settings.
|
|
47
|
+
- **Note size.** Many small notes: precise search hits, whole-file reads stay cheap, more links to maintain. Fewer large notes: `sections`, `peek`, and the `lines` column carry the cost down to line-range reads. Both work.
|
|
48
|
+
- **Recurring questions.** Save a scenario an agent will repeat under `queries`, naming the verb it runs: `{ "sql": "..." }` for filters and reports, or `{ "search": "...", "preset": "raw", "k": 5 }` for a ranked search. Either runs as `sense <name>`, and `sense check` validates both kinds against the real tree, so a broken saved scenario fails at check time, not mid-task.
|
|
49
|
+
- **Where decisions live.** Choices that should outlive one conversation can be recorded in the agent's own instruction or skill files, or in a note in the tree itself; a one-off search over an existing corpus needs none of that.
|
package/dist/cjs/cli/query.d.cts
DELETED
package/dist/cjs/cli/query.d.ts
DELETED
|
@@ -1 +0,0 @@
|
|
|
1
|
-
{"version":3,"sources":["/Users/kevin/Dev/OpenSource/ai/sensemaking/src/cli/query.ts"],"sourcesContent":["import { USAGE } from './index.ts';\nimport { CONFIG, FORMAT, formatOf, parse, runSql } from './shared.ts';\nimport type { Command } from './types.ts';\n\nconst query: Command = (ctx) => {\n const usage = `usage: ${ctx.name} ${USAGE.query}`;\n const { values, positionals } = parse(ctx.argv, usage, { ...FORMAT, ...CONFIG });\n const [sql, ...params] = positionals;\n if (!sql) ctx.usageError(usage);\n const format = formatOf(values);\n runSql(ctx.resolveConfig(values.config as string | undefined), sql, params, format, 'ad-hoc query');\n};\nexport default query;\n"],"names":["query","ctx","usage","USAGE","name","parse","argv","FORMAT","CONFIG","values","positionals","sql","params","usageError","format","formatOf","runSql","resolveConfig","config"],"mappings":";;;;+BAYA;;;eAAA;;;uBAZsB;wBACkC;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AAGxD,IAAMA,QAAiB,eAACC;IACtB,IAAMC,QAAQ,AAAC,UAAqBC,OAAZF,IAAIG,IAAI,EAAC,KAAe,OAAZD,cAAK,CAACH,KAAK;IAC/C,IAAgCK,SAAAA,IAAAA,eAAK,EAACJ,IAAIK,IAAI,EAAEJ,OAAO,mBAAKK,gBAAM,EAAKC,gBAAM,IAArEC,SAAwBJ,OAAxBI,QAAQC,cAAgBL,OAAhBK;IAChB,IAAyBA,yBAAAA,cAAlBC,MAAkBD,iBAAb,AAAGE,SAAUF,mBAAb;IACZ,IAAI,CAACC,KAAKV,IAAIY,UAAU,CAACX;IACzB,IAAMY,SAASC,IAAAA,kBAAQ,EAACN;IACxBO,IAAAA,gBAAM,EAACf,IAAIgB,aAAa,CAACR,OAAOS,MAAM,GAAyBP,KAAKC,QAAQE,QAAQ;AACtF;IACA,WAAed"}
|
package/dist/esm/cli/query.d.ts
DELETED
|
@@ -1 +0,0 @@
|
|
|
1
|
-
{"version":3,"sources":["/Users/kevin/Dev/OpenSource/ai/sensemaking/src/cli/query.ts"],"sourcesContent":["import { USAGE } from './index.ts';\nimport { CONFIG, FORMAT, formatOf, parse, runSql } from './shared.ts';\nimport type { Command } from './types.ts';\n\nconst query: Command = (ctx) => {\n const usage = `usage: ${ctx.name} ${USAGE.query}`;\n const { values, positionals } = parse(ctx.argv, usage, { ...FORMAT, ...CONFIG });\n const [sql, ...params] = positionals;\n if (!sql) ctx.usageError(usage);\n const format = formatOf(values);\n runSql(ctx.resolveConfig(values.config as string | undefined), sql, params, format, 'ad-hoc query');\n};\nexport default query;\n"],"names":["USAGE","CONFIG","FORMAT","formatOf","parse","runSql","query","ctx","usage","name","values","positionals","argv","sql","params","usageError","format","resolveConfig","config"],"mappings":"AAAA,SAASA,KAAK,QAAQ,aAAa;AACnC,SAASC,MAAM,EAAEC,MAAM,EAAEC,QAAQ,EAAEC,KAAK,EAAEC,MAAM,QAAQ,cAAc;AAGtE,MAAMC,QAAiB,CAACC;IACtB,MAAMC,QAAQ,CAAC,OAAO,EAAED,IAAIE,IAAI,CAAC,CAAC,EAAET,MAAMM,KAAK,EAAE;IACjD,MAAM,EAAEI,MAAM,EAAEC,WAAW,EAAE,GAAGP,MAAMG,IAAIK,IAAI,EAAEJ,OAAO;QAAE,GAAGN,MAAM;QAAE,GAAGD,MAAM;IAAC;IAC9E,MAAM,CAACY,KAAK,GAAGC,OAAO,GAAGH;IACzB,IAAI,CAACE,KAAKN,IAAIQ,UAAU,CAACP;IACzB,MAAMQ,SAASb,SAASO;IACxBL,OAAOE,IAAIU,aAAa,CAACP,OAAOQ,MAAM,GAAyBL,KAAKC,QAAQE,QAAQ;AACtF;AACA,eAAeV,MAAM"}
|