sensemaking 0.8.0 → 0.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (102) hide show
  1. package/README.md +48 -39
  2. package/dist/cjs/cli/check.js +24 -43
  3. package/dist/cjs/cli/check.js.map +1 -1
  4. package/dist/cjs/cli/index.js +2 -2
  5. package/dist/cjs/cli/index.js.map +1 -1
  6. package/dist/cjs/cli/named.js +18 -8
  7. package/dist/cjs/cli/named.js.map +1 -1
  8. package/dist/cjs/cli/search.d.cts +3 -0
  9. package/dist/cjs/cli/search.d.ts +3 -0
  10. package/dist/cjs/cli/{find.js → search.js} +7 -5
  11. package/dist/cjs/cli/search.js.map +1 -0
  12. package/dist/cjs/cli/shared.js.map +1 -1
  13. package/dist/cjs/cli/status.js +29 -12
  14. package/dist/cjs/cli/status.js.map +1 -1
  15. package/dist/cjs/cli/types.d.cts +3 -1
  16. package/dist/cjs/cli/types.d.ts +3 -1
  17. package/dist/cjs/cli.js +25 -5
  18. package/dist/cjs/cli.js.map +1 -1
  19. package/dist/cjs/commands.d.cts +11 -2
  20. package/dist/cjs/commands.d.ts +11 -2
  21. package/dist/cjs/commands.js +86 -21
  22. package/dist/cjs/commands.js.map +1 -1
  23. package/dist/cjs/config.d.cts +41 -14
  24. package/dist/cjs/config.d.ts +41 -14
  25. package/dist/cjs/config.js +448 -160
  26. package/dist/cjs/config.js.map +1 -1
  27. package/dist/cjs/db.d.cts +1 -1
  28. package/dist/cjs/db.d.ts +1 -1
  29. package/dist/cjs/db.js +208 -62
  30. package/dist/cjs/db.js.map +1 -1
  31. package/dist/cjs/errors.d.cts +1 -1
  32. package/dist/cjs/errors.d.ts +1 -1
  33. package/dist/cjs/errors.js.map +1 -1
  34. package/dist/cjs/features/embed.js +12 -2
  35. package/dist/cjs/features/embed.js.map +1 -1
  36. package/dist/cjs/features/types.d.cts +2 -1
  37. package/dist/cjs/features/types.d.ts +2 -1
  38. package/dist/cjs/index.d.cts +3 -3
  39. package/dist/cjs/index.d.ts +3 -3
  40. package/dist/cjs/index.js +6 -3
  41. package/dist/cjs/index.js.map +1 -1
  42. package/dist/cjs/output.d.cts +15 -0
  43. package/dist/cjs/output.d.ts +15 -0
  44. package/dist/cjs/output.js +26 -5
  45. package/dist/cjs/output.js.map +1 -1
  46. package/dist/cjs/scan.d.cts +3 -0
  47. package/dist/cjs/scan.d.ts +3 -0
  48. package/dist/cjs/scan.js +84 -4
  49. package/dist/cjs/scan.js.map +1 -1
  50. package/dist/esm/cli/check.js +20 -41
  51. package/dist/esm/cli/check.js.map +1 -1
  52. package/dist/esm/cli/index.js +1 -1
  53. package/dist/esm/cli/index.js.map +1 -1
  54. package/dist/esm/cli/named.js +17 -9
  55. package/dist/esm/cli/named.js.map +1 -1
  56. package/dist/esm/cli/search.d.ts +3 -0
  57. package/dist/esm/cli/search.js +16 -0
  58. package/dist/esm/cli/search.js.map +1 -0
  59. package/dist/esm/cli/shared.js +1 -1
  60. package/dist/esm/cli/shared.js.map +1 -1
  61. package/dist/esm/cli/status.js +10 -10
  62. package/dist/esm/cli/status.js.map +1 -1
  63. package/dist/esm/cli/types.d.ts +3 -1
  64. package/dist/esm/cli/types.js.map +1 -1
  65. package/dist/esm/cli.js +20 -4
  66. package/dist/esm/cli.js.map +1 -1
  67. package/dist/esm/commands.d.ts +11 -2
  68. package/dist/esm/commands.js +76 -24
  69. package/dist/esm/commands.js.map +1 -1
  70. package/dist/esm/config.d.ts +41 -14
  71. package/dist/esm/config.js +354 -114
  72. package/dist/esm/config.js.map +1 -1
  73. package/dist/esm/db.d.ts +1 -1
  74. package/dist/esm/db.js +53 -6
  75. package/dist/esm/db.js.map +1 -1
  76. package/dist/esm/errors.d.ts +1 -1
  77. package/dist/esm/errors.js.map +1 -1
  78. package/dist/esm/features/embed.js +12 -2
  79. package/dist/esm/features/embed.js.map +1 -1
  80. package/dist/esm/features/types.d.ts +2 -1
  81. package/dist/esm/features/types.js.map +1 -1
  82. package/dist/esm/index.d.ts +3 -3
  83. package/dist/esm/index.js +1 -1
  84. package/dist/esm/index.js.map +1 -1
  85. package/dist/esm/output.d.ts +15 -0
  86. package/dist/esm/output.js +19 -5
  87. package/dist/esm/output.js.map +1 -1
  88. package/dist/esm/scan.d.ts +3 -0
  89. package/dist/esm/scan.js +32 -4
  90. package/dist/esm/scan.js.map +1 -1
  91. package/package.json +9 -3
  92. package/schema.json +74 -54
  93. package/skills/sense/EXAMPLES.md +8 -6
  94. package/skills/sense/SKILL.md +51 -48
  95. package/skills/sense-setup/EXAMPLES.md +121 -0
  96. package/skills/sense-setup/SKILL.md +71 -66
  97. package/dist/cjs/cli/find.d.cts +0 -3
  98. package/dist/cjs/cli/find.d.ts +0 -3
  99. package/dist/cjs/cli/find.js.map +0 -1
  100. package/dist/esm/cli/find.d.ts +0 -3
  101. package/dist/esm/cli/find.js +0 -14
  102. package/dist/esm/cli/find.js.map +0 -1
@@ -1,82 +1,91 @@
1
1
  ---
2
2
  name: sense-setup
3
- description: Set up the sense CLI on a markdown tree and make the tree-design decisions that shape it — sense init, enabling features (links, sections, rank, embed), and the trade-offs of frontmatter conventions, summaries, folder layout, and note size. Use when creating or restructuring a markdown knowledge base, running sense init, editing sense.config.json, enabling semantic search, or deciding how notes should be written for an agent to query later.
3
+ description: Set up the sense CLI on a markdown tree and make the tree-design decisions that shape it — sense init, presets (which files, which settings, vectors on or off), and the trade-offs of frontmatter conventions, summaries, folder layout, and note size. Use when creating or restructuring a markdown knowledge base, running sense init, editing sense.config.json, configuring search scope or vectors, or deciding how notes should be written for an agent to query later.
4
4
  ---
5
5
 
6
6
  # sense: setup and tree design
7
7
 
8
8
  Querying an existing tree is the `sense` skill. This one covers making a tree:
9
- installing, configuring features, and the design decisions a tree owner faces.
9
+ installing, writing presets, and the design decisions a tree owner faces.
10
+ Worked configurations for common tree shapes: [EXAMPLES.md](EXAMPLES.md).
10
11
 
11
12
  ## Setup
12
13
 
13
14
  - `npm install -g sensemaking`, then `sense init` at the tree root writes
14
- `sense.config.json` (all opt-out features on, `embed` off). Config discovery
15
- walks up from cwd; `--config <path>` overrides.
16
- - `scan.include` globs resolve relative to the config file, never the cwd.
17
- - `sense map` and `sense status` report feature state
18
- (`features: links, sections, rank · off: embed (features.embed)`), so the
19
- current config is always discoverable from output.
15
+ `sense.config.json` — two presets (`default`, and `large` showing what a big
16
+ vault tunes), everything on. Config discovery walks up from cwd;
17
+ `--config <path>` overrides.
18
+ - Globs resolve relative to the config file, never the cwd.
19
+ - `sense status` and `sense map` show each preset's coverage (files matched,
20
+ embedded count), so what a config actually indexes is always visible in
21
+ output. A config edit that changes coverage rebuilds the cache and names the
22
+ preset that caused it on stderr.
20
23
 
21
- ## Features
24
+ ## Presets
22
25
 
23
- | feature | powers | cost when on | config key |
24
- |---|---|---|---|
25
- | `links` | backlinks, dead-link queries, `find`'s link expansion, `peek`'s link lists | link re-resolution at reconcile | `features.links` (default on) |
26
- | `sections` | `peek`'s outline, line-range reads, per-section token estimates | heading extraction at parse | `features.sections` (default on) |
27
- | `rank` | `map`'s hubs, `_rank` in any ORDER BY | PageRank pass at reconcile; requires `links` | `features.rank` (default on) |
28
- | `embed` | `find --semantic` (meaning-based expansion) | the first semantic query embeds the whole tree (minutes on tens of thousands of notes, progress on stderr); after that, a model load per invocation (measured in BENCHMARKING.md as `semantic_find_ms` minus `find_ms`) | `features.embed` (default off) |
26
+ A preset is a named, self-contained bundle of settings. `default` (required) is
27
+ what bare commands use; every other preset is addressed by name
28
+ (`sense search "..." --preset raw`, or `"preset": "raw"` in a saved search).
29
+ No inheritance: what a preset states is all it does.
29
30
 
30
- - `embed` accepts `true` (built-in static model) or an object: `model` (Hugging
31
- Face id or local path), `type` (`static` pure-JS built-in, or `api` for any
32
- OpenAI-compatible `/v1/embeddings` endpoint — Ollama, LM Studio, llama.cpp,
33
- hosted), `url`, `key` (env var name). A local `model` path is fully offline.
34
- - The two types differ in what they match, not only in cost: `static` is a
35
- context-free distilled model that handles paraphrase and reworded concepts
36
- (a query like "delegating without micromanaging" can reach a note that uses
37
- neither word); tight near-synonyms and domain jargon ("heart attack" for
38
- "myocardial infarction") are where it misses and an `api` transformer model
39
- tends to succeed. The gain concentrates where the searcher's vocabulary
40
- differs from the notes' — measured in the retrieval eval (BENCHMARKING.md,
41
- "Retrieval quality"): on a vocabulary-gap corpus semantic expansion adds
42
- recall; where vocabulary overlaps, BM25 plus link expansion already answers
43
- most queries and vectors mostly reorder.
44
- - Enabling `embed` changes no default `find` result — expansion runs only when
45
- a query passes `--semantic`. Invoking `--semantic` on a tree without `embed`
46
- is an error naming the config key.
47
- - Toggling any feature rebuilds the cache on the next query (safe, automatic).
48
- - Disabled features degrade output, visibly: `peek` prints
49
- `sections: off (features.sections)` rather than an empty outline.
50
- - Every query reconciles for itself, so nothing has to be running. On a
51
- large tree that changes in bulk — a sync, a generated batch — whoever
52
- queries next pays that re-parse; `sense watch` moves it into the
53
- background instead. It changes latency, never answers, and the OS
54
- supervises it rather than the CLI (WATCH.md ships launchd and systemd
55
- units).
31
+ | field | means | default |
32
+ |---|---|---|
33
+ | `include` / `exclude` | which files this preset covers (globs) | required |
34
+ | `k` | how many results a search returns | 10 |
35
+ | `semantic` | vectors for this preset's files and searches | on; only ever written as `false` |
36
+ | `where` | a standing SQL filter on frontmatter | none |
37
+
38
+ **Indexing derives from presets.** A file is indexed if any preset includes it;
39
+ it is embedded if any covering preset has semantic on. Consequences worth
40
+ designing around:
41
+
42
+ - A layer of the tree covered only by a `semantic: false` preset (raw sources,
43
+ archives, generated output) is fully searchable lexically and by SQL but
44
+ costs no vector work — the main scale lever.
45
+ - Files no preset includes are not indexed at all.
46
+ - Presets may overlap; they are views, not partitions.
47
+ - The first semantic search embeds everything covered (progress on stderr;
48
+ minutes on tens of thousands of notes, seconds on small trees). Vectors use
49
+ the built-in static model unless a top-level
50
+ `"embed": { "model", "type": "static"|"api", "url", "key" }` block points at
51
+ a Model2Vec model, local path, or OpenAI-compatible endpoint. `static`
52
+ handles paraphrase and reworded concepts; tight domain jargon ("heart
53
+ attack" for "myocardial infarction") is where an `api` transformer model
54
+ tends to do better — measured in BENCHMARKING.md, "Retrieval quality".
55
+ - Global `features` (`links`, `sections`, `rank`) still toggle tree-wide;
56
+ most trees never touch them.
57
+
58
+ **Large vaults**: everything except the vector build is measured linear to
59
+ 100k notes with no tuning (BENCHMARKING.md). The knobs that matter are `k`
60
+ (more, smaller results — rows carry `lines` section ranges, so agents read
61
+ sections, not files) and `semantic: false` on the layers that don't earn
62
+ vectors.
56
63
 
57
64
  ## Tree design decisions
58
65
 
59
66
  These belong to the tree's owner. sense works with any of them and reads no
60
- instruction files of its own; each choice below only changes what queries can
61
- do, and every consequence is listed so the choice can be made deliberately.
67
+ instruction files of its own; each choice only changes what queries can do.
62
68
 
63
69
  - **Frontmatter fields.** Columns are discovered per tree — whatever keys notes
64
70
  declare become queryable. Consistent fields across notes make SQL filters
65
71
  and named queries possible (`WHERE status = 'active'`). SQLite's compiled
66
72
  column limit (2,000; sqlite.org/limits.html) bounds distinct keys per tree —
67
- the crawl stops with an error naming the count and the levers, which in
68
- practice only a generator writing unbounded keys ever hits. Reserved keys
73
+ the crawl stops with an error naming the count and the levers. Reserved keys
69
74
  (dropped with a warning): `path`, `_mtime`, `_size`, `_rank`, `content`,
70
75
  `links`, `sections`. Values keep their YAML type: strings TEXT, whole numbers
71
76
  and booleans INTEGER (`true` is 1), fractions REAL, lists and maps JSON text;
72
77
  `map` prints the observed type per field.
78
+ - **Presets are path-shaped; frontmatter is state-shaped.** A preset's coverage
79
+ must be computable from the path alone (it decides indexing, baked into the
80
+ cache). Volatile state (`status`, `project`, dates) lives in frontmatter and
81
+ filters at query time (`where`, `has()`, `datetime()`). A state worth
82
+ different *indexing* (retired memory, superseded sources) is a state worth
83
+ moving the file — the archive-folder pattern in EXAMPLES.md.
73
84
  - **What a note omits is also a filter.** A layer that deliberately carries none
74
85
  of the fields the saved views filter on is excluded from all of them without
75
- any view naming the layer; only a view filtering solely on a field the layer
76
- does share needs an explicit condition (`AND type != 'raw'`). Sparse fields cut
77
- both ways: less of the tree filters when you want breadth, and exactly this
78
- separation when layers differ in authority — source extractions vs.
79
- conclusions, generated output vs. notes.
86
+ any view naming the layer. Sparse fields cut both ways: less of the tree
87
+ filters when you want breadth, and exactly this separation when layers
88
+ differ in authority.
80
89
  - **Dates.** `datetime()` comparisons work for dates written as ISO 8601 —
81
90
  the only format it parses. A tree that mixes date formats can store them,
82
91
  but can't compare them in SQL.
@@ -84,22 +93,18 @@ do, and every consequence is listed so the choice can be made deliberately.
84
93
  every result row (often answering a question with no file read) and is a
85
94
  weighted search field ranked above body text. The cost is writing and
86
95
  maintaining the line as notes change.
87
- - **Folder shape.** sense is structure-indifferent: globs find the files,
88
- paths are queryable text, links resolve by basename at any depth. Folders
89
- are for the humans and agents navigating the tree, not for the index —
90
- flat and nested trees query identically.
91
- - **Note size.** Many small notes: precise `find` hits, whole-file reads stay
92
- cheap, more links to maintain. Fewer large notes: `sections` and `peek`
93
- carry the cost down to line-range reads. Both work; per-section token
94
- estimates exist either way.
95
- - **Recurring questions.** A scenario an agent will run repeatedly can be saved
96
- with its settings: a SQL string for filters and reports, or a saved find
97
- (`"hot": { "find": "...", "k": 20, "semantic": true }`) for searches — either
98
- runs as `sense <name>`, and `sense check` validates both kinds against the
99
- real tree, so a broken saved scenario fails at check time, not mid-task.
96
+ - **Folder shape.** Globs find the files, paths are queryable text, links
97
+ resolve by basename at any depth — but presets make folders meaningful:
98
+ a folder is the natural unit that gets its own coverage and settings.
99
+ - **Note size.** Many small notes: precise search hits, whole-file reads stay
100
+ cheap, more links to maintain. Fewer large notes: `sections`, `peek`, and
101
+ the `lines` column carry the cost down to line-range reads. Both work.
102
+ - **Recurring questions.** Save a scenario an agent will repeat: a SQL string
103
+ for filters and reports, or a saved search
104
+ (`"hot": { "search": "...", "preset": "raw", "k": 5 }`). Either runs as
105
+ `sense <name>`, and `sense check` validates both kinds against the real
106
+ tree, so a broken saved scenario fails at check time, not mid-task.
100
107
  - **Where decisions live.** Choices that should outlive one conversation can
101
108
  be recorded in the agent's own instruction or skill files, or in a note in
102
109
  the tree itself; a one-off search over an existing corpus needs none of
103
- that. Whether to settle a choice with the user or proceed on the corpus as
104
- found depends on whether the agent is only querying or also authoring —
105
- an authoring agent's choices compound; a querying agent's don't.
110
+ that.
@@ -1,3 +0,0 @@
1
- import type { Command } from './types.js';
2
- declare const findCmd: Command;
3
- export default findCmd;
@@ -1,3 +0,0 @@
1
- import type { Command } from './types.js';
2
- declare const findCmd: Command;
3
- export default findCmd;
@@ -1 +0,0 @@
1
- {"version":3,"sources":["/Users/kevin/Dev/OpenSource/ai/sensemaking/src/cli/find.ts"],"sourcesContent":["import { find } from '../commands.ts';\nimport { printRows } from '../output.ts';\nimport { parseK, withDb } from './shared.ts';\nimport type { Command } from './types.ts';\n\nconst findCmd: Command = async (ctx) => {\n const [terms] = ctx.rest;\n if (!terms) ctx.usageError(`usage: ${ctx.name} find \"<terms>\" [--where \"<sql>\"] [--k n] [--semantic]`);\n const k = parseK(ctx);\n await withDb(ctx, async (db, cfg) => printRows(await find(db, cfg, terms, { k, where: ctx.values.where, semantic: ctx.values.semantic }), ctx.format));\n};\nexport default findCmd;\n"],"names":["findCmd","ctx","terms","k","rest","usageError","name","parseK","withDb","db","cfg","find","where","values","semantic","printRows","format"],"mappings":";;;;+BAWA;;;eAAA;;;0BAXqB;wBACK;wBACK;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AAG/B,IAAMA,UAAmB,iBAAOC;;YACdA,WAATC,OAEDC;;;;oBAFUF,6BAAAA,IAAIG,IAAI,MAAjBF,QAASD;oBAChB,IAAI,CAACC,OAAOD,IAAII,UAAU,CAAC,AAAC,UAAkB,OAATJ,IAAIK,IAAI,EAAC;oBACxCH,IAAII,IAAAA,gBAAM,EAACN;oBACjB;;wBAAMO,IAAAA,gBAAM,EAACP,KAAK,SAAOQ,IAAIC;;;;;4CAAkB;;gDAAMC,IAAAA,gBAAI,EAACF,IAAIC,KAAKR,OAAO;oDAAEC,GAAAA;oDAAGS,OAAOX,IAAIY,MAAM,CAACD,KAAK;oDAAEE,UAAUb,IAAIY,MAAM,CAACC,QAAQ;gDAAC;;;;;gDAAjGC,mBAAS;oDAAC;oDAA2Fd,IAAIe,MAAM;;;;;;;;;oBAApJ;;;;;;IACF;;IACA,WAAehB"}
@@ -1,3 +0,0 @@
1
- import type { Command } from './types.js';
2
- declare const findCmd: Command;
3
- export default findCmd;
@@ -1,14 +0,0 @@
1
- import { find } from '../commands.js';
2
- import { printRows } from '../output.js';
3
- import { parseK, withDb } from './shared.js';
4
- const findCmd = async (ctx)=>{
5
- const [terms] = ctx.rest;
6
- if (!terms) ctx.usageError(`usage: ${ctx.name} find "<terms>" [--where "<sql>"] [--k n] [--semantic]`);
7
- const k = parseK(ctx);
8
- await withDb(ctx, async (db, cfg)=>printRows(await find(db, cfg, terms, {
9
- k,
10
- where: ctx.values.where,
11
- semantic: ctx.values.semantic
12
- }), ctx.format));
13
- };
14
- export default findCmd;
@@ -1 +0,0 @@
1
- {"version":3,"sources":["/Users/kevin/Dev/OpenSource/ai/sensemaking/src/cli/find.ts"],"sourcesContent":["import { find } from '../commands.ts';\nimport { printRows } from '../output.ts';\nimport { parseK, withDb } from './shared.ts';\nimport type { Command } from './types.ts';\n\nconst findCmd: Command = async (ctx) => {\n const [terms] = ctx.rest;\n if (!terms) ctx.usageError(`usage: ${ctx.name} find \"<terms>\" [--where \"<sql>\"] [--k n] [--semantic]`);\n const k = parseK(ctx);\n await withDb(ctx, async (db, cfg) => printRows(await find(db, cfg, terms, { k, where: ctx.values.where, semantic: ctx.values.semantic }), ctx.format));\n};\nexport default findCmd;\n"],"names":["find","printRows","parseK","withDb","findCmd","ctx","terms","rest","usageError","name","k","db","cfg","where","values","semantic","format"],"mappings":"AAAA,SAASA,IAAI,QAAQ,iBAAiB;AACtC,SAASC,SAAS,QAAQ,eAAe;AACzC,SAASC,MAAM,EAAEC,MAAM,QAAQ,cAAc;AAG7C,MAAMC,UAAmB,OAAOC;IAC9B,MAAM,CAACC,MAAM,GAAGD,IAAIE,IAAI;IACxB,IAAI,CAACD,OAAOD,IAAIG,UAAU,CAAC,CAAC,OAAO,EAAEH,IAAII,IAAI,CAAC,sDAAsD,CAAC;IACrG,MAAMC,IAAIR,OAAOG;IACjB,MAAMF,OAAOE,KAAK,OAAOM,IAAIC,MAAQX,UAAU,MAAMD,KAAKW,IAAIC,KAAKN,OAAO;YAAEI;YAAGG,OAAOR,IAAIS,MAAM,CAACD,KAAK;YAAEE,UAAUV,IAAIS,MAAM,CAACC,QAAQ;QAAC,IAAIV,IAAIW,MAAM;AACtJ;AACA,eAAeZ,QAAQ"}