sensemaking 0.10.0 → 0.11.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +6 -7
- package/dist/cjs/cli/index.d.cts +1 -3
- package/dist/cjs/cli/index.d.ts +1 -3
- package/dist/cjs/cli/index.js +1 -13
- package/dist/cjs/cli/index.js.map +1 -1
- package/dist/cjs/cli/named.js +3 -1
- package/dist/cjs/cli/named.js.map +1 -1
- package/dist/cjs/cli/shared.d.cts +1 -1
- package/dist/cjs/cli/shared.d.ts +1 -1
- package/dist/cjs/cli/shared.js +21 -1
- package/dist/cjs/cli/shared.js.map +1 -1
- package/dist/cjs/cli/sql.js +30 -2
- package/dist/cjs/cli/sql.js.map +1 -1
- package/dist/cjs/cli/status.d.cts +1 -0
- package/dist/cjs/cli/status.d.ts +1 -0
- package/dist/cjs/cli/status.js +76 -13
- package/dist/cjs/cli/status.js.map +1 -1
- package/dist/cjs/commands.d.cts +2 -1
- package/dist/cjs/commands.d.ts +2 -1
- package/dist/cjs/commands.js +32 -12
- package/dist/cjs/commands.js.map +1 -1
- package/dist/cjs/db.d.cts +2 -2
- package/dist/cjs/db.d.ts +2 -2
- package/dist/cjs/db.js +11 -12
- package/dist/cjs/db.js.map +1 -1
- package/dist/cjs/errors.d.cts +1 -1
- package/dist/cjs/errors.d.ts +1 -1
- package/dist/cjs/errors.js.map +1 -1
- package/dist/cjs/features/embed.d.cts +1 -0
- package/dist/cjs/features/embed.d.ts +1 -0
- package/dist/cjs/features/embed.js +31 -9
- package/dist/cjs/features/embed.js.map +1 -1
- package/dist/cjs/index.d.cts +1 -1
- package/dist/cjs/index.d.ts +1 -1
- package/dist/cjs/index.js +3 -3
- package/dist/cjs/index.js.map +1 -1
- package/dist/cjs/output.d.cts +3 -0
- package/dist/cjs/output.d.ts +3 -0
- package/dist/cjs/output.js +6 -1
- package/dist/cjs/output.js.map +1 -1
- package/dist/cjs/scan.d.cts +1 -0
- package/dist/cjs/scan.d.ts +1 -0
- package/dist/cjs/scan.js +83 -12
- package/dist/cjs/scan.js.map +1 -1
- package/dist/esm/cli/index.d.ts +1 -3
- package/dist/esm/cli/index.js +1 -5
- package/dist/esm/cli/index.js.map +1 -1
- package/dist/esm/cli/named.js +3 -1
- package/dist/esm/cli/named.js.map +1 -1
- package/dist/esm/cli/shared.d.ts +1 -1
- package/dist/esm/cli/shared.js +21 -1
- package/dist/esm/cli/shared.js.map +1 -1
- package/dist/esm/cli/sql.js +5 -2
- package/dist/esm/cli/sql.js.map +1 -1
- package/dist/esm/cli/status.d.ts +1 -0
- package/dist/esm/cli/status.js +70 -13
- package/dist/esm/cli/status.js.map +1 -1
- package/dist/esm/commands.d.ts +2 -1
- package/dist/esm/commands.js +30 -10
- package/dist/esm/commands.js.map +1 -1
- package/dist/esm/db.d.ts +2 -2
- package/dist/esm/db.js +12 -10
- package/dist/esm/db.js.map +1 -1
- package/dist/esm/errors.d.ts +1 -1
- package/dist/esm/errors.js.map +1 -1
- package/dist/esm/features/embed.d.ts +1 -0
- package/dist/esm/features/embed.js +29 -9
- package/dist/esm/features/embed.js.map +1 -1
- package/dist/esm/index.d.ts +1 -1
- package/dist/esm/index.js +1 -1
- package/dist/esm/index.js.map +1 -1
- package/dist/esm/output.d.ts +3 -0
- package/dist/esm/output.js +6 -1
- package/dist/esm/output.js.map +1 -1
- package/dist/esm/scan.d.ts +1 -0
- package/dist/esm/scan.js +82 -13
- package/dist/esm/scan.js.map +1 -1
- package/package.json +1 -1
- package/schema.json +1 -1
- package/skills/sense/SKILL.md +16 -10
- package/skills/sense-setup/SKILL.md +2 -2
- package/dist/cjs/cli/check.d.cts +0 -3
- package/dist/cjs/cli/check.d.ts +0 -3
- package/dist/cjs/cli/check.js +0 -392
- package/dist/cjs/cli/check.js.map +0 -1
- package/dist/cjs/cli/rebuild.d.cts +0 -3
- package/dist/cjs/cli/rebuild.d.ts +0 -3
- package/dist/cjs/cli/rebuild.js +0 -50
- package/dist/cjs/cli/rebuild.js.map +0 -1
- package/dist/esm/cli/check.d.ts +0 -3
- package/dist/esm/cli/check.js +0 -88
- package/dist/esm/cli/check.js.map +0 -1
- package/dist/esm/cli/rebuild.d.ts +0 -3
- package/dist/esm/cli/rebuild.js +0 -13
- package/dist/esm/cli/rebuild.js.map +0 -1
package/dist/esm/scan.js
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
import { globSync, readFileSync, statSync } from 'node:fs';
|
|
2
2
|
import { join, sep } from 'node:path';
|
|
3
3
|
import removeMarkdown from 'remove-markdown';
|
|
4
|
-
import { parseDocument } from 'yaml';
|
|
4
|
+
import { isCollection, parseDocument, visit } from 'yaml';
|
|
5
5
|
import { embedEnabled, presetNames, presetSemanticEnabled } from './config.js';
|
|
6
6
|
// Filesystem -> rows. Pure data in, data + warnings out; db.ts does the SQL.
|
|
7
7
|
// Frontmatter keys that would collide with table columns. Exported so db.ts's upsert can tell
|
|
@@ -11,10 +11,20 @@ export const RESERVED_COLUMNS = new Set([
|
|
|
11
11
|
'_mtime',
|
|
12
12
|
'_size',
|
|
13
13
|
'_rank',
|
|
14
|
+
'_parse_error',
|
|
14
15
|
'content',
|
|
15
16
|
'links',
|
|
16
17
|
'sections'
|
|
17
18
|
]);
|
|
19
|
+
// YAML error codes whose recovery is unambiguous, so the parse is accepted rather than
|
|
20
|
+
// quarantined. Only one qualifies: YAML 1.2 reserves `@` and `` ` `` at the start of a plain
|
|
21
|
+
// scalar for future use, so they can never be valid and the text can only be what was typed
|
|
22
|
+
// (`aliases: [@handle]` -> ["@handle"]). Every other code has a second reading -- an unquoted
|
|
23
|
+
// `:` swallows the keys after it, an unquoted `[..](..)` drops the URL, a duplicate key picks
|
|
24
|
+
// one value in silence -- so it writes values nobody wrote. See plans/frontmatter-parse-policy.md.
|
|
25
|
+
const ACCEPTED_YAML_CODES = new Set([
|
|
26
|
+
'BAD_SCALAR_START'
|
|
27
|
+
]);
|
|
18
28
|
function normalizeText(value) {
|
|
19
29
|
if (value === null || value === undefined) return '';
|
|
20
30
|
return String(value).replace(/\s+/g, ' ').trim();
|
|
@@ -108,32 +118,90 @@ function splitFrontmatter(raw) {
|
|
|
108
118
|
body: rest.slice(close.index + close[0].length)
|
|
109
119
|
};
|
|
110
120
|
}
|
|
111
|
-
//
|
|
112
|
-
//
|
|
121
|
+
// A well-formed document can still hold a value nobody meant: `created: {{date}}` is valid
|
|
122
|
+
// YAML for a flow map used as a mapping key, so it raises no error and stores
|
|
123
|
+
// {"{ date }": null}. No error code can catch that, but yaml notices the stringified key, so
|
|
124
|
+
// this reports it with the path instead (yaml's own warning has none, fires once per document,
|
|
125
|
+
// and is what trains readers to discard stderr).
|
|
126
|
+
function warnStringifiedKeys(relPath, doc, warnings) {
|
|
127
|
+
let found = false;
|
|
128
|
+
// Nested, not top level: `created: {{date}}` puts the collection key one level down, inside
|
|
129
|
+
// the flow map that `{{...}}` parses as.
|
|
130
|
+
visit(doc, {
|
|
131
|
+
Pair (_key, pair) {
|
|
132
|
+
if (!isCollection(pair.key)) return undefined;
|
|
133
|
+
found = true;
|
|
134
|
+
return visit.BREAK;
|
|
135
|
+
}
|
|
136
|
+
});
|
|
137
|
+
// One per file: a template repeats the same mistake on every field it stamps.
|
|
138
|
+
if (found) warnings.push(`warning: ${relPath} frontmatter has a key that is itself a list or mapping, stored as text; this is usually an unrendered template placeholder like {{date}}`);
|
|
139
|
+
}
|
|
140
|
+
// Accept a clean parse, and one whose every error is unambiguous (ACCEPTED_YAML_CODES).
|
|
141
|
+
// Anything else is quarantined: no frontmatter columns at all, and `_parse_error` carries the
|
|
142
|
+
// reason. Recovering it would write values nobody wrote, which is worse than absence because
|
|
143
|
+
// no query can see it. The file is still indexed -- content, links and sections never touch
|
|
144
|
+
// frontmatter -- so a broken note stays searchable while it is being hunted for.
|
|
145
|
+
// yaml's message continues onto a source excerpt, so the first line is the sentence -- minus
|
|
146
|
+
// the colon that introduced the part being dropped.
|
|
147
|
+
function firstLine(message) {
|
|
148
|
+
return message.split('\n')[0].replace(/:\s*$/, '');
|
|
149
|
+
}
|
|
113
150
|
function parseFrontmatter(relPath, fm, warnings) {
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
151
|
+
// logLevel silences yaml's own pathless warnings; warnStringifiedKeys re-reports the one
|
|
152
|
+
// that carries information, with the file it came from.
|
|
153
|
+
const doc = parseDocument(fm, {
|
|
154
|
+
logLevel: 'silent'
|
|
155
|
+
});
|
|
156
|
+
const refused = doc.errors.filter((err)=>!ACCEPTED_YAML_CODES.has(err.code));
|
|
157
|
+
if (refused.length > 0) {
|
|
158
|
+
const detail = refused.length > 1 ? ` (and ${refused.length - 1} more)` : '';
|
|
159
|
+
const parseError = `${firstLine(refused[0].message)}${detail}`;
|
|
160
|
+
warnings.push(`warning: ${relPath} frontmatter did not parse, so none of it is indexed: ${parseError}`);
|
|
161
|
+
return {
|
|
162
|
+
data: {},
|
|
163
|
+
parseError
|
|
164
|
+
};
|
|
117
165
|
}
|
|
118
166
|
let data;
|
|
119
167
|
try {
|
|
120
168
|
data = doc.toJS();
|
|
121
169
|
} catch (err) {
|
|
122
|
-
|
|
123
|
-
|
|
170
|
+
// Reaches here with doc.errors empty: `title: **Bold**` parses, then opens an alias on
|
|
171
|
+
// materialisation. An empty error list is not a successful parse.
|
|
172
|
+
const parseError = firstLine(err.message);
|
|
173
|
+
warnings.push(`warning: ${relPath} frontmatter did not parse, so none of it is indexed: ${parseError}`);
|
|
174
|
+
return {
|
|
175
|
+
data: {},
|
|
176
|
+
parseError
|
|
177
|
+
};
|
|
124
178
|
}
|
|
125
|
-
if (data === null || data === undefined) return {
|
|
179
|
+
if (data === null || data === undefined) return {
|
|
180
|
+
data: {},
|
|
181
|
+
parseError: null
|
|
182
|
+
};
|
|
126
183
|
if (typeof data !== 'object' || Array.isArray(data)) {
|
|
127
|
-
|
|
128
|
-
|
|
184
|
+
const parseError = 'frontmatter is not a key-value mapping';
|
|
185
|
+
warnings.push(`warning: ${relPath} ${parseError}; none of it is indexed`);
|
|
186
|
+
return {
|
|
187
|
+
data: {},
|
|
188
|
+
parseError
|
|
189
|
+
};
|
|
129
190
|
}
|
|
130
|
-
|
|
191
|
+
warnStringifiedKeys(relPath, doc, warnings);
|
|
192
|
+
return {
|
|
193
|
+
data: data,
|
|
194
|
+
parseError: null
|
|
195
|
+
};
|
|
131
196
|
}
|
|
132
197
|
export function parseFile(file, extractors = []) {
|
|
133
198
|
const raw = readFileSync(file.absPath, 'utf8');
|
|
134
199
|
const warnings = [];
|
|
135
200
|
const { fm, body: content } = splitFrontmatter(raw);
|
|
136
|
-
const data = fm === null ? {
|
|
201
|
+
const { data, parseError } = fm === null ? {
|
|
202
|
+
data: {},
|
|
203
|
+
parseError: null
|
|
204
|
+
} : parseFrontmatter(file.relPath, fm, warnings);
|
|
137
205
|
const mapped = {};
|
|
138
206
|
for (const key of Object.keys(data)){
|
|
139
207
|
if (RESERVED_COLUMNS.has(key)) {
|
|
@@ -156,6 +224,7 @@ export function parseFile(file, extractors = []) {
|
|
|
156
224
|
size: file.size,
|
|
157
225
|
presets: file.presets,
|
|
158
226
|
data: mapped,
|
|
227
|
+
parseError,
|
|
159
228
|
search,
|
|
160
229
|
extracted: Object.fromEntries(extractors.filter((f)=>f.extract).map((f)=>{
|
|
161
230
|
var _f_extract;
|
package/dist/esm/scan.js.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"sources":["/Users/kevin/Dev/OpenSource/ai/sensemaking/src/scan.ts"],"sourcesContent":["import { globSync, readFileSync, statSync } from 'node:fs';\nimport { join, sep } from 'node:path';\nimport removeMarkdown from 'remove-markdown';\nimport { parseDocument } from 'yaml';\nimport type { Config } from './config.ts';\nimport { embedEnabled, presetNames, presetSemanticEnabled } from './config.ts';\nimport type { Feature } from './features/types.ts';\n\n// Filesystem -> rows. Pure data in, data + warnings out; db.ts does the SQL.\n\n// Frontmatter keys that would collide with table columns. Exported so db.ts's upsert can tell\n// a feature-owned column (`_rank`) from a parsed one and leave it alone on reparse.\nexport const RESERVED_COLUMNS = new Set(['path', '_mtime', '_size', '_rank', 'content', 'links', 'sections']);\n\nfunction normalizeText(value: unknown): string {\n if (value === null || value === undefined) return '';\n return String(value).replace(/\\s+/g, ' ').trim();\n}\n\n// Keeps URL query strings, asset filenames, and HTML attributes out of the index: rare terms\n// carry high IDF, so they outrank prose. remove-markdown misses wikilinks and tables.\nfunction stripText(value: string): string {\n const withoutWikilinks = value.replace(/\\[\\[([^\\]|]+)\\|([^\\]]+)\\]\\]/g, '$2').replace(/\\[\\[([^\\]]+)\\]\\]/g, '$1');\n const withoutMarkdown = removeMarkdown(withoutWikilinks);\n const withoutTables = withoutMarkdown.replace(/^\\s*\\|?[-\\s|:]+\\|\\s*$/gm, '').replace(/\\|/g, ' ');\n return normalizeText(withoutTables);\n}\n\nexport interface FileStat {\n relPath: string;\n absPath: string;\n mtimeMs: number;\n size: number;\n presets: string[]; // every declared preset covering this file (>= 1; union, overlap allowed)\n embed: boolean; // true iff a model is named and some covering preset has semantic on\n}\n\n// Presets are views, not partitions: they overlap freely, and a file's covering set (not one\n// owner) drives indexing. Globs resolve relative to baseDir; unmatched files are not indexed.\nexport function toPosixPath(relPath: string, separator: string = sep): string {\n return separator === '\\\\' ? relPath.split(separator).join('/') : relPath;\n}\n\n// Every command pays listFiles before it answers (the freshness check stats each file), so\n// per-file work here is the hottest path in the package. Everything derivable from the config\n// alone is computed once, above the loop.\nconst NO_THROW = { throwIfNoEntry: false } as const;\n\nexport function listFiles(cfg: Config, baseDir: string): FileStat[] {\n const coverage = new Map<string, Set<string>>();\n const posixNeeded = sep === '\\\\';\n for (const name of presetNames(cfg)) {\n const preset = cfg.presets[name];\n for (const matched of globSync(preset.include, { cwd: baseDir, exclude: preset.exclude })) {\n const relPath = posixNeeded ? toPosixPath(matched) : matched;\n const set = coverage.get(relPath) ?? new Set<string>();\n set.add(name);\n coverage.set(relPath, set);\n }\n }\n\n // Which presets want vectors is a property of the config, not of any file.\n const embedding = embedEnabled(cfg);\n const semanticPresets = embedding ? new Set(presetNames(cfg).filter((name) => presetSemanticEnabled(cfg, name))) : null;\n\n const files: FileStat[] = [];\n for (const relPath of [...coverage.keys()].sort()) {\n const absPath = join(baseDir, relPath); // join re-applies the platform separator for fs calls\n // node:fs glob matches directories and dangling symlinks; fast-glob returned neither, so\n // one stat filters both back out (throwIfNoEntry keeps a dangling link from throwing).\n const st = statSync(absPath, NO_THROW);\n if (!st?.isFile()) continue;\n const presets = [...(coverage.get(relPath) as Set<string>)].sort();\n const embed = semanticPresets !== null && presets.some((name) => semanticPresets.has(name));\n files.push({ relPath, absPath, mtimeMs: st.mtimeMs, size: st.size, presets, embed });\n }\n return files;\n}\n\nexport interface ParsedDoc {\n relPath: string;\n mtimeMs: number;\n size: number;\n presets: string[];\n data: Record<string, string | number | bigint | null>;\n // title/summary are duplicated from frontmatter so bm25() can weight them above the body text.\n search: { title: string; summary: string; text: string };\n // Per-feature extraction results, keyed by feature name; features store them at reconcile.\n extracted: Record<string, unknown>;\n}\n\n// Storage class follows the YAML scalar. Booleans store as 1/0, so `WHERE flag = 1` matches\n// and `WHERE flag = 'true'` cannot; `map` prints observed types so the mismatch is visible.\nfunction mapValue(value: unknown): string | number | bigint | null {\n if (value === null || value === undefined) return null;\n if (typeof value === 'boolean') return BigInt(value ? 1 : 0);\n if (typeof value === 'number') return Number.isSafeInteger(value) ? BigInt(value) : value;\n if (typeof value === 'string') return value;\n return JSON.stringify(value);\n}\n\n// The delimiter split is all this package used gray-matter for.\nfunction splitFrontmatter(raw: string): { fm: string | null; body: string } {\n const open = raw.match(/^---\\r?\\n/);\n if (!open) return { fm: null, body: raw };\n const rest = raw.slice(open[0].length);\n const close = rest.match(/^---\\r?(\\n|$)/m);\n if (!close || close.index === undefined) return { fm: null, body: raw };\n return { fm: rest.slice(0, close.index), body: rest.slice(close.index + close[0].length) };\n}\n\n// Lenient by design: parseDocument collects syntax errors as data and still yields values,\n// so Obsidian-style frontmatter (e.g. an alias starting with @) survives with a warning.\nfunction parseFrontmatter(relPath: string, fm: string, warnings: string[]): Record<string, unknown> {\n const doc = parseDocument(fm);\n if (doc.errors.length > 0) {\n warnings.push(`warning: ${relPath} frontmatter has ${doc.errors.length} syntax error(s) (${doc.errors[0].message.split('\\n')[0]}); parsed leniently`);\n }\n let data: unknown;\n try {\n data = doc.toJS();\n } catch (err) {\n warnings.push(`warning: ${relPath} has unparseable frontmatter (${(err as Error).message.split('\\n')[0]}); indexing without it`);\n return {};\n }\n if (data === null || data === undefined) return {};\n if (typeof data !== 'object' || Array.isArray(data)) {\n warnings.push(`warning: ${relPath} frontmatter is not a key-value mapping; ignoring it`);\n return {};\n }\n return data as Record<string, unknown>;\n}\n\nexport function parseFile(file: FileStat, extractors: Feature[] = []): { doc: ParsedDoc; warnings: string[] } {\n const raw = readFileSync(file.absPath, 'utf8');\n const warnings: string[] = [];\n\n const { fm, body: content } = splitFrontmatter(raw);\n const data = fm === null ? {} : parseFrontmatter(file.relPath, fm, warnings);\n const mapped: Record<string, string | number | bigint | null> = {};\n\n for (const key of Object.keys(data)) {\n if (RESERVED_COLUMNS.has(key)) {\n warnings.push(`warning: ${file.relPath} has a frontmatter key named \"${key}\", which is reserved; ignoring it`);\n continue;\n }\n mapped[key] = mapValue(data[key]);\n }\n\n // title/summary are plain YAML strings -- whitespace-collapse only;\n // the prose gets the full markdown strip.\n const search = { title: normalizeText(data.title), summary: normalizeText(data.summary), text: stripText(content) };\n\n return {\n doc: {\n relPath: file.relPath,\n mtimeMs: file.mtimeMs,\n size: file.size,\n presets: file.presets,\n data: mapped,\n search,\n extracted: Object.fromEntries(extractors.filter((f) => f.extract).map((f) => [f.name, f.extract?.(raw, content, search)])),\n },\n warnings,\n };\n}\n"],"names":["globSync","readFileSync","statSync","join","sep","removeMarkdown","parseDocument","embedEnabled","presetNames","presetSemanticEnabled","RESERVED_COLUMNS","Set","normalizeText","value","undefined","String","replace","trim","stripText","withoutWikilinks","withoutMarkdown","withoutTables","toPosixPath","relPath","separator","split","NO_THROW","throwIfNoEntry","listFiles","cfg","baseDir","coverage","Map","posixNeeded","name","preset","presets","matched","include","cwd","exclude","set","get","add","embedding","semanticPresets","filter","files","keys","sort","absPath","st","isFile","embed","some","has","push","mtimeMs","size","mapValue","BigInt","Number","isSafeInteger","JSON","stringify","splitFrontmatter","raw","open","match","fm","body","rest","slice","length","close","index","parseFrontmatter","warnings","doc","errors","message","data","toJS","err","Array","isArray","parseFile","file","extractors","content","mapped","key","Object","search","title","summary","text","extracted","fromEntries","f","extract","map"],"mappings":"AAAA,SAASA,QAAQ,EAAEC,YAAY,EAAEC,QAAQ,QAAQ,UAAU;AAC3D,SAASC,IAAI,EAAEC,GAAG,QAAQ,YAAY;AACtC,OAAOC,oBAAoB,kBAAkB;AAC7C,SAASC,aAAa,QAAQ,OAAO;AAErC,SAASC,YAAY,EAAEC,WAAW,EAAEC,qBAAqB,QAAQ,cAAc;AAG/E,6EAA6E;AAE7E,8FAA8F;AAC9F,oFAAoF;AACpF,OAAO,MAAMC,mBAAmB,IAAIC,IAAI;IAAC;IAAQ;IAAU;IAAS;IAAS;IAAW;IAAS;CAAW,EAAE;AAE9G,SAASC,cAAcC,KAAc;IACnC,IAAIA,UAAU,QAAQA,UAAUC,WAAW,OAAO;IAClD,OAAOC,OAAOF,OAAOG,OAAO,CAAC,QAAQ,KAAKC,IAAI;AAChD;AAEA,6FAA6F;AAC7F,sFAAsF;AACtF,SAASC,UAAUL,KAAa;IAC9B,MAAMM,mBAAmBN,MAAMG,OAAO,CAAC,gCAAgC,MAAMA,OAAO,CAAC,qBAAqB;IAC1G,MAAMI,kBAAkBf,eAAec;IACvC,MAAME,gBAAgBD,gBAAgBJ,OAAO,CAAC,2BAA2B,IAAIA,OAAO,CAAC,OAAO;IAC5F,OAAOJ,cAAcS;AACvB;AAWA,6FAA6F;AAC7F,8FAA8F;AAC9F,OAAO,SAASC,YAAYC,OAAe,EAAEC,YAAoBpB,GAAG;IAClE,OAAOoB,cAAc,OAAOD,QAAQE,KAAK,CAACD,WAAWrB,IAAI,CAAC,OAAOoB;AACnE;AAEA,2FAA2F;AAC3F,8FAA8F;AAC9F,0CAA0C;AAC1C,MAAMG,WAAW;IAAEC,gBAAgB;AAAM;AAEzC,OAAO,SAASC,UAAUC,GAAW,EAAEC,OAAe;IACpD,MAAMC,WAAW,IAAIC;IACrB,MAAMC,cAAc7B,QAAQ;IAC5B,KAAK,MAAM8B,QAAQ1B,YAAYqB,KAAM;QACnC,MAAMM,SAASN,IAAIO,OAAO,CAACF,KAAK;QAChC,KAAK,MAAMG,WAAWrC,SAASmC,OAAOG,OAAO,EAAE;YAAEC,KAAKT;YAASU,SAASL,OAAOK,OAAO;QAAC,GAAI;gBAE7ET;YADZ,MAAMR,UAAUU,cAAcX,YAAYe,WAAWA;YACrD,MAAMI,OAAMV,gBAAAA,SAASW,GAAG,CAACnB,sBAAbQ,2BAAAA,gBAAyB,IAAIpB;YACzC8B,IAAIE,GAAG,CAACT;YACRH,SAASU,GAAG,CAAClB,SAASkB;QACxB;IACF;IAEA,2EAA2E;IAC3E,MAAMG,YAAYrC,aAAasB;IAC/B,MAAMgB,kBAAkBD,YAAY,IAAIjC,IAAIH,YAAYqB,KAAKiB,MAAM,CAAC,CAACZ,OAASzB,sBAAsBoB,KAAKK,UAAU;IAEnH,MAAMa,QAAoB,EAAE;IAC5B,KAAK,MAAMxB,WAAW;WAAIQ,SAASiB,IAAI;KAAG,CAACC,IAAI,GAAI;QACjD,MAAMC,UAAU/C,KAAK2B,SAASP,UAAU,sDAAsD;QAC9F,yFAAyF;QACzF,uFAAuF;QACvF,MAAM4B,KAAKjD,SAASgD,SAASxB;QAC7B,IAAI,EAACyB,eAAAA,yBAAAA,GAAIC,MAAM,KAAI;QACnB,MAAMhB,UAAU;eAAKL,SAASW,GAAG,CAACnB;SAAyB,CAAC0B,IAAI;QAChE,MAAMI,QAAQR,oBAAoB,QAAQT,QAAQkB,IAAI,CAAC,CAACpB,OAASW,gBAAgBU,GAAG,CAACrB;QACrFa,MAAMS,IAAI,CAAC;YAAEjC;YAAS2B;YAASO,SAASN,GAAGM,OAAO;YAAEC,MAAMP,GAAGO,IAAI;YAAEtB;YAASiB;QAAM;IACpF;IACA,OAAON;AACT;AAcA,4FAA4F;AAC5F,4FAA4F;AAC5F,SAASY,SAAS9C,KAAc;IAC9B,IAAIA,UAAU,QAAQA,UAAUC,WAAW,OAAO;IAClD,IAAI,OAAOD,UAAU,WAAW,OAAO+C,OAAO/C,QAAQ,IAAI;IAC1D,IAAI,OAAOA,UAAU,UAAU,OAAOgD,OAAOC,aAAa,CAACjD,SAAS+C,OAAO/C,SAASA;IACpF,IAAI,OAAOA,UAAU,UAAU,OAAOA;IACtC,OAAOkD,KAAKC,SAAS,CAACnD;AACxB;AAEA,gEAAgE;AAChE,SAASoD,iBAAiBC,GAAW;IACnC,MAAMC,OAAOD,IAAIE,KAAK,CAAC;IACvB,IAAI,CAACD,MAAM,OAAO;QAAEE,IAAI;QAAMC,MAAMJ;IAAI;IACxC,MAAMK,OAAOL,IAAIM,KAAK,CAACL,IAAI,CAAC,EAAE,CAACM,MAAM;IACrC,MAAMC,QAAQH,KAAKH,KAAK,CAAC;IACzB,IAAI,CAACM,SAASA,MAAMC,KAAK,KAAK7D,WAAW,OAAO;QAAEuD,IAAI;QAAMC,MAAMJ;IAAI;IACtE,OAAO;QAAEG,IAAIE,KAAKC,KAAK,CAAC,GAAGE,MAAMC,KAAK;QAAGL,MAAMC,KAAKC,KAAK,CAACE,MAAMC,KAAK,GAAGD,KAAK,CAAC,EAAE,CAACD,MAAM;IAAE;AAC3F;AAEA,2FAA2F;AAC3F,yFAAyF;AACzF,SAASG,iBAAiBrD,OAAe,EAAE8C,EAAU,EAAEQ,QAAkB;IACvE,MAAMC,MAAMxE,cAAc+D;IAC1B,IAAIS,IAAIC,MAAM,CAACN,MAAM,GAAG,GAAG;QACzBI,SAASrB,IAAI,CAAC,CAAC,SAAS,EAAEjC,QAAQ,iBAAiB,EAAEuD,IAAIC,MAAM,CAACN,MAAM,CAAC,kBAAkB,EAAEK,IAAIC,MAAM,CAAC,EAAE,CAACC,OAAO,CAACvD,KAAK,CAAC,KAAK,CAAC,EAAE,CAAC,mBAAmB,CAAC;IACtJ;IACA,IAAIwD;IACJ,IAAI;QACFA,OAAOH,IAAII,IAAI;IACjB,EAAE,OAAOC,KAAK;QACZN,SAASrB,IAAI,CAAC,CAAC,SAAS,EAAEjC,QAAQ,8BAA8B,EAAE,AAAC4D,IAAcH,OAAO,CAACvD,KAAK,CAAC,KAAK,CAAC,EAAE,CAAC,sBAAsB,CAAC;QAC/H,OAAO,CAAC;IACV;IACA,IAAIwD,SAAS,QAAQA,SAASnE,WAAW,OAAO,CAAC;IACjD,IAAI,OAAOmE,SAAS,YAAYG,MAAMC,OAAO,CAACJ,OAAO;QACnDJ,SAASrB,IAAI,CAAC,CAAC,SAAS,EAAEjC,QAAQ,oDAAoD,CAAC;QACvF,OAAO,CAAC;IACV;IACA,OAAO0D;AACT;AAEA,OAAO,SAASK,UAAUC,IAAc,EAAEC,aAAwB,EAAE;IAClE,MAAMtB,MAAMjE,aAAasF,KAAKrC,OAAO,EAAE;IACvC,MAAM2B,WAAqB,EAAE;IAE7B,MAAM,EAAER,EAAE,EAAEC,MAAMmB,OAAO,EAAE,GAAGxB,iBAAiBC;IAC/C,MAAMe,OAAOZ,OAAO,OAAO,CAAC,IAAIO,iBAAiBW,KAAKhE,OAAO,EAAE8C,IAAIQ;IACnE,MAAMa,SAA0D,CAAC;IAEjE,KAAK,MAAMC,OAAOC,OAAO5C,IAAI,CAACiC,MAAO;QACnC,IAAIvE,iBAAiB6C,GAAG,CAACoC,MAAM;YAC7Bd,SAASrB,IAAI,CAAC,CAAC,SAAS,EAAE+B,KAAKhE,OAAO,CAAC,8BAA8B,EAAEoE,IAAI,iCAAiC,CAAC;YAC7G;QACF;QACAD,MAAM,CAACC,IAAI,GAAGhC,SAASsB,IAAI,CAACU,IAAI;IAClC;IAEA,oEAAoE;IACpE,0CAA0C;IAC1C,MAAME,SAAS;QAAEC,OAAOlF,cAAcqE,KAAKa,KAAK;QAAGC,SAASnF,cAAcqE,KAAKc,OAAO;QAAGC,MAAM9E,UAAUuE;IAAS;IAElH,OAAO;QACLX,KAAK;YACHvD,SAASgE,KAAKhE,OAAO;YACrBkC,SAAS8B,KAAK9B,OAAO;YACrBC,MAAM6B,KAAK7B,IAAI;YACftB,SAASmD,KAAKnD,OAAO;YACrB6C,MAAMS;YACNG;YACAI,WAAWL,OAAOM,WAAW,CAACV,WAAW1C,MAAM,CAAC,CAACqD,IAAMA,EAAEC,OAAO,EAAEC,GAAG,CAAC,CAACF;oBAAeA;uBAAT;oBAACA,EAAEjE,IAAI;qBAAEiE,aAAAA,EAAEC,OAAO,cAATD,iCAAAA,gBAAAA,GAAYjC,KAAKuB,SAASI;iBAAQ;;QAC1H;QACAhB;IACF;AACF"}
|
|
1
|
+
{"version":3,"sources":["/Users/kevin/Dev/OpenSource/ai/sensemaking/src/scan.ts"],"sourcesContent":["import { globSync, readFileSync, statSync } from 'node:fs';\nimport { join, sep } from 'node:path';\nimport removeMarkdown from 'remove-markdown';\nimport { isCollection, parseDocument, visit } from 'yaml';\nimport type { Config } from './config.ts';\nimport { embedEnabled, presetNames, presetSemanticEnabled } from './config.ts';\nimport type { Feature } from './features/types.ts';\n\n// Filesystem -> rows. Pure data in, data + warnings out; db.ts does the SQL.\n\n// Frontmatter keys that would collide with table columns. Exported so db.ts's upsert can tell\n// a feature-owned column (`_rank`) from a parsed one and leave it alone on reparse.\nexport const RESERVED_COLUMNS = new Set(['path', '_mtime', '_size', '_rank', '_parse_error', 'content', 'links', 'sections']);\n\n// YAML error codes whose recovery is unambiguous, so the parse is accepted rather than\n// quarantined. Only one qualifies: YAML 1.2 reserves `@` and `` ` `` at the start of a plain\n// scalar for future use, so they can never be valid and the text can only be what was typed\n// (`aliases: [@handle]` -> [\"@handle\"]). Every other code has a second reading -- an unquoted\n// `:` swallows the keys after it, an unquoted `[..](..)` drops the URL, a duplicate key picks\n// one value in silence -- so it writes values nobody wrote. See plans/frontmatter-parse-policy.md.\nconst ACCEPTED_YAML_CODES = new Set(['BAD_SCALAR_START']);\n\nfunction normalizeText(value: unknown): string {\n if (value === null || value === undefined) return '';\n return String(value).replace(/\\s+/g, ' ').trim();\n}\n\n// Keeps URL query strings, asset filenames, and HTML attributes out of the index: rare terms\n// carry high IDF, so they outrank prose. remove-markdown misses wikilinks and tables.\nfunction stripText(value: string): string {\n const withoutWikilinks = value.replace(/\\[\\[([^\\]|]+)\\|([^\\]]+)\\]\\]/g, '$2').replace(/\\[\\[([^\\]]+)\\]\\]/g, '$1');\n const withoutMarkdown = removeMarkdown(withoutWikilinks);\n const withoutTables = withoutMarkdown.replace(/^\\s*\\|?[-\\s|:]+\\|\\s*$/gm, '').replace(/\\|/g, ' ');\n return normalizeText(withoutTables);\n}\n\nexport interface FileStat {\n relPath: string;\n absPath: string;\n mtimeMs: number;\n size: number;\n presets: string[]; // every declared preset covering this file (>= 1; union, overlap allowed)\n embed: boolean; // true iff a model is named and some covering preset has semantic on\n}\n\n// Presets are views, not partitions: they overlap freely, and a file's covering set (not one\n// owner) drives indexing. Globs resolve relative to baseDir; unmatched files are not indexed.\nexport function toPosixPath(relPath: string, separator: string = sep): string {\n return separator === '\\\\' ? relPath.split(separator).join('/') : relPath;\n}\n\n// Every command pays listFiles before it answers (the freshness check stats each file), so\n// per-file work here is the hottest path in the package. Everything derivable from the config\n// alone is computed once, above the loop.\nconst NO_THROW = { throwIfNoEntry: false } as const;\n\nexport function listFiles(cfg: Config, baseDir: string): FileStat[] {\n const coverage = new Map<string, Set<string>>();\n const posixNeeded = sep === '\\\\';\n for (const name of presetNames(cfg)) {\n const preset = cfg.presets[name];\n for (const matched of globSync(preset.include, { cwd: baseDir, exclude: preset.exclude })) {\n const relPath = posixNeeded ? toPosixPath(matched) : matched;\n const set = coverage.get(relPath) ?? new Set<string>();\n set.add(name);\n coverage.set(relPath, set);\n }\n }\n\n // Which presets want vectors is a property of the config, not of any file.\n const embedding = embedEnabled(cfg);\n const semanticPresets = embedding ? new Set(presetNames(cfg).filter((name) => presetSemanticEnabled(cfg, name))) : null;\n\n const files: FileStat[] = [];\n for (const relPath of [...coverage.keys()].sort()) {\n const absPath = join(baseDir, relPath); // join re-applies the platform separator for fs calls\n // node:fs glob matches directories and dangling symlinks; fast-glob returned neither, so\n // one stat filters both back out (throwIfNoEntry keeps a dangling link from throwing).\n const st = statSync(absPath, NO_THROW);\n if (!st?.isFile()) continue;\n const presets = [...(coverage.get(relPath) as Set<string>)].sort();\n const embed = semanticPresets !== null && presets.some((name) => semanticPresets.has(name));\n files.push({ relPath, absPath, mtimeMs: st.mtimeMs, size: st.size, presets, embed });\n }\n return files;\n}\n\nexport interface ParsedDoc {\n relPath: string;\n mtimeMs: number;\n size: number;\n presets: string[];\n data: Record<string, string | number | bigint | null>;\n // NULL when the frontmatter parsed, the first YAML message otherwise. In the row rather than\n // a side table so `SELECT *` and any `IS NULL` investigation trip over it without being asked.\n parseError: string | null;\n // title/summary are duplicated from frontmatter so bm25() can weight them above the body text.\n search: { title: string; summary: string; text: string };\n // Per-feature extraction results, keyed by feature name; features store them at reconcile.\n extracted: Record<string, unknown>;\n}\n\n// Storage class follows the YAML scalar. Booleans store as 1/0, so `WHERE flag = 1` matches\n// and `WHERE flag = 'true'` cannot; `map` prints observed types so the mismatch is visible.\nfunction mapValue(value: unknown): string | number | bigint | null {\n if (value === null || value === undefined) return null;\n if (typeof value === 'boolean') return BigInt(value ? 1 : 0);\n if (typeof value === 'number') return Number.isSafeInteger(value) ? BigInt(value) : value;\n if (typeof value === 'string') return value;\n return JSON.stringify(value);\n}\n\n// The delimiter split is all this package used gray-matter for.\nfunction splitFrontmatter(raw: string): { fm: string | null; body: string } {\n const open = raw.match(/^---\\r?\\n/);\n if (!open) return { fm: null, body: raw };\n const rest = raw.slice(open[0].length);\n const close = rest.match(/^---\\r?(\\n|$)/m);\n if (!close || close.index === undefined) return { fm: null, body: raw };\n return { fm: rest.slice(0, close.index), body: rest.slice(close.index + close[0].length) };\n}\n\n// A well-formed document can still hold a value nobody meant: `created: {{date}}` is valid\n// YAML for a flow map used as a mapping key, so it raises no error and stores\n// {\"{ date }\": null}. No error code can catch that, but yaml notices the stringified key, so\n// this reports it with the path instead (yaml's own warning has none, fires once per document,\n// and is what trains readers to discard stderr).\nfunction warnStringifiedKeys(relPath: string, doc: ReturnType<typeof parseDocument>, warnings: string[]): void {\n let found = false;\n // Nested, not top level: `created: {{date}}` puts the collection key one level down, inside\n // the flow map that `{{...}}` parses as.\n visit(doc, {\n Pair(_key, pair) {\n if (!isCollection(pair.key)) return undefined;\n found = true;\n return visit.BREAK;\n },\n });\n // One per file: a template repeats the same mistake on every field it stamps.\n if (found) warnings.push(`warning: ${relPath} frontmatter has a key that is itself a list or mapping, stored as text; this is usually an unrendered template placeholder like {{date}}`);\n}\n\n// Accept a clean parse, and one whose every error is unambiguous (ACCEPTED_YAML_CODES).\n// Anything else is quarantined: no frontmatter columns at all, and `_parse_error` carries the\n// reason. Recovering it would write values nobody wrote, which is worse than absence because\n// no query can see it. The file is still indexed -- content, links and sections never touch\n// frontmatter -- so a broken note stays searchable while it is being hunted for.\n// yaml's message continues onto a source excerpt, so the first line is the sentence -- minus\n// the colon that introduced the part being dropped.\nfunction firstLine(message: string): string {\n return message.split('\\n')[0].replace(/:\\s*$/, '');\n}\n\nfunction parseFrontmatter(relPath: string, fm: string, warnings: string[]): { data: Record<string, unknown>; parseError: string | null } {\n // logLevel silences yaml's own pathless warnings; warnStringifiedKeys re-reports the one\n // that carries information, with the file it came from.\n const doc = parseDocument(fm, { logLevel: 'silent' });\n const refused = doc.errors.filter((err) => !ACCEPTED_YAML_CODES.has(err.code));\n if (refused.length > 0) {\n const detail = refused.length > 1 ? ` (and ${refused.length - 1} more)` : '';\n const parseError = `${firstLine(refused[0].message)}${detail}`;\n warnings.push(`warning: ${relPath} frontmatter did not parse, so none of it is indexed: ${parseError}`);\n return { data: {}, parseError };\n }\n\n let data: unknown;\n try {\n data = doc.toJS();\n } catch (err) {\n // Reaches here with doc.errors empty: `title: **Bold**` parses, then opens an alias on\n // materialisation. An empty error list is not a successful parse.\n const parseError = firstLine((err as Error).message);\n warnings.push(`warning: ${relPath} frontmatter did not parse, so none of it is indexed: ${parseError}`);\n return { data: {}, parseError };\n }\n\n if (data === null || data === undefined) return { data: {}, parseError: null };\n if (typeof data !== 'object' || Array.isArray(data)) {\n const parseError = 'frontmatter is not a key-value mapping';\n warnings.push(`warning: ${relPath} ${parseError}; none of it is indexed`);\n return { data: {}, parseError };\n }\n warnStringifiedKeys(relPath, doc, warnings);\n return { data: data as Record<string, unknown>, parseError: null };\n}\n\nexport function parseFile(file: FileStat, extractors: Feature[] = []): { doc: ParsedDoc; warnings: string[] } {\n const raw = readFileSync(file.absPath, 'utf8');\n const warnings: string[] = [];\n\n const { fm, body: content } = splitFrontmatter(raw);\n const { data, parseError } = fm === null ? { data: {} as Record<string, unknown>, parseError: null } : parseFrontmatter(file.relPath, fm, warnings);\n const mapped: Record<string, string | number | bigint | null> = {};\n\n for (const key of Object.keys(data)) {\n if (RESERVED_COLUMNS.has(key)) {\n warnings.push(`warning: ${file.relPath} has a frontmatter key named \"${key}\", which is reserved; ignoring it`);\n continue;\n }\n mapped[key] = mapValue(data[key]);\n }\n\n // title/summary are plain YAML strings -- whitespace-collapse only;\n // the prose gets the full markdown strip.\n const search = { title: normalizeText(data.title), summary: normalizeText(data.summary), text: stripText(content) };\n\n return {\n doc: {\n relPath: file.relPath,\n mtimeMs: file.mtimeMs,\n size: file.size,\n presets: file.presets,\n data: mapped,\n parseError,\n search,\n extracted: Object.fromEntries(extractors.filter((f) => f.extract).map((f) => [f.name, f.extract?.(raw, content, search)])),\n },\n warnings,\n };\n}\n"],"names":["globSync","readFileSync","statSync","join","sep","removeMarkdown","isCollection","parseDocument","visit","embedEnabled","presetNames","presetSemanticEnabled","RESERVED_COLUMNS","Set","ACCEPTED_YAML_CODES","normalizeText","value","undefined","String","replace","trim","stripText","withoutWikilinks","withoutMarkdown","withoutTables","toPosixPath","relPath","separator","split","NO_THROW","throwIfNoEntry","listFiles","cfg","baseDir","coverage","Map","posixNeeded","name","preset","presets","matched","include","cwd","exclude","set","get","add","embedding","semanticPresets","filter","files","keys","sort","absPath","st","isFile","embed","some","has","push","mtimeMs","size","mapValue","BigInt","Number","isSafeInteger","JSON","stringify","splitFrontmatter","raw","open","match","fm","body","rest","slice","length","close","index","warnStringifiedKeys","doc","warnings","found","Pair","_key","pair","key","BREAK","firstLine","message","parseFrontmatter","logLevel","refused","errors","err","code","detail","parseError","data","toJS","Array","isArray","parseFile","file","extractors","content","mapped","Object","search","title","summary","text","extracted","fromEntries","f","extract","map"],"mappings":"AAAA,SAASA,QAAQ,EAAEC,YAAY,EAAEC,QAAQ,QAAQ,UAAU;AAC3D,SAASC,IAAI,EAAEC,GAAG,QAAQ,YAAY;AACtC,OAAOC,oBAAoB,kBAAkB;AAC7C,SAASC,YAAY,EAAEC,aAAa,EAAEC,KAAK,QAAQ,OAAO;AAE1D,SAASC,YAAY,EAAEC,WAAW,EAAEC,qBAAqB,QAAQ,cAAc;AAG/E,6EAA6E;AAE7E,8FAA8F;AAC9F,oFAAoF;AACpF,OAAO,MAAMC,mBAAmB,IAAIC,IAAI;IAAC;IAAQ;IAAU;IAAS;IAAS;IAAgB;IAAW;IAAS;CAAW,EAAE;AAE9H,uFAAuF;AACvF,6FAA6F;AAC7F,4FAA4F;AAC5F,8FAA8F;AAC9F,8FAA8F;AAC9F,mGAAmG;AACnG,MAAMC,sBAAsB,IAAID,IAAI;IAAC;CAAmB;AAExD,SAASE,cAAcC,KAAc;IACnC,IAAIA,UAAU,QAAQA,UAAUC,WAAW,OAAO;IAClD,OAAOC,OAAOF,OAAOG,OAAO,CAAC,QAAQ,KAAKC,IAAI;AAChD;AAEA,6FAA6F;AAC7F,sFAAsF;AACtF,SAASC,UAAUL,KAAa;IAC9B,MAAMM,mBAAmBN,MAAMG,OAAO,CAAC,gCAAgC,MAAMA,OAAO,CAAC,qBAAqB;IAC1G,MAAMI,kBAAkBlB,eAAeiB;IACvC,MAAME,gBAAgBD,gBAAgBJ,OAAO,CAAC,2BAA2B,IAAIA,OAAO,CAAC,OAAO;IAC5F,OAAOJ,cAAcS;AACvB;AAWA,6FAA6F;AAC7F,8FAA8F;AAC9F,OAAO,SAASC,YAAYC,OAAe,EAAEC,YAAoBvB,GAAG;IAClE,OAAOuB,cAAc,OAAOD,QAAQE,KAAK,CAACD,WAAWxB,IAAI,CAAC,OAAOuB;AACnE;AAEA,2FAA2F;AAC3F,8FAA8F;AAC9F,0CAA0C;AAC1C,MAAMG,WAAW;IAAEC,gBAAgB;AAAM;AAEzC,OAAO,SAASC,UAAUC,GAAW,EAAEC,OAAe;IACpD,MAAMC,WAAW,IAAIC;IACrB,MAAMC,cAAchC,QAAQ;IAC5B,KAAK,MAAMiC,QAAQ3B,YAAYsB,KAAM;QACnC,MAAMM,SAASN,IAAIO,OAAO,CAACF,KAAK;QAChC,KAAK,MAAMG,WAAWxC,SAASsC,OAAOG,OAAO,EAAE;YAAEC,KAAKT;YAASU,SAASL,OAAOK,OAAO;QAAC,GAAI;gBAE7ET;YADZ,MAAMR,UAAUU,cAAcX,YAAYe,WAAWA;YACrD,MAAMI,OAAMV,gBAAAA,SAASW,GAAG,CAACnB,sBAAbQ,2BAAAA,gBAAyB,IAAIrB;YACzC+B,IAAIE,GAAG,CAACT;YACRH,SAASU,GAAG,CAAClB,SAASkB;QACxB;IACF;IAEA,2EAA2E;IAC3E,MAAMG,YAAYtC,aAAauB;IAC/B,MAAMgB,kBAAkBD,YAAY,IAAIlC,IAAIH,YAAYsB,KAAKiB,MAAM,CAAC,CAACZ,OAAS1B,sBAAsBqB,KAAKK,UAAU;IAEnH,MAAMa,QAAoB,EAAE;IAC5B,KAAK,MAAMxB,WAAW;WAAIQ,SAASiB,IAAI;KAAG,CAACC,IAAI,GAAI;QACjD,MAAMC,UAAUlD,KAAK8B,SAASP,UAAU,sDAAsD;QAC9F,yFAAyF;QACzF,uFAAuF;QACvF,MAAM4B,KAAKpD,SAASmD,SAASxB;QAC7B,IAAI,EAACyB,eAAAA,yBAAAA,GAAIC,MAAM,KAAI;QACnB,MAAMhB,UAAU;eAAKL,SAASW,GAAG,CAACnB;SAAyB,CAAC0B,IAAI;QAChE,MAAMI,QAAQR,oBAAoB,QAAQT,QAAQkB,IAAI,CAAC,CAACpB,OAASW,gBAAgBU,GAAG,CAACrB;QACrFa,MAAMS,IAAI,CAAC;YAAEjC;YAAS2B;YAASO,SAASN,GAAGM,OAAO;YAAEC,MAAMP,GAAGO,IAAI;YAAEtB;YAASiB;QAAM;IACpF;IACA,OAAON;AACT;AAiBA,4FAA4F;AAC5F,4FAA4F;AAC5F,SAASY,SAAS9C,KAAc;IAC9B,IAAIA,UAAU,QAAQA,UAAUC,WAAW,OAAO;IAClD,IAAI,OAAOD,UAAU,WAAW,OAAO+C,OAAO/C,QAAQ,IAAI;IAC1D,IAAI,OAAOA,UAAU,UAAU,OAAOgD,OAAOC,aAAa,CAACjD,SAAS+C,OAAO/C,SAASA;IACpF,IAAI,OAAOA,UAAU,UAAU,OAAOA;IACtC,OAAOkD,KAAKC,SAAS,CAACnD;AACxB;AAEA,gEAAgE;AAChE,SAASoD,iBAAiBC,GAAW;IACnC,MAAMC,OAAOD,IAAIE,KAAK,CAAC;IACvB,IAAI,CAACD,MAAM,OAAO;QAAEE,IAAI;QAAMC,MAAMJ;IAAI;IACxC,MAAMK,OAAOL,IAAIM,KAAK,CAACL,IAAI,CAAC,EAAE,CAACM,MAAM;IACrC,MAAMC,QAAQH,KAAKH,KAAK,CAAC;IACzB,IAAI,CAACM,SAASA,MAAMC,KAAK,KAAK7D,WAAW,OAAO;QAAEuD,IAAI;QAAMC,MAAMJ;IAAI;IACtE,OAAO;QAAEG,IAAIE,KAAKC,KAAK,CAAC,GAAGE,MAAMC,KAAK;QAAGL,MAAMC,KAAKC,KAAK,CAACE,MAAMC,KAAK,GAAGD,KAAK,CAAC,EAAE,CAACD,MAAM;IAAE;AAC3F;AAEA,2FAA2F;AAC3F,8EAA8E;AAC9E,6FAA6F;AAC7F,+FAA+F;AAC/F,iDAAiD;AACjD,SAASG,oBAAoBrD,OAAe,EAAEsD,GAAqC,EAAEC,QAAkB;IACrG,IAAIC,QAAQ;IACZ,4FAA4F;IAC5F,yCAAyC;IACzC1E,MAAMwE,KAAK;QACTG,MAAKC,IAAI,EAAEC,IAAI;YACb,IAAI,CAAC/E,aAAa+E,KAAKC,GAAG,GAAG,OAAOrE;YACpCiE,QAAQ;YACR,OAAO1E,MAAM+E,KAAK;QACpB;IACF;IACA,8EAA8E;IAC9E,IAAIL,OAAOD,SAAStB,IAAI,CAAC,CAAC,SAAS,EAAEjC,QAAQ,yIAAyI,CAAC;AACzL;AAEA,wFAAwF;AACxF,8FAA8F;AAC9F,6FAA6F;AAC7F,4FAA4F;AAC5F,iFAAiF;AACjF,6FAA6F;AAC7F,oDAAoD;AACpD,SAAS8D,UAAUC,OAAe;IAChC,OAAOA,QAAQ7D,KAAK,CAAC,KAAK,CAAC,EAAE,CAACT,OAAO,CAAC,SAAS;AACjD;AAEA,SAASuE,iBAAiBhE,OAAe,EAAE8C,EAAU,EAAES,QAAkB;IACvE,yFAAyF;IACzF,wDAAwD;IACxD,MAAMD,MAAMzE,cAAciE,IAAI;QAAEmB,UAAU;IAAS;IACnD,MAAMC,UAAUZ,IAAIa,MAAM,CAAC5C,MAAM,CAAC,CAAC6C,MAAQ,CAAChF,oBAAoB4C,GAAG,CAACoC,IAAIC,IAAI;IAC5E,IAAIH,QAAQhB,MAAM,GAAG,GAAG;QACtB,MAAMoB,SAASJ,QAAQhB,MAAM,GAAG,IAAI,CAAC,MAAM,EAAEgB,QAAQhB,MAAM,GAAG,EAAE,MAAM,CAAC,GAAG;QAC1E,MAAMqB,aAAa,GAAGT,UAAUI,OAAO,CAAC,EAAE,CAACH,OAAO,IAAIO,QAAQ;QAC9Df,SAAStB,IAAI,CAAC,CAAC,SAAS,EAAEjC,QAAQ,sDAAsD,EAAEuE,YAAY;QACtG,OAAO;YAAEC,MAAM,CAAC;YAAGD;QAAW;IAChC;IAEA,IAAIC;IACJ,IAAI;QACFA,OAAOlB,IAAImB,IAAI;IACjB,EAAE,OAAOL,KAAK;QACZ,uFAAuF;QACvF,kEAAkE;QAClE,MAAMG,aAAaT,UAAU,AAACM,IAAcL,OAAO;QACnDR,SAAStB,IAAI,CAAC,CAAC,SAAS,EAAEjC,QAAQ,sDAAsD,EAAEuE,YAAY;QACtG,OAAO;YAAEC,MAAM,CAAC;YAAGD;QAAW;IAChC;IAEA,IAAIC,SAAS,QAAQA,SAASjF,WAAW,OAAO;QAAEiF,MAAM,CAAC;QAAGD,YAAY;IAAK;IAC7E,IAAI,OAAOC,SAAS,YAAYE,MAAMC,OAAO,CAACH,OAAO;QACnD,MAAMD,aAAa;QACnBhB,SAAStB,IAAI,CAAC,CAAC,SAAS,EAAEjC,QAAQ,CAAC,EAAEuE,WAAW,uBAAuB,CAAC;QACxE,OAAO;YAAEC,MAAM,CAAC;YAAGD;QAAW;IAChC;IACAlB,oBAAoBrD,SAASsD,KAAKC;IAClC,OAAO;QAAEiB,MAAMA;QAAiCD,YAAY;IAAK;AACnE;AAEA,OAAO,SAASK,UAAUC,IAAc,EAAEC,aAAwB,EAAE;IAClE,MAAMnC,MAAMpE,aAAasG,KAAKlD,OAAO,EAAE;IACvC,MAAM4B,WAAqB,EAAE;IAE7B,MAAM,EAAET,EAAE,EAAEC,MAAMgC,OAAO,EAAE,GAAGrC,iBAAiBC;IAC/C,MAAM,EAAE6B,IAAI,EAAED,UAAU,EAAE,GAAGzB,OAAO,OAAO;QAAE0B,MAAM,CAAC;QAA8BD,YAAY;IAAK,IAAIP,iBAAiBa,KAAK7E,OAAO,EAAE8C,IAAIS;IAC1I,MAAMyB,SAA0D,CAAC;IAEjE,KAAK,MAAMpB,OAAOqB,OAAOxD,IAAI,CAAC+C,MAAO;QACnC,IAAItF,iBAAiB8C,GAAG,CAAC4B,MAAM;YAC7BL,SAAStB,IAAI,CAAC,CAAC,SAAS,EAAE4C,KAAK7E,OAAO,CAAC,8BAA8B,EAAE4D,IAAI,iCAAiC,CAAC;YAC7G;QACF;QACAoB,MAAM,CAACpB,IAAI,GAAGxB,SAASoC,IAAI,CAACZ,IAAI;IAClC;IAEA,oEAAoE;IACpE,0CAA0C;IAC1C,MAAMsB,SAAS;QAAEC,OAAO9F,cAAcmF,KAAKW,KAAK;QAAGC,SAAS/F,cAAcmF,KAAKY,OAAO;QAAGC,MAAM1F,UAAUoF;IAAS;IAElH,OAAO;QACLzB,KAAK;YACHtD,SAAS6E,KAAK7E,OAAO;YACrBkC,SAAS2C,KAAK3C,OAAO;YACrBC,MAAM0C,KAAK1C,IAAI;YACftB,SAASgE,KAAKhE,OAAO;YACrB2D,MAAMQ;YACNT;YACAW;YACAI,WAAWL,OAAOM,WAAW,CAACT,WAAWvD,MAAM,CAAC,CAACiE,IAAMA,EAAEC,OAAO,EAAEC,GAAG,CAAC,CAACF;oBAAeA;uBAAT;oBAACA,EAAE7E,IAAI;qBAAE6E,aAAAA,EAAEC,OAAO,cAATD,iCAAAA,gBAAAA,GAAY7C,KAAKoC,SAASG;iBAAQ;;QAC1H;QACA3B;IACF;AACF"}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "sensemaking",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.11.0",
|
|
4
4
|
"description": "Query and search your markdown notes with context-aware progressive disclosure: SQL over frontmatter, links, and text, plus semantic search and link-graph ranking. No server, no build step",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"markdown",
|
package/schema.json
CHANGED
|
@@ -85,7 +85,7 @@
|
|
|
85
85
|
},
|
|
86
86
|
"queries": {
|
|
87
87
|
"type": "object",
|
|
88
|
-
"description": "Saved queries runnable as `sense <name> [params...]`, each naming the verb it runs, one to one with the two commands: `{ sql }` runs like `sense sql`, `{ search }` like `sense search`. A bare string is rejected -- it silently meant SQL. `{ sql }`: `?` placeholders bind to CLI positional args in order; deterministic and enumerating, including raw FTS5 `MATCH` for word search under your own SQL -- \"0 rows = not in the tree\" lives here. Tables: `frontmatter` (one row per file, one column per discovered frontmatter key, plus `path`/`_mtime`/`_size`/`_rank
|
|
88
|
+
"description": "Saved queries runnable as `sense <name> [params...]`, each naming the verb it runs, one to one with the two commands: `{ sql }` runs like `sense sql`, `{ search }` like `sense search`. A bare string is rejected -- it silently meant SQL. `{ sql }`: `?` placeholders bind to CLI positional args in order; deterministic and enumerating, including raw FTS5 `MATCH` for word search under your own SQL -- \"0 rows = not in the tree\" lives here. Tables: `frontmatter` (one row per file, one column per discovered frontmatter key, plus `path`/`_mtime`/`_size`/`_rank`/`_parse_error`, the last being NULL when the frontmatter parsed and the YAML message when it did not, in which case no other column is populated), `content` (FTS5: `title`, `summary`, `text`, `path`), `links` (`src`, `target`, `dst`), `sections` (`path`, `idx`, `level`, `heading`, `start_line`, `end_line`, `tokens`), and `preset_files` (`preset`, `path`) -- which presets cover which files. `has(field, value)`: array membership on a JSON-array field, substring match on a string (so has(f.status, 'active') also matches 'inactive'), false on NULL. Exact matches: `=` for scalars, `EXISTS (SELECT 1 FROM json_each(f.tags) WHERE value = ?)` for array members. Canonical query: `SELECT f.path, content.title, content.summary, CASE WHEN length(content.text) <= 16384 THEN snippet(content, -1, '«', '»', '…', 10) END AS hit FROM frontmatter f JOIN content ON content.path = f.path WHERE content MATCH ? ORDER BY bm25(content, 10.0, 5.0, 1.0) LIMIT 10`. The CASE bounds snippet(), which re-tokenizes each matched document and costs seconds per query once a tree holds a megabyte-scale note; `search` applies the same bound internally. Saved search object: one text driving every engine the scoped preset has -- word match, links, vectors -- fused into one ranked list, `via` labeling which engine produced each row; `search` text must be non-empty (a saved query saves a question -- a scope without one is just flags). `preset` names one declared preset (defaults to `default`); `include` is an ad hoc glob scope that replaces the preset's include/exclude entirely, same as `search --include`; `where` and `k` behave like the `search` command's flags. `sense <name>` behaves like `sense search <search> [--preset] [--include] [--where] [--k]` with zero flags; it takes no positional parameters. Running an entry validates it: a typo'd column or unknown preset errors and exits nonzero, and a parameterised entry validates with any argument, since SQL is prepared before parameters bind. Nothing asserts on the result itself; a returned row set is the reader's judgment. Reserved frontmatter keys: `path`, `_mtime`, `_size`, `_rank`, `_parse_error`, `content`, `links`, `sections`. Reserved names (unreachable as saved queries, any shape): `init`, `sql`, `search`, `map`, `peek`, `path`, `related`, `download`, `watch`, `status`.",
|
|
89
89
|
"additionalProperties": {
|
|
90
90
|
"oneOf": [
|
|
91
91
|
{
|
package/skills/sense/SKILL.md
CHANGED
|
@@ -1,11 +1,11 @@
|
|
|
1
1
|
---
|
|
2
2
|
name: sense
|
|
3
|
-
description: Query a markdown tree with the sense CLI: filter notes by frontmatter, full-text search the prose, follow wikilinks/backlinks, trace how notes connect (link path,
|
|
3
|
+
description: "Query a markdown tree with the sense CLI: filter notes by frontmatter, full-text search the prose, follow wikilinks/backlinks, trace how notes connect (link path, similar-but-unlinked), and read note outlines. Use when the user wants to query, filter, count, search, or report on a folder of markdown notes, when you need to find which notes discuss a topic before reading them, when you want a note's backlinks or structure, when you want to know how two notes connect, when a directory has a sense.config.json, or when asked to add a saved entry to one."
|
|
4
4
|
---
|
|
5
5
|
|
|
6
6
|
# sense
|
|
7
7
|
|
|
8
|
-
SQL over a markdown tree, kept fresh by a filesystem check on every query. Every file becomes rows in `frontmatter` (one column per key, plus `path`/`_mtime`/`_size`/`_rank`), `content` (FTS5: `title`, `summary`, `text`), `links` (`src`, `target`, `dst`; `NULL` dst = dead link),
|
|
8
|
+
SQL over a markdown tree, kept fresh by a filesystem check on every query. Every file becomes rows in `frontmatter` (one column per key, plus `path`/`_mtime`/`_size`/`_rank`/`_parse_error`), `content` (FTS5: `title`, `summary`, `text`), `links` (`src`, `target`, `dst`; `NULL` dst = dead link), `sections` (heading outline with line ranges and token estimates), and `preset_files` (`path`, `preset`: which presets cover which files). Features add their own storage; `map` and `status` report which are on.
|
|
9
9
|
|
|
10
10
|
## What each tool is for
|
|
11
11
|
|
|
@@ -32,18 +32,18 @@ sense related notes/pricing-model.md # notes similar by meaning it do
|
|
|
32
32
|
sense map
|
|
33
33
|
sense sql "<statement>" [params...] # ad-hoc SQL; ? binds positional args, count-checked
|
|
34
34
|
sense <name> [params...] # a saved query from sense.config.json
|
|
35
|
-
sense --list | status |
|
|
35
|
+
sense --list | status | download
|
|
36
36
|
```
|
|
37
37
|
|
|
38
38
|
- Terms pass verbatim to FTS5 MATCH. Bare words AND-join (one absent word means zero rows), so write `OR` yourself when you want any-word matching; double-quote punctuated terms (`"customer-facing"`, `"founder's"`); invalid syntax is an error, not a rewrite. The same rules apply to search commands you write into subagent briefs.
|
|
39
39
|
- When a search misses, the recall levers are: OR-in synonyms and concrete instances (the index only knows the words in the files; a note about a specific tool rarely names its category), raise `--k` (a row costs tens of tokens), and widen the scope (`--preset`, or `--include` for an ad-hoc glob). Vector rows already cover the meaning-over-words gap by default. Each widening adds candidates and dilutes ranking, so the noise trade-off runs both ways.
|
|
40
40
|
- A frontmatter query enumerates its matches deterministically; search ranks by term overlap, so results shift as phrasing shifts. Trade-off: a query needs a known field, search doesn't.
|
|
41
41
|
- The `via` column says what produced each row: `match` (words hit), `link` (connected to notes that hit), `vector` (near in meaning), and combinations. The `lines` column, when set, points at the section that earned the row (the best-matching chunk on vector rows, the term cluster's section on large lexical notes) and is a direct `Read` range; null means the whole note is the reference.
|
|
42
|
-
- Scope is one vocabulary shared by `search`, `map`, `peek`, `path`, and `related`: bare command uses the config's `default` preset; `--preset <name>` picks another (unknown names error, listing what's declared); `--include <glob>` and `--exclude <glob>` are ad-hoc globs for one command, each overriding its own side of the preset, so one does not clear the other; `--no-exclude` drops the preset's `exclude` for one command, the only way to widen past it without editing config (it widens the query scope, not the index: a file no preset covers is never indexed). `--where` takes any SQL condition against frontmatter alias `f`, not only field equality: `"f.status = 'active' AND has(f.tags, 'x')"`, `"datetime(f.created) >= datetime(?)"`. There is no whole-index flag: a broad `default` preset, or a declared `all` preset (`include ["**/*"]`), is the whole tree. `sense status` shows every preset with its coverage.
|
|
42
|
+
- Scope is one vocabulary shared by `search`, `map`, `peek`, `path`, and `related`: bare command uses the config's `default` preset; `--preset <name>` picks another (unknown names error, listing what's declared); `--include <glob>` and `--exclude <glob>` are ad-hoc globs for one command, each overriding its own side of the preset, so one does not clear the other; `--no-exclude` drops the preset's `exclude` for one command, the only way to widen past it without editing config (it widens the query scope, not the index: a file no preset covers is never indexed). `--where` takes any SQL condition against frontmatter alias `f`, not only field equality: `"f.status = 'active' AND has(f.tags, 'x')"`, `"datetime(f.created) >= datetime(?)"`. There is no whole-index flag: a broad `default` preset, or a declared `all` preset (`include ["**/*"]`), is the whole tree. `sense status` shows every preset with its coverage. `sql` scopes differently: it runs over the whole index by default, and `--preset <name>` *binds* the scope as a temporary `scope(path)` table your statement joins, rather than filtering behind the query's back (`JOIN scope ON scope."path" = f.path`). Naming a preset without joining `scope` is a usage error, since it would return everything while reading as scoped. Without the flag, join `preset_files` directly, which is the same coverage under a preset name you write into the SQL.
|
|
43
43
|
- `score` is a rank-fusion value: it ranks rows within one result set and is not comparable across queries, not a relevance magnitude. It encodes how many signals fired and at what rank, so a perfect lexical hit and a weak vector-only hit can read the same number. With vectors active, rows carry `similarity`: the cosine (-1 to 1) of the query against that file's best-matching chunk (the same chunk the `lines` range points at). It orders vector evidence within a result set; the range it spans depends on the corpus and the embedding model, and compresses on small trees, where even a nonsense query has a moderately near neighbour somewhere. Compare similarities within a result set rather than against a fixed cutoff carried between trees.
|
|
44
44
|
- Absence evidence lives in the labels: a `semantic: false` preset (or a tree with no `embed` block) returns 0 rows when the words are nowhere in it. Default `search` always returns up to `k` rows (nearest-neighbour search has a nearest neighbour for any input), so a result of only `via: vector` rows IS the absence signal for the words themselves; `similarity` and the snippet are the evidence for judging whether a vector row is a real conceptual hit.
|
|
45
|
-
- A `queries` entry names the verb it runs, mirroring the two commands: `"dead-links": { "sql": "SELECT src, target FROM links WHERE dst IS NULL" }` runs as `sense dead-links`, and `"hot": { "search": "pricing OR billing", "preset": "raw", "k": 20 }` runs as `sense hot` with its settings baked in, so repeat runs need no flags. An invocation-level `--preset`, `--k`, or `--where` overrides a saved search's value; `--list` labels each entry `(sql)` or `(search)`.
|
|
46
|
-
-
|
|
45
|
+
- A `queries` entry names the verb it runs, mirroring the two commands: `"dead-links": { "sql": "SELECT src, target FROM links WHERE dst IS NULL AND lower(target) NOT GLOB '*.[a-z0-9]*'" }` runs as `sense dead-links`, and `"hot": { "search": "pricing OR billing", "preset": "raw", "k": 20 }` runs as `sense hot` with its settings baked in, so repeat runs need no flags. An invocation-level `--preset`, `--k`, or `--where` overrides a saved search's value; `--list` labels each entry `(sql)` or `(search)`.
|
|
46
|
+
- Running an entry is how it is validated: a typo'd column, stale SQL, bad FTS5 syntax or an unknown preset errors and exits nonzero. A parameterised entry validates with any argument, since SQL is prepared before parameters bind (`sense by-tag zzz` reports `no such column` if the column is wrong, `(0 rows)` if it is right). To sweep a whole config after editing it: `for q in $(sense --list | awk '{print $1}'); do sense "$q" --format json >/dev/null || echo "broken: $q"; done`. Whether an empty result is good or bad is the reader's judgment: a dead-link query returning rows means broken citations to fix.
|
|
47
47
|
|
|
48
48
|
## SQL
|
|
49
49
|
|
|
@@ -53,8 +53,10 @@ The commands are shorthands over those tables; anything they don't express, SQL
|
|
|
53
53
|
sense sql "SELECT name FROM pragma_table_info('frontmatter')" # what fields exist
|
|
54
54
|
sense sql "SELECT DISTINCT status FROM frontmatter" # what values a field takes
|
|
55
55
|
sense sql "SELECT src FROM links WHERE dst = ?" notes/pricing-model.md # backlinks
|
|
56
|
-
sense sql "SELECT
|
|
56
|
+
sense sql "SELECT f.path FROM frontmatter f JOIN scope ON scope.path = f.path" --preset default # scope SQL to a preset
|
|
57
|
+
sense sql "SELECT f.path FROM frontmatter f JOIN preset_files p ON p.path = f.path AND p.preset = 'default'" # the same, preset named in the SQL
|
|
57
58
|
sense sql "SELECT path FROM frontmatter WHERE path NOT IN (SELECT dst FROM links WHERE dst IS NOT NULL) AND path NOT IN (SELECT src FROM links)" # linked neither way (fine if intentional; linking is optional)
|
|
59
|
+
sense sql "SELECT src, target FROM links WHERE dst IS NULL AND lower(target) NOT GLOB '*.[a-z0-9]*'" # broken wikilinks, attachments excluded
|
|
58
60
|
sense sql "SELECT heading, start_line, tokens FROM sections WHERE path = ?" a.md # budget a read
|
|
59
61
|
sense sql "SELECT j.value, COUNT(*) n FROM frontmatter, json_each(frontmatter.tags) j GROUP BY j.value ORDER BY n DESC" # count per array member
|
|
60
62
|
```
|
|
@@ -77,10 +79,13 @@ SELECT DISTINCT b.dst FROM links a JOIN links b ON a.src = b.src
|
|
|
77
79
|
WHERE a.dst = ? AND b.dst IS NOT NULL AND b.dst <> a.dst;
|
|
78
80
|
```
|
|
79
81
|
|
|
82
|
+
- A saved `{ sql }` written against `scope` is preset-agnostic: `sense <name> --preset raw` re-points the same statement at another layer, so one entry serves every preset instead of one copy each.
|
|
83
|
+
- `content MATCH` only works against the fts5 table by its own name, never through an alias or a view: `FROM content c ... WHERE c MATCH 'x'` fails with `no such column: c`. This is why `--preset` binds a table to join rather than shadowing the tables.
|
|
80
84
|
- `content MATCH` takes FTS5 syntax: `a OR b`, `"phrase"`, `pref*`, `NEAR(a b, 5)`, `summary: term`. Stemmed; markdown stripped at index time. Double-quote any term with punctuation. Bare `customer-facing` errors (`-` reads as a column filter), bare apostrophes are syntax errors: write `"customer-facing"`, `"founder's"`.
|
|
81
85
|
- Rank with `ORDER BY bm25(content, 10.0, 5.0, 1.0)` (title > summary > body); excerpt with `snippet(content, -1, '«', '»', '…', 10)`. snippet() re-tokenizes each matched doc and its cost grows superlinearly with doc size, measured ~10 s per query on a tree holding one 1 MB note. `search` bounds this itself (docs past 16 KB get an equivalent excerpt another way); in hand-written SQL, guard it: `CASE WHEN length(text) <= 16384 THEN snippet(...) END`, or select `title`/`summary` instead of an excerpt.
|
|
82
86
|
- Select `content.title`/`content.summary` (always exist, empty when absent) rather than `f.title`/`f.summary` (discovered columns; error on trees that never declare them).
|
|
83
87
|
- Frontmatter values keep their YAML type: strings are TEXT, whole numbers and booleans are INTEGER (`true` stores as 1, so `WHERE flag = 1` matches and `WHERE flag = 'true'` matches nothing), fractions are REAL, lists and maps are JSON text. `map` prints the observed type per field, and a field showing two types (`integer,text`) has drifted across notes.
|
|
88
|
+
- **Dead links need the attachment filter.** `dst IS NULL` alone is not "broken link": a wikilink to anything that is not markdown (`[[Board.base]]`, `![[Pasted image.png]]`, `[[spec.pdf]]`) can never resolve, because sense indexes markdown and resolution only tries the exact path or `+.md`. Those are out of the index's universe, not broken. On a 1,400-note Obsidian vault the unfiltered query returns 143 rows where 14 are real. Exclude anything carrying a file extension, as in the recipe above, and widen the exclusion if your notes have dotted titles (`[[Node.js]]` carries one too, so a stricter list -- `'*.png'`, `'*.pdf'`, `'*.base'`, and whatever else your vault attaches -- is safer on a tree whose titles use dots). Scope it with `preset_files` as well: template and skill files are full of `[[Note Name]]` examples that are deliberately unresolved.
|
|
84
89
|
- `has(field, value)`: array membership on JSON-array fields, substring on strings, false on NULL. This is the `includes()` convention. Substring means `has(f.status, 'active')` also matches `inactive`; exact scalar match is `f.status = ?`, deliberate substring is `LIKE`, exact array membership is `EXISTS (SELECT 1 FROM json_each(f.tags) WHERE value = ?)`. To aggregate per member instead, use `json_each(frontmatter.<field>)` (above) -- GROUP BY on the raw column splits `["a","b"]` and `["b","a"]` into separate buckets.
|
|
85
90
|
- Date fields are stored as written. Compare through `datetime()`, which normalizes ISO 8601 timezone offsets to UTC: `WHERE datetime(created) >= datetime(?)`. Bare string comparison is only safe when every note uses the same offset.
|
|
86
91
|
- To bound what a query puts into context: `snippet()` excerpts just the matching text, `LIMIT` caps row counts, and selecting `path`/`title`/`summary` keeps rows small. `SELECT text FROM content` returns the tree's entire prose (sense warns past 50 KB). Aggregates (`COUNT`, `GROUP BY`) are already bounded. `SELECT * FROM frontmatter` is always safe: prose is not a frontmatter column.
|
|
@@ -90,9 +95,10 @@ Worked traces: [EXAMPLES.md](EXAMPLES.md).
|
|
|
90
95
|
## Setup and upkeep
|
|
91
96
|
|
|
92
97
|
- Missing CLI: `npm install -g sensemaking`. Missing config: `sense init` at the tree root. Discovery walks up from cwd; `--config <path>` overrides. Setting up or restructuring a tree (presets, frontmatter conventions, note design) is the `sense-setup` skill.
|
|
93
|
-
- `map` and `status` report each preset's coverage (files matched, embedded count). Indexing derives from presets, so the coverage numbers are how you see what a config actually indexes and embeds. A scope with fewer signals just uses fewer (a semantic-off preset searches lexically); a saved search naming an unknown preset errors
|
|
98
|
+
- `map` and `status` report each preset's coverage (files matched, embedded count). Indexing derives from presets, so the coverage numbers are how you see what a config actually indexes and embeds. A scope with fewer signals just uses fewer (a semantic-off preset searches lexically); a saved search naming an unknown preset errors when run, listing the declared ones.
|
|
94
99
|
- Save a query into `sense.config.json` only when it will be reused; run ad-hoc otherwise.
|
|
95
100
|
- A one-line `summary:` per note is optional and pays twice: it appears in result rows and is a weighted search field. Date comparisons work for dates written as ISO 8601 (`2026-08-12`, or with time and offset), the only format `datetime()` parses. Field names in examples (`status`, `tags`, `created`) are illustrative; your tree defines its own.
|
|
96
|
-
- Reserved frontmatter keys (dropped with a warning): `path`, `_mtime`, `_size`, `_rank`, `content`, `links`, `sections`.
|
|
101
|
+
- Reserved frontmatter keys (dropped with a warning): `path`, `_mtime`, `_size`, `_rank`, `_parse_error`, `content`, `links`, `sections`.
|
|
102
|
+
- A note whose frontmatter does not parse is indexed with **no** frontmatter columns and `_parse_error` set to the YAML message, which carries the line. Nothing is half-recovered: a non-NULL value is a value the author wrote. So a NULL column means the key was absent *or* the note did not parse, and `_parse_error` is how you tell: `WHERE status IS NULL AND _parse_error IS NULL` is "genuinely missing status". List what needs fixing with `sense sql "SELECT path, _parse_error FROM frontmatter WHERE _parse_error IS NOT NULL"`; fixing a file clears it on the next command. `sense status` reports the count.
|
|
97
103
|
- Exit codes: `0` ok, `1` error (SQLite message verbatim), `2` usage (unknown query, wrong param count).
|
|
98
|
-
- Doubted cache: `sense
|
|
104
|
+
- Doubted cache: delete the directory `sense status` prints on its `cache:` line. Rarely needed; every query reconciles first.
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
---
|
|
2
2
|
name: sense-setup
|
|
3
|
-
description: Set up the sense CLI on a markdown tree and make the tree-design decisions that shape it: sense init, presets (which files, which settings, whether the scope searches by meaning), the embed block that names the model, and the trade-offs of frontmatter conventions, summaries, folder layout, and note size. Use when creating or restructuring a markdown knowledge base, running sense init, editing sense.config.json, configuring search scope or vectors, or deciding how notes should be written for an agent to query later.
|
|
3
|
+
description: "Set up the sense CLI on a markdown tree and make the tree-design decisions that shape it: sense init, presets (which files, which settings, whether the scope searches by meaning), the embed block that names the model, and the trade-offs of frontmatter conventions, summaries, folder layout, and note size. Use when creating or restructuring a markdown knowledge base, running sense init, editing sense.config.json, configuring search scope or vectors, or deciding how notes should be written for an agent to query later."
|
|
4
4
|
---
|
|
5
5
|
|
|
6
6
|
# sense: setup and tree design
|
|
@@ -45,5 +45,5 @@ These belong to the tree's owner. sense works with any of them and reads no inst
|
|
|
45
45
|
- **Summaries.** A one-line `summary:` is optional and pays twice: it shows in every result row (often answering a question with no file read) and is a weighted search field ranked above body text. The cost is writing and maintaining the line as notes change.
|
|
46
46
|
- **Folder shape.** Globs find the files, paths are queryable text, links resolve by basename at any depth, but presets make folders meaningful: a folder is the natural unit that gets its own coverage and settings.
|
|
47
47
|
- **Note size.** Many small notes: precise search hits, whole-file reads stay cheap, more links to maintain. Fewer large notes: `sections`, `peek`, and the `lines` column carry the cost down to line-range reads. Both work.
|
|
48
|
-
- **Recurring questions.** Save a scenario an agent will repeat under `queries`, naming the verb it runs: `{ "sql": "..." }` for filters and reports, or `{ "search": "...", "preset": "raw", "k": 5 }` for a ranked search. Either runs as `sense <name>`, and
|
|
48
|
+
- **Recurring questions.** Save a scenario an agent will repeat under `queries`, naming the verb it runs: `{ "sql": "..." }` for filters and reports, or `{ "search": "...", "preset": "raw", "k": 5 }` for a ranked search. Either runs as `sense <name>`, and running one is how it is validated: a typo'd column or an unknown preset errors and exits nonzero. A parameterised entry validates with any argument, since SQL is prepared before parameters bind.
|
|
49
49
|
- **Where decisions live.** Choices that should outlive one conversation can be recorded in the agent's own instruction or skill files, or in a note in the tree itself; a one-off search over an existing corpus needs none of that.
|
package/dist/cjs/cli/check.d.cts
DELETED
package/dist/cjs/cli/check.d.ts
DELETED