sensemaking 0.11.4 → 0.12.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (72) hide show
  1. package/README.md +4 -3
  2. package/bin/cli.js +5 -2
  3. package/dist/cjs/cli/index.d.cts +3 -3
  4. package/dist/cjs/cli/index.d.ts +3 -3
  5. package/dist/cjs/cli/index.js +3 -3
  6. package/dist/cjs/cli/index.js.map +1 -1
  7. package/dist/cjs/cli/named.js +2 -2
  8. package/dist/cjs/cli/named.js.map +1 -1
  9. package/dist/cjs/cli/related.js +1 -1
  10. package/dist/cjs/cli/related.js.map +1 -1
  11. package/dist/cjs/cli/search.js +1 -1
  12. package/dist/cjs/cli/search.js.map +1 -1
  13. package/dist/cjs/cli/shared.d.cts +3 -1
  14. package/dist/cjs/cli/shared.d.ts +3 -1
  15. package/dist/cjs/cli/shared.js +35 -5
  16. package/dist/cjs/cli/shared.js.map +1 -1
  17. package/dist/cjs/cli/sql.js +1 -1
  18. package/dist/cjs/cli/sql.js.map +1 -1
  19. package/dist/cjs/cli.js +14 -1
  20. package/dist/cjs/cli.js.map +1 -1
  21. package/dist/cjs/commands.js +19 -5
  22. package/dist/cjs/commands.js.map +1 -1
  23. package/dist/cjs/config.d.cts +4 -0
  24. package/dist/cjs/config.d.ts +4 -0
  25. package/dist/cjs/config.js +39 -4
  26. package/dist/cjs/config.js.map +1 -1
  27. package/dist/cjs/db.d.cts +1 -1
  28. package/dist/cjs/db.d.ts +1 -1
  29. package/dist/cjs/db.js +158 -14
  30. package/dist/cjs/db.js.map +1 -1
  31. package/dist/cjs/output.d.cts +4 -1
  32. package/dist/cjs/output.d.ts +4 -1
  33. package/dist/cjs/output.js +158 -6
  34. package/dist/cjs/output.js.map +1 -1
  35. package/dist/cjs/segment.d.cts +2 -0
  36. package/dist/cjs/segment.d.ts +2 -0
  37. package/dist/cjs/segment.js +98 -0
  38. package/dist/cjs/segment.js.map +1 -0
  39. package/dist/esm/cli/index.d.ts +3 -3
  40. package/dist/esm/cli/index.js +3 -3
  41. package/dist/esm/cli/index.js.map +1 -1
  42. package/dist/esm/cli/named.js +3 -3
  43. package/dist/esm/cli/named.js.map +1 -1
  44. package/dist/esm/cli/related.js +2 -2
  45. package/dist/esm/cli/related.js.map +1 -1
  46. package/dist/esm/cli/search.js +2 -2
  47. package/dist/esm/cli/search.js.map +1 -1
  48. package/dist/esm/cli/shared.d.ts +3 -1
  49. package/dist/esm/cli/shared.js +32 -5
  50. package/dist/esm/cli/shared.js.map +1 -1
  51. package/dist/esm/cli/sql.js +2 -2
  52. package/dist/esm/cli/sql.js.map +1 -1
  53. package/dist/esm/cli.js +14 -1
  54. package/dist/esm/cli.js.map +1 -1
  55. package/dist/esm/commands.js +19 -5
  56. package/dist/esm/commands.js.map +1 -1
  57. package/dist/esm/config.d.ts +4 -0
  58. package/dist/esm/config.js +37 -4
  59. package/dist/esm/config.js.map +1 -1
  60. package/dist/esm/db.d.ts +1 -1
  61. package/dist/esm/db.js +140 -15
  62. package/dist/esm/db.js.map +1 -1
  63. package/dist/esm/output.d.ts +4 -1
  64. package/dist/esm/output.js +104 -4
  65. package/dist/esm/output.js.map +1 -1
  66. package/dist/esm/segment.d.ts +2 -0
  67. package/dist/esm/segment.js +65 -0
  68. package/dist/esm/segment.js.map +1 -0
  69. package/package.json +4 -1
  70. package/schema.json +12 -1
  71. package/skills/sense/SKILL.md +7 -6
  72. package/skills/sense-setup/SKILL.md +1 -0
@@ -1 +1 @@
1
- {"version":3,"sources":["/Users/kevin/Dev/OpenSource/ai/sensemaking/src/output.ts"],"sourcesContent":["// rows -> table (default) | json\n\nexport type Row = Record<string, unknown>;\n\n// Warned about on stderr (keeps --format json stdout machine-readable), never truncated.\nconst OVERSIZE_BYTES = 50_000;\n\n// Flatten embedded newlines so they can't break table column alignment; JSON output is left alone.\nfunction cell(value: unknown): string {\n return String(value ?? '').replace(/\\s*\\n\\s*/g, ' ');\n}\n\nexport function printRows(rows: Row[], format: 'table' | 'json'): void {\n const rendered = renderRows(rows, format);\n console.log(rendered);\n if (rendered.length > OVERSIZE_BYTES) {\n console.warn(`warning: result is ${Math.round(rendered.length / 1000)} KB across ${rows.length} row(s); consider a LIMIT, fewer columns, or snippet() instead of whole values`);\n }\n}\n\n// Tables are for reading, and a row wider than the terminal wraps unreadably: widest columns\n// give up space first, down to a floor. --format json is never truncated.\nconst MIN_COLUMN = 12;\nconst TRUNCATED = '…';\n\nfunction fitWidths(widths: number[], limit: number): number[] {\n const gaps = (widths.length - 1) * 2;\n const fitted = [...widths];\n for (;;) {\n const total = fitted.reduce((a, b) => a + b, 0) + gaps;\n if (total <= limit) return fitted;\n const widest = fitted.indexOf(Math.max(...fitted));\n if (fitted[widest] <= MIN_COLUMN) return fitted;\n fitted[widest] = Math.max(MIN_COLUMN, fitted[widest] - Math.max(1, total - limit));\n }\n}\n\nfunction renderRows(rows: Row[], format: 'table' | 'json', width = process.stdout.columns): string {\n if (format === 'json') return JSON.stringify(rows, null, 2);\n if (rows.length === 0) return '(0 rows)';\n\n const columns = Object.keys(rows[0]);\n const cells = rows.map((row) => columns.map((col) => cell(row[col])));\n const natural = columns.map((col, i) => Math.max(col.length, ...cells.map((row) => row[i].length)));\n const widths = width && width > MIN_COLUMN ? fitWidths(natural, width) : natural;\n\n const clip = (value: string, w: number) => (value.length <= w ? value.padEnd(w) : `${value.slice(0, w - 1)}${TRUNCATED}`);\n const formatRow = (values: string[]) =>\n values\n .map((value, i) => clip(value, widths[i]))\n .join(' ')\n .trimEnd();\n\n return [formatRow(columns), widths.map((w) => '-'.repeat(w)).join(' '), ...cells.map(formatRow)].join('\\n');\n}\n\n// Text renderers for the top-level commands; cli.ts prints what these return.\n\n// Names the config key that turns a feature back on. Only the three features-block toggles\n// reach here; `embed` has its own status line.\nexport function featureOffHint(name: string): string {\n return `features.${name}`;\n}\n\nexport function featuresLine(features: { on: string[]; off: string[] }): string {\n const off = features.off.length > 0 ? ` · off: ${features.off.map((name) => `${name} (${featureOffHint(name)})`).join(', ')}` : '';\n return `features: ${features.on.join(', ')}${off}`;\n}\n\nexport function presetsLines(presets: Array<{ name: string; files: number; embedded: number; semantic?: boolean }>): string[] {\n // \"0 embedded\" is ambiguous between a scope that declined vectors and one that has not built\n // them yet, so a semantic-off preset says which.\n return ['presets:', ...presets.map((p) => ` ${p.name}: ${p.files} file(s), ${p.embedded} embedded${p.semantic === false ? ' (semantic: false)' : ''}`)];\n}\n\nexport function renderMap(result: { docs: { count: number; bytes: number }; fields: Row[]; fieldsTotal: number; features: { on: string[]; off: string[] }; presets: Array<{ name: string; files: number; embedded: number; semantic?: boolean }>; hubs: Row[]; recent: Row[] }): string {\n const parts = [`docs: ${result.docs.count} (${Math.round(result.docs.bytes / 1024)} KB)`, featuresLine(result.features), '', ...presetsLines(result.presets), '', renderRows(result.fields, 'table')];\n if (result.fieldsTotal > result.fields.length) parts.push(`(+${result.fieldsTotal - result.fields.length} more fields)`);\n if (result.hubs.length > 0) parts.push('\\nhubs (by link rank):', renderRows(result.hubs, 'table'));\n else if (result.features.off.includes('rank')) parts.push(`\\nhubs: off (${featureOffHint('rank')})`);\n parts.push('\\nrecent:', renderRows(result.recent, 'table'));\n return parts.join('\\n');\n}\n\nexport function renderPeek(result: { path: string; tokens: number; frontmatter: Row; parseError?: string | null; sections: Row[]; outbound: string[]; backlinks: string[]; unresolved: string[]; sectionsTotal: number; outboundTotal: number; backlinksTotal: number; unresolvedTotal: number; off: string[] }): string {\n const lines = [`${result.path} (~${result.tokens} tokens)`];\n // Otherwise a refused parse is indistinguishable from a note that has no frontmatter, which\n // is the confusion the whole quarantine design exists to remove.\n if (result.parseError) lines.push(` frontmatter: did not parse, so none of it is indexed (${result.parseError})`);\n for (const [key, value] of Object.entries(result.frontmatter)) lines.push(` ${key}: ${value}`);\n if (result.sections.length > 0) {\n lines.push('', 'sections:');\n for (const s of result.sections) lines.push(` ${'#'.repeat(s.level as number)} ${s.heading} [L${s.start_line}-${s.end_line}, ~${s.tokens}t]`);\n if (result.sectionsTotal > result.sections.length) lines.push(` (+${result.sectionsTotal - result.sections.length} more sections -- sections table has all of them)`);\n } else if (result.off.includes('sections')) lines.push('', 'sections: off (features.sections)');\n if (result.off.includes('links')) {\n lines.push('', 'links: off (features.links)');\n } else {\n const linkLine = (label: string, shown: string[], total: number) => {\n if (total > 0) lines.push(`${label} (${total}): ${shown.join(', ')}${total > shown.length ? `, +${total - shown.length} more` : ''}`);\n };\n if (result.outboundTotal + result.unresolvedTotal + result.backlinksTotal > 0) lines.push('');\n linkLine('links out', result.outbound, result.outboundTotal);\n linkLine('unresolved', result.unresolved, result.unresolvedTotal);\n linkLine('backlinks', result.backlinks, result.backlinksTotal);\n }\n return lines.join('\\n');\n}\n"],"names":["OVERSIZE_BYTES","cell","value","String","replace","printRows","rows","format","rendered","renderRows","console","log","length","warn","Math","round","MIN_COLUMN","TRUNCATED","fitWidths","widths","limit","gaps","fitted","total","reduce","a","b","widest","indexOf","max","width","process","stdout","columns","JSON","stringify","Object","keys","cells","map","row","col","natural","i","clip","w","padEnd","slice","formatRow","values","join","trimEnd","repeat","featureOffHint","name","featuresLine","features","off","on","presetsLines","presets","p","files","embedded","semantic","renderMap","result","parts","docs","count","bytes","fields","fieldsTotal","push","hubs","includes","recent","renderPeek","lines","path","tokens","parseError","key","entries","frontmatter","sections","s","level","heading","start_line","end_line","sectionsTotal","linkLine","label","shown","outboundTotal","unresolvedTotal","backlinksTotal","outbound","unresolved","backlinks"],"mappings":"AAAA,iCAAiC;AAIjC,yFAAyF;AACzF,MAAMA,iBAAiB;AAEvB,mGAAmG;AACnG,SAASC,KAAKC,KAAc;IAC1B,OAAOC,OAAOD,kBAAAA,mBAAAA,QAAS,IAAIE,OAAO,CAAC,aAAa;AAClD;AAEA,OAAO,SAASC,UAAUC,IAAW,EAAEC,MAAwB;IAC7D,MAAMC,WAAWC,WAAWH,MAAMC;IAClCG,QAAQC,GAAG,CAACH;IACZ,IAAIA,SAASI,MAAM,GAAGZ,gBAAgB;QACpCU,QAAQG,IAAI,CAAC,CAAC,mBAAmB,EAAEC,KAAKC,KAAK,CAACP,SAASI,MAAM,GAAG,MAAM,WAAW,EAAEN,KAAKM,MAAM,CAAC,8EAA8E,CAAC;IAChL;AACF;AAEA,6FAA6F;AAC7F,0EAA0E;AAC1E,MAAMI,aAAa;AACnB,MAAMC,YAAY;AAElB,SAASC,UAAUC,MAAgB,EAAEC,KAAa;IAChD,MAAMC,OAAO,AAACF,CAAAA,OAAOP,MAAM,GAAG,CAAA,IAAK;IACnC,MAAMU,SAAS;WAAIH;KAAO;IAC1B,OAAS;QACP,MAAMI,QAAQD,OAAOE,MAAM,CAAC,CAACC,GAAGC,IAAMD,IAAIC,GAAG,KAAKL;QAClD,IAAIE,SAASH,OAAO,OAAOE;QAC3B,MAAMK,SAASL,OAAOM,OAAO,CAACd,KAAKe,GAAG,IAAIP;QAC1C,IAAIA,MAAM,CAACK,OAAO,IAAIX,YAAY,OAAOM;QACzCA,MAAM,CAACK,OAAO,GAAGb,KAAKe,GAAG,CAACb,YAAYM,MAAM,CAACK,OAAO,GAAGb,KAAKe,GAAG,CAAC,GAAGN,QAAQH;IAC7E;AACF;AAEA,SAASX,WAAWH,IAAW,EAAEC,MAAwB,EAAEuB,QAAQC,QAAQC,MAAM,CAACC,OAAO;IACvF,IAAI1B,WAAW,QAAQ,OAAO2B,KAAKC,SAAS,CAAC7B,MAAM,MAAM;IACzD,IAAIA,KAAKM,MAAM,KAAK,GAAG,OAAO;IAE9B,MAAMqB,UAAUG,OAAOC,IAAI,CAAC/B,IAAI,CAAC,EAAE;IACnC,MAAMgC,QAAQhC,KAAKiC,GAAG,CAAC,CAACC,MAAQP,QAAQM,GAAG,CAAC,CAACE,MAAQxC,KAAKuC,GAAG,CAACC,IAAI;IAClE,MAAMC,UAAUT,QAAQM,GAAG,CAAC,CAACE,KAAKE,IAAM7B,KAAKe,GAAG,CAACY,IAAI7B,MAAM,KAAK0B,MAAMC,GAAG,CAAC,CAACC,MAAQA,GAAG,CAACG,EAAE,CAAC/B,MAAM;IAChG,MAAMO,SAASW,SAASA,QAAQd,aAAaE,UAAUwB,SAASZ,SAASY;IAEzE,MAAME,OAAO,CAAC1C,OAAe2C,IAAe3C,MAAMU,MAAM,IAAIiC,IAAI3C,MAAM4C,MAAM,CAACD,KAAK,GAAG3C,MAAM6C,KAAK,CAAC,GAAGF,IAAI,KAAK5B,WAAW;IACxH,MAAM+B,YAAY,CAACC,SACjBA,OACGV,GAAG,CAAC,CAACrC,OAAOyC,IAAMC,KAAK1C,OAAOiB,MAAM,CAACwB,EAAE,GACvCO,IAAI,CAAC,MACLC,OAAO;IAEZ,OAAO;QAACH,UAAUf;QAAUd,OAAOoB,GAAG,CAAC,CAACM,IAAM,IAAIO,MAAM,CAACP,IAAIK,IAAI,CAAC;WAAUZ,MAAMC,GAAG,CAACS;KAAW,CAACE,IAAI,CAAC;AACzG;AAEA,8EAA8E;AAE9E,2FAA2F;AAC3F,+CAA+C;AAC/C,OAAO,SAASG,eAAeC,IAAY;IACzC,OAAO,CAAC,SAAS,EAAEA,MAAM;AAC3B;AAEA,OAAO,SAASC,aAAaC,QAAyC;IACpE,MAAMC,MAAMD,SAASC,GAAG,CAAC7C,MAAM,GAAG,IAAI,CAAC,QAAQ,EAAE4C,SAASC,GAAG,CAAClB,GAAG,CAAC,CAACe,OAAS,GAAGA,KAAK,EAAE,EAAED,eAAeC,MAAM,CAAC,CAAC,EAAEJ,IAAI,CAAC,OAAO,GAAG;IAChI,OAAO,CAAC,UAAU,EAAEM,SAASE,EAAE,CAACR,IAAI,CAAC,QAAQO,KAAK;AACpD;AAEA,OAAO,SAASE,aAAaC,OAAqF;IAChH,6FAA6F;IAC7F,iDAAiD;IACjD,OAAO;QAAC;WAAeA,QAAQrB,GAAG,CAAC,CAACsB,IAAM,CAAC,EAAE,EAAEA,EAAEP,IAAI,CAAC,EAAE,EAAEO,EAAEC,KAAK,CAAC,UAAU,EAAED,EAAEE,QAAQ,CAAC,SAAS,EAAEF,EAAEG,QAAQ,KAAK,QAAQ,uBAAuB,IAAI;KAAE;AAC1J;AAEA,OAAO,SAASC,UAAUC,MAAoP;IAC5Q,MAAMC,QAAQ;QAAC,CAAC,MAAM,EAAED,OAAOE,IAAI,CAACC,KAAK,CAAC,EAAE,EAAEvD,KAAKC,KAAK,CAACmD,OAAOE,IAAI,CAACE,KAAK,GAAG,MAAM,IAAI,CAAC;QAAEf,aAAaW,OAAOV,QAAQ;QAAG;WAAOG,aAAaO,OAAON,OAAO;QAAG;QAAInD,WAAWyD,OAAOK,MAAM,EAAE;KAAS;IACrM,IAAIL,OAAOM,WAAW,GAAGN,OAAOK,MAAM,CAAC3D,MAAM,EAAEuD,MAAMM,IAAI,CAAC,CAAC,EAAE,EAAEP,OAAOM,WAAW,GAAGN,OAAOK,MAAM,CAAC3D,MAAM,CAAC,aAAa,CAAC;IACvH,IAAIsD,OAAOQ,IAAI,CAAC9D,MAAM,GAAG,GAAGuD,MAAMM,IAAI,CAAC,0BAA0BhE,WAAWyD,OAAOQ,IAAI,EAAE;SACpF,IAAIR,OAAOV,QAAQ,CAACC,GAAG,CAACkB,QAAQ,CAAC,SAASR,MAAMM,IAAI,CAAC,CAAC,aAAa,EAAEpB,eAAe,QAAQ,CAAC,CAAC;IACnGc,MAAMM,IAAI,CAAC,aAAahE,WAAWyD,OAAOU,MAAM,EAAE;IAClD,OAAOT,MAAMjB,IAAI,CAAC;AACpB;AAEA,OAAO,SAAS2B,WAAWX,MAAoR;IAC7S,MAAMY,QAAQ;QAAC,GAAGZ,OAAOa,IAAI,CAAC,IAAI,EAAEb,OAAOc,MAAM,CAAC,QAAQ,CAAC;KAAC;IAC5D,4FAA4F;IAC5F,iEAAiE;IACjE,IAAId,OAAOe,UAAU,EAAEH,MAAML,IAAI,CAAC,CAAC,wDAAwD,EAAEP,OAAOe,UAAU,CAAC,CAAC,CAAC;IACjH,KAAK,MAAM,CAACC,KAAKhF,MAAM,IAAIkC,OAAO+C,OAAO,CAACjB,OAAOkB,WAAW,EAAGN,MAAML,IAAI,CAAC,CAAC,EAAE,EAAES,IAAI,EAAE,EAAEhF,OAAO;IAC9F,IAAIgE,OAAOmB,QAAQ,CAACzE,MAAM,GAAG,GAAG;QAC9BkE,MAAML,IAAI,CAAC,IAAI;QACf,KAAK,MAAMa,KAAKpB,OAAOmB,QAAQ,CAAEP,MAAML,IAAI,CAAC,CAAC,EAAE,EAAE,IAAIrB,MAAM,CAACkC,EAAEC,KAAK,EAAY,CAAC,EAAED,EAAEE,OAAO,CAAC,IAAI,EAAEF,EAAEG,UAAU,CAAC,CAAC,EAAEH,EAAEI,QAAQ,CAAC,GAAG,EAAEJ,EAAEN,MAAM,CAAC,EAAE,CAAC;QAC9I,IAAId,OAAOyB,aAAa,GAAGzB,OAAOmB,QAAQ,CAACzE,MAAM,EAAEkE,MAAML,IAAI,CAAC,CAAC,IAAI,EAAEP,OAAOyB,aAAa,GAAGzB,OAAOmB,QAAQ,CAACzE,MAAM,CAAC,iDAAiD,CAAC;IACvK,OAAO,IAAIsD,OAAOT,GAAG,CAACkB,QAAQ,CAAC,aAAaG,MAAML,IAAI,CAAC,IAAI;IAC3D,IAAIP,OAAOT,GAAG,CAACkB,QAAQ,CAAC,UAAU;QAChCG,MAAML,IAAI,CAAC,IAAI;IACjB,OAAO;QACL,MAAMmB,WAAW,CAACC,OAAeC,OAAiBvE;YAChD,IAAIA,QAAQ,GAAGuD,MAAML,IAAI,CAAC,GAAGoB,MAAM,EAAE,EAAEtE,MAAM,GAAG,EAAEuE,MAAM5C,IAAI,CAAC,QAAQ3B,QAAQuE,MAAMlF,MAAM,GAAG,CAAC,GAAG,EAAEW,QAAQuE,MAAMlF,MAAM,CAAC,KAAK,CAAC,GAAG,IAAI;QACtI;QACA,IAAIsD,OAAO6B,aAAa,GAAG7B,OAAO8B,eAAe,GAAG9B,OAAO+B,cAAc,GAAG,GAAGnB,MAAML,IAAI,CAAC;QAC1FmB,SAAS,aAAa1B,OAAOgC,QAAQ,EAAEhC,OAAO6B,aAAa;QAC3DH,SAAS,cAAc1B,OAAOiC,UAAU,EAAEjC,OAAO8B,eAAe;QAChEJ,SAAS,aAAa1B,OAAOkC,SAAS,EAAElC,OAAO+B,cAAc;IAC/D;IACA,OAAOnB,MAAM5B,IAAI,CAAC;AACpB"}
1
+ {"version":3,"sources":["/Users/kevin/Dev/OpenSource/ai/sensemaking/src/output.ts"],"sourcesContent":["// rows -> table (default) | json | csv\n\nexport type Row = Record<string, unknown>;\n\n// map/peek/status/path render structures rather than a row set, so csv has nothing to mean\n// there; the commands that emit rows take RowFormat.\nexport type Format = 'table' | 'json';\nexport type RowFormat = Format | 'csv';\n\n// Warned about on stderr (keeps --format json stdout machine-readable), never truncated.\nconst OVERSIZE_BYTES = 50_000;\n\n// Flatten embedded newlines so they can't break table column alignment; JSON output is left alone.\nfunction cell(value: unknown): string {\n return String(value ?? '').replace(/\\s*\\n\\s*/g, ' ');\n}\n\n// RFC 4180, and deliberately not cell(): csv is the format for redirecting a result to a\n// file, so it keeps every character, embedded newlines included. What it cannot carry is\n// NULL vs empty string, or a value's type -- that is what json is for.\nfunction csvField(value: unknown): string {\n const text = value === null || value === undefined ? '' : String(value);\n return /[\"\\r\\n,]/.test(text) ? `\"${text.replace(/\"/g, '\"\"')}\"` : text;\n}\n\nfunction csvLine(values: unknown[]): string {\n return values.map(csvField).join(',');\n}\n\nfunction warnOversize(bytes: number, rowCount: number): void {\n if (bytes <= OVERSIZE_BYTES) return;\n console.warn(`warning: result is ${Math.round(bytes / 1000)} KB across ${rowCount} row(s); consider a LIMIT, fewer columns, snippet() instead of whole values, or --format csv redirected to a file`);\n}\n\n// Header names come from a row wherever there is one: a statement's column list keeps\n// duplicates that the row object has already collapsed, so `SELECT a AS x, b AS x` would print\n// b's value under both headers. `columns` is the fallback that labels a 0-row result, and with\n// no rows and no statement (a bounded command that found nothing) csv writes nothing at all --\n// a bare newline would read to a csv parser as one empty record.\nexport function printRows(rows: Row[], format: RowFormat, columns?: string[]): void {\n if (format === 'csv') {\n // Collapsed the same way a row object collapses them, so a statement naming a column\n // twice describes itself identically whether it matched anything or not.\n const names = rows.length > 0 ? Object.keys(rows[0]) : [...new Set(columns ?? [])];\n if (names.length === 0) return;\n const rendered = [csvLine(names), ...rows.map((row) => csvLine(names.map((name) => row[name])))].join('\\n');\n console.log(rendered);\n // Bytes, not code units -- CJK is 3 bytes per character.\n warnOversize(Buffer.byteLength(rendered), rows.length);\n return;\n }\n const rendered = renderRows(rows, format);\n console.log(rendered);\n warnOversize(Buffer.byteLength(rendered), rows.length);\n}\n\n// Streaming counterpart for the one unbounded caller, `sense sql`: rows arrive from an\n// iterator, so a large result is never held here as rows and again as a rendered string.\n// Table buffers by necessity (a column is as wide as its widest cell, which the last row can\n// change); json and csv write row by row. The writes are synchronous, so when stdout is a pipe\n// Node buffers what the reader has not taken yet -- the saving is this module's copies, not\n// the operating system's.\nexport function printRowStream(rows: Iterable<Row>, format: RowFormat, columns: string[]): void {\n const iterator = rows[Symbol.iterator]();\n // One step before any write. FTS5 parses a MATCH expression at step time rather than at\n // prepare time, so a bad search string throws here, with stdout still clean for the error.\n const first = iterator.next();\n const rest: Iterable<Row> = { [Symbol.iterator]: () => iterator };\n\n if (format === 'table') {\n printRows(first.done ? [] : [first.value, ...rest], format, columns);\n return;\n }\n\n let bytes = 0;\n let rowCount = 0;\n const write = (text: string): void => {\n // Bytes, not code units -- CJK is 3 bytes per character.\n bytes += Buffer.byteLength(text);\n process.stdout.write(text);\n };\n\n if (format === 'json') {\n // Byte-identical to JSON.stringify(rows, null, 2) + newline: each row re-indented one\n // level, comma-joined, inside the array brackets written here.\n // JSON.stringify throws on BigInt, so a row with an int64 past 2^53 (setReadBigInts) is\n // converted via the replacer: a value that fits a safe integer stays a number, a larger one\n // becomes its decimal string, since a JSON number cannot carry it losslessly.\n const MAX_SAFE = BigInt(Number.MAX_SAFE_INTEGER);\n const bigintSafe = (_key: string, value: unknown) => (typeof value === 'bigint' ? (value >= -MAX_SAFE && value <= MAX_SAFE ? Number(value) : value.toString()) : value);\n const indent = (row: Row) =>\n JSON.stringify(row, bigintSafe, 2)\n .split('\\n')\n .map((line) => ` ${line}`)\n .join('\\n');\n if (first.done) {\n write('[]\\n');\n } else {\n write('[\\n');\n write(indent(first.value));\n rowCount = 1;\n for (const row of rest) {\n write(',\\n');\n write(indent(row));\n rowCount++;\n }\n write('\\n]\\n');\n }\n } else if (first.done) {\n const names = [...new Set(columns)];\n if (names.length > 0) write(`${csvLine(names)}\\n`);\n } else {\n const names = Object.keys(first.value);\n write(`${csvLine(names)}\\n`);\n write(`${csvLine(names.map((name) => first.value[name]))}\\n`);\n rowCount = 1;\n for (const row of rest) {\n write(`${csvLine(names.map((name) => row[name]))}\\n`);\n rowCount++;\n }\n }\n warnOversize(bytes, rowCount);\n}\n\n// Tables are for reading, and a row wider than the terminal wraps unreadably: widest columns\n// give up space first, down to a floor. --format json is never truncated.\nconst MIN_COLUMN = 12;\nconst TRUNCATED = '…';\n\nfunction fitWidths(widths: number[], limit: number): number[] {\n const gaps = (widths.length - 1) * 2;\n const fitted = [...widths];\n for (;;) {\n const total = fitted.reduce((a, b) => a + b, 0) + gaps;\n if (total <= limit) return fitted;\n const widest = fitted.indexOf(Math.max(...fitted));\n if (fitted[widest] <= MIN_COLUMN) return fitted;\n fitted[widest] = Math.max(MIN_COLUMN, fitted[widest] - Math.max(1, total - limit));\n }\n}\n\nfunction renderRows(rows: Row[], format: Format, width = process.stdout.columns): string {\n if (format === 'json') return JSON.stringify(rows, null, 2);\n if (rows.length === 0) return '(0 rows)';\n\n const columns = Object.keys(rows[0]);\n const cells = rows.map((row) => columns.map((col) => cell(row[col])));\n const natural = columns.map((col, i) => Math.max(col.length, ...cells.map((row) => row[i].length)));\n const widths = width && width > MIN_COLUMN ? fitWidths(natural, width) : natural;\n\n const clip = (value: string, w: number) => (value.length <= w ? value.padEnd(w) : `${value.slice(0, w - 1)}${TRUNCATED}`);\n const formatRow = (values: string[]) =>\n values\n .map((value, i) => clip(value, widths[i]))\n .join(' ')\n .trimEnd();\n\n return [formatRow(columns), widths.map((w) => '-'.repeat(w)).join(' '), ...cells.map(formatRow)].join('\\n');\n}\n\n// Text renderers for the top-level commands; cli.ts prints what these return.\n\n// Names the config key that turns a feature back on. Only the three features-block toggles\n// reach here; `embed` has its own status line.\nexport function featureOffHint(name: string): string {\n return `features.${name}`;\n}\n\nexport function featuresLine(features: { on: string[]; off: string[] }): string {\n const off = features.off.length > 0 ? ` · off: ${features.off.map((name) => `${name} (${featureOffHint(name)})`).join(', ')}` : '';\n return `features: ${features.on.join(', ')}${off}`;\n}\n\nexport function presetsLines(presets: Array<{ name: string; files: number; embedded: number; semantic?: boolean }>): string[] {\n // \"0 embedded\" is ambiguous between a scope that declined vectors and one that has not built\n // them yet, so a semantic-off preset says which.\n return ['presets:', ...presets.map((p) => ` ${p.name}: ${p.files} file(s), ${p.embedded} embedded${p.semantic === false ? ' (semantic: false)' : ''}`)];\n}\n\nexport function renderMap(result: { docs: { count: number; bytes: number }; fields: Row[]; fieldsTotal: number; features: { on: string[]; off: string[] }; presets: Array<{ name: string; files: number; embedded: number; semantic?: boolean }>; hubs: Row[]; recent: Row[] }): string {\n const parts = [`docs: ${result.docs.count} (${Math.round(result.docs.bytes / 1024)} KB)`, featuresLine(result.features), '', ...presetsLines(result.presets), '', renderRows(result.fields, 'table')];\n if (result.fieldsTotal > result.fields.length) parts.push(`(+${result.fieldsTotal - result.fields.length} more fields)`);\n if (result.hubs.length > 0) parts.push('\\nhubs (by link rank):', renderRows(result.hubs, 'table'));\n else if (result.features.off.includes('rank')) parts.push(`\\nhubs: off (${featureOffHint('rank')})`);\n parts.push('\\nrecent:', renderRows(result.recent, 'table'));\n return parts.join('\\n');\n}\n\nexport function renderPeek(result: { path: string; tokens: number; frontmatter: Row; parseError?: string | null; sections: Row[]; outbound: string[]; backlinks: string[]; unresolved: string[]; sectionsTotal: number; outboundTotal: number; backlinksTotal: number; unresolvedTotal: number; off: string[] }): string {\n const lines = [`${result.path} (~${result.tokens} tokens)`];\n // Otherwise a refused parse is indistinguishable from a note that has no frontmatter, which\n // is the confusion the whole quarantine design exists to remove.\n if (result.parseError) lines.push(` frontmatter: did not parse, so none of it is indexed (${result.parseError})`);\n for (const [key, value] of Object.entries(result.frontmatter)) lines.push(` ${key}: ${value}`);\n if (result.sections.length > 0) {\n lines.push('', 'sections:');\n for (const s of result.sections) lines.push(` ${'#'.repeat(s.level as number)} ${s.heading} [L${s.start_line}-${s.end_line}, ~${s.tokens}t]`);\n if (result.sectionsTotal > result.sections.length) lines.push(` (+${result.sectionsTotal - result.sections.length} more sections -- sections table has all of them)`);\n } else if (result.off.includes('sections')) lines.push('', 'sections: off (features.sections)');\n if (result.off.includes('links')) {\n lines.push('', 'links: off (features.links)');\n } else {\n const linkLine = (label: string, shown: string[], total: number) => {\n if (total > 0) lines.push(`${label} (${total}): ${shown.join(', ')}${total > shown.length ? `, +${total - shown.length} more` : ''}`);\n };\n if (result.outboundTotal + result.unresolvedTotal + result.backlinksTotal > 0) lines.push('');\n linkLine('links out', result.outbound, result.outboundTotal);\n linkLine('unresolved', result.unresolved, result.unresolvedTotal);\n linkLine('backlinks', result.backlinks, result.backlinksTotal);\n }\n return lines.join('\\n');\n}\n"],"names":["OVERSIZE_BYTES","cell","value","String","replace","csvField","text","undefined","test","csvLine","values","map","join","warnOversize","bytes","rowCount","console","warn","Math","round","printRows","rows","format","columns","names","length","Object","keys","Set","rendered","row","name","log","Buffer","byteLength","renderRows","printRowStream","iterator","Symbol","first","next","rest","done","write","process","stdout","MAX_SAFE","BigInt","Number","MAX_SAFE_INTEGER","bigintSafe","_key","toString","indent","JSON","stringify","split","line","MIN_COLUMN","TRUNCATED","fitWidths","widths","limit","gaps","fitted","total","reduce","a","b","widest","indexOf","max","width","cells","col","natural","i","clip","w","padEnd","slice","formatRow","trimEnd","repeat","featureOffHint","featuresLine","features","off","on","presetsLines","presets","p","files","embedded","semantic","renderMap","result","parts","docs","count","fields","fieldsTotal","push","hubs","includes","recent","renderPeek","lines","path","tokens","parseError","key","entries","frontmatter","sections","s","level","heading","start_line","end_line","sectionsTotal","linkLine","label","shown","outboundTotal","unresolvedTotal","backlinksTotal","outbound","unresolved","backlinks"],"mappings":"AAAA,uCAAuC;AASvC,yFAAyF;AACzF,MAAMA,iBAAiB;AAEvB,mGAAmG;AACnG,SAASC,KAAKC,KAAc;IAC1B,OAAOC,OAAOD,kBAAAA,mBAAAA,QAAS,IAAIE,OAAO,CAAC,aAAa;AAClD;AAEA,yFAAyF;AACzF,yFAAyF;AACzF,uEAAuE;AACvE,SAASC,SAASH,KAAc;IAC9B,MAAMI,OAAOJ,UAAU,QAAQA,UAAUK,YAAY,KAAKJ,OAAOD;IACjE,OAAO,WAAWM,IAAI,CAACF,QAAQ,CAAC,CAAC,EAAEA,KAAKF,OAAO,CAAC,MAAM,MAAM,CAAC,CAAC,GAAGE;AACnE;AAEA,SAASG,QAAQC,MAAiB;IAChC,OAAOA,OAAOC,GAAG,CAACN,UAAUO,IAAI,CAAC;AACnC;AAEA,SAASC,aAAaC,KAAa,EAAEC,QAAgB;IACnD,IAAID,SAASd,gBAAgB;IAC7BgB,QAAQC,IAAI,CAAC,CAAC,mBAAmB,EAAEC,KAAKC,KAAK,CAACL,QAAQ,MAAM,WAAW,EAAEC,SAAS,iHAAiH,CAAC;AACtM;AAEA,sFAAsF;AACtF,+FAA+F;AAC/F,+FAA+F;AAC/F,+FAA+F;AAC/F,iEAAiE;AACjE,OAAO,SAASK,UAAUC,IAAW,EAAEC,MAAiB,EAAEC,OAAkB;IAC1E,IAAID,WAAW,OAAO;QACpB,qFAAqF;QACrF,yEAAyE;QACzE,MAAME,QAAQH,KAAKI,MAAM,GAAG,IAAIC,OAAOC,IAAI,CAACN,IAAI,CAAC,EAAE,IAAI;eAAI,IAAIO,IAAIL,oBAAAA,qBAAAA,UAAW,EAAE;SAAE;QAClF,IAAIC,MAAMC,MAAM,KAAK,GAAG;QACxB,MAAMI,WAAW;YAACpB,QAAQe;eAAWH,KAAKV,GAAG,CAAC,CAACmB,MAAQrB,QAAQe,MAAMb,GAAG,CAAC,CAACoB,OAASD,GAAG,CAACC,KAAK;SAAI,CAACnB,IAAI,CAAC;QACtGI,QAAQgB,GAAG,CAACH;QACZ,yDAAyD;QACzDhB,aAAaoB,OAAOC,UAAU,CAACL,WAAWR,KAAKI,MAAM;QACrD;IACF;IACA,MAAMI,WAAWM,WAAWd,MAAMC;IAClCN,QAAQgB,GAAG,CAACH;IACZhB,aAAaoB,OAAOC,UAAU,CAACL,WAAWR,KAAKI,MAAM;AACvD;AAEA,uFAAuF;AACvF,yFAAyF;AACzF,6FAA6F;AAC7F,+FAA+F;AAC/F,4FAA4F;AAC5F,0BAA0B;AAC1B,OAAO,SAASW,eAAef,IAAmB,EAAEC,MAAiB,EAAEC,OAAiB;IACtF,MAAMc,WAAWhB,IAAI,CAACiB,OAAOD,QAAQ,CAAC;IACtC,wFAAwF;IACxF,2FAA2F;IAC3F,MAAME,QAAQF,SAASG,IAAI;IAC3B,MAAMC,OAAsB;QAAE,CAACH,OAAOD,QAAQ,CAAC,EAAE,IAAMA;IAAS;IAEhE,IAAIf,WAAW,SAAS;QACtBF,UAAUmB,MAAMG,IAAI,GAAG,EAAE,GAAG;YAACH,MAAMrC,KAAK;eAAKuC;SAAK,EAAEnB,QAAQC;QAC5D;IACF;IAEA,IAAIT,QAAQ;IACZ,IAAIC,WAAW;IACf,MAAM4B,QAAQ,CAACrC;QACb,yDAAyD;QACzDQ,SAASmB,OAAOC,UAAU,CAAC5B;QAC3BsC,QAAQC,MAAM,CAACF,KAAK,CAACrC;IACvB;IAEA,IAAIgB,WAAW,QAAQ;QACrB,sFAAsF;QACtF,+DAA+D;QAC/D,wFAAwF;QACxF,4FAA4F;QAC5F,8EAA8E;QAC9E,MAAMwB,WAAWC,OAAOC,OAAOC,gBAAgB;QAC/C,MAAMC,aAAa,CAACC,MAAcjD,QAAoB,OAAOA,UAAU,WAAYA,SAAS,CAAC4C,YAAY5C,SAAS4C,WAAWE,OAAO9C,SAASA,MAAMkD,QAAQ,KAAMlD;QACjK,MAAMmD,SAAS,CAACvB,MACdwB,KAAKC,SAAS,CAACzB,KAAKoB,YAAY,GAC7BM,KAAK,CAAC,MACN7C,GAAG,CAAC,CAAC8C,OAAS,CAAC,EAAE,EAAEA,MAAM,EACzB7C,IAAI,CAAC;QACV,IAAI2B,MAAMG,IAAI,EAAE;YACdC,MAAM;QACR,OAAO;YACLA,MAAM;YACNA,MAAMU,OAAOd,MAAMrC,KAAK;YACxBa,WAAW;YACX,KAAK,MAAMe,OAAOW,KAAM;gBACtBE,MAAM;gBACNA,MAAMU,OAAOvB;gBACbf;YACF;YACA4B,MAAM;QACR;IACF,OAAO,IAAIJ,MAAMG,IAAI,EAAE;QACrB,MAAMlB,QAAQ;eAAI,IAAII,IAAIL;SAAS;QACnC,IAAIC,MAAMC,MAAM,GAAG,GAAGkB,MAAM,GAAGlC,QAAQe,OAAO,EAAE,CAAC;IACnD,OAAO;QACL,MAAMA,QAAQE,OAAOC,IAAI,CAACY,MAAMrC,KAAK;QACrCyC,MAAM,GAAGlC,QAAQe,OAAO,EAAE,CAAC;QAC3BmB,MAAM,GAAGlC,QAAQe,MAAMb,GAAG,CAAC,CAACoB,OAASQ,MAAMrC,KAAK,CAAC6B,KAAK,GAAG,EAAE,CAAC;QAC5DhB,WAAW;QACX,KAAK,MAAMe,OAAOW,KAAM;YACtBE,MAAM,GAAGlC,QAAQe,MAAMb,GAAG,CAAC,CAACoB,OAASD,GAAG,CAACC,KAAK,GAAG,EAAE,CAAC;YACpDhB;QACF;IACF;IACAF,aAAaC,OAAOC;AACtB;AAEA,6FAA6F;AAC7F,0EAA0E;AAC1E,MAAM2C,aAAa;AACnB,MAAMC,YAAY;AAElB,SAASC,UAAUC,MAAgB,EAAEC,KAAa;IAChD,MAAMC,OAAO,AAACF,CAAAA,OAAOpC,MAAM,GAAG,CAAA,IAAK;IACnC,MAAMuC,SAAS;WAAIH;KAAO;IAC1B,OAAS;QACP,MAAMI,QAAQD,OAAOE,MAAM,CAAC,CAACC,GAAGC,IAAMD,IAAIC,GAAG,KAAKL;QAClD,IAAIE,SAASH,OAAO,OAAOE;QAC3B,MAAMK,SAASL,OAAOM,OAAO,CAACpD,KAAKqD,GAAG,IAAIP;QAC1C,IAAIA,MAAM,CAACK,OAAO,IAAIX,YAAY,OAAOM;QACzCA,MAAM,CAACK,OAAO,GAAGnD,KAAKqD,GAAG,CAACb,YAAYM,MAAM,CAACK,OAAO,GAAGnD,KAAKqD,GAAG,CAAC,GAAGN,QAAQH;IAC7E;AACF;AAEA,SAAS3B,WAAWd,IAAW,EAAEC,MAAc,EAAEkD,QAAQ5B,QAAQC,MAAM,CAACtB,OAAO;IAC7E,IAAID,WAAW,QAAQ,OAAOgC,KAAKC,SAAS,CAAClC,MAAM,MAAM;IACzD,IAAIA,KAAKI,MAAM,KAAK,GAAG,OAAO;IAE9B,MAAMF,UAAUG,OAAOC,IAAI,CAACN,IAAI,CAAC,EAAE;IACnC,MAAMoD,QAAQpD,KAAKV,GAAG,CAAC,CAACmB,MAAQP,QAAQZ,GAAG,CAAC,CAAC+D,MAAQzE,KAAK6B,GAAG,CAAC4C,IAAI;IAClE,MAAMC,UAAUpD,QAAQZ,GAAG,CAAC,CAAC+D,KAAKE,IAAM1D,KAAKqD,GAAG,CAACG,IAAIjD,MAAM,KAAKgD,MAAM9D,GAAG,CAAC,CAACmB,MAAQA,GAAG,CAAC8C,EAAE,CAACnD,MAAM;IAChG,MAAMoC,SAASW,SAASA,QAAQd,aAAaE,UAAUe,SAASH,SAASG;IAEzE,MAAME,OAAO,CAAC3E,OAAe4E,IAAe5E,MAAMuB,MAAM,IAAIqD,IAAI5E,MAAM6E,MAAM,CAACD,KAAK,GAAG5E,MAAM8E,KAAK,CAAC,GAAGF,IAAI,KAAKnB,WAAW;IACxH,MAAMsB,YAAY,CAACvE,SACjBA,OACGC,GAAG,CAAC,CAACT,OAAO0E,IAAMC,KAAK3E,OAAO2D,MAAM,CAACe,EAAE,GACvChE,IAAI,CAAC,MACLsE,OAAO;IAEZ,OAAO;QAACD,UAAU1D;QAAUsC,OAAOlD,GAAG,CAAC,CAACmE,IAAM,IAAIK,MAAM,CAACL,IAAIlE,IAAI,CAAC;WAAU6D,MAAM9D,GAAG,CAACsE;KAAW,CAACrE,IAAI,CAAC;AACzG;AAEA,8EAA8E;AAE9E,2FAA2F;AAC3F,+CAA+C;AAC/C,OAAO,SAASwE,eAAerD,IAAY;IACzC,OAAO,CAAC,SAAS,EAAEA,MAAM;AAC3B;AAEA,OAAO,SAASsD,aAAaC,QAAyC;IACpE,MAAMC,MAAMD,SAASC,GAAG,CAAC9D,MAAM,GAAG,IAAI,CAAC,QAAQ,EAAE6D,SAASC,GAAG,CAAC5E,GAAG,CAAC,CAACoB,OAAS,GAAGA,KAAK,EAAE,EAAEqD,eAAerD,MAAM,CAAC,CAAC,EAAEnB,IAAI,CAAC,OAAO,GAAG;IAChI,OAAO,CAAC,UAAU,EAAE0E,SAASE,EAAE,CAAC5E,IAAI,CAAC,QAAQ2E,KAAK;AACpD;AAEA,OAAO,SAASE,aAAaC,OAAqF;IAChH,6FAA6F;IAC7F,iDAAiD;IACjD,OAAO;QAAC;WAAeA,QAAQ/E,GAAG,CAAC,CAACgF,IAAM,CAAC,EAAE,EAAEA,EAAE5D,IAAI,CAAC,EAAE,EAAE4D,EAAEC,KAAK,CAAC,UAAU,EAAED,EAAEE,QAAQ,CAAC,SAAS,EAAEF,EAAEG,QAAQ,KAAK,QAAQ,uBAAuB,IAAI;KAAE;AAC1J;AAEA,OAAO,SAASC,UAAUC,MAAoP;IAC5Q,MAAMC,QAAQ;QAAC,CAAC,MAAM,EAAED,OAAOE,IAAI,CAACC,KAAK,CAAC,EAAE,EAAEjF,KAAKC,KAAK,CAAC6E,OAAOE,IAAI,CAACpF,KAAK,GAAG,MAAM,IAAI,CAAC;QAAEuE,aAAaW,OAAOV,QAAQ;QAAG;WAAOG,aAAaO,OAAON,OAAO;QAAG;QAAIvD,WAAW6D,OAAOI,MAAM,EAAE;KAAS;IACrM,IAAIJ,OAAOK,WAAW,GAAGL,OAAOI,MAAM,CAAC3E,MAAM,EAAEwE,MAAMK,IAAI,CAAC,CAAC,EAAE,EAAEN,OAAOK,WAAW,GAAGL,OAAOI,MAAM,CAAC3E,MAAM,CAAC,aAAa,CAAC;IACvH,IAAIuE,OAAOO,IAAI,CAAC9E,MAAM,GAAG,GAAGwE,MAAMK,IAAI,CAAC,0BAA0BnE,WAAW6D,OAAOO,IAAI,EAAE;SACpF,IAAIP,OAAOV,QAAQ,CAACC,GAAG,CAACiB,QAAQ,CAAC,SAASP,MAAMK,IAAI,CAAC,CAAC,aAAa,EAAElB,eAAe,QAAQ,CAAC,CAAC;IACnGa,MAAMK,IAAI,CAAC,aAAanE,WAAW6D,OAAOS,MAAM,EAAE;IAClD,OAAOR,MAAMrF,IAAI,CAAC;AACpB;AAEA,OAAO,SAAS8F,WAAWV,MAAoR;IAC7S,MAAMW,QAAQ;QAAC,GAAGX,OAAOY,IAAI,CAAC,IAAI,EAAEZ,OAAOa,MAAM,CAAC,QAAQ,CAAC;KAAC;IAC5D,4FAA4F;IAC5F,iEAAiE;IACjE,IAAIb,OAAOc,UAAU,EAAEH,MAAML,IAAI,CAAC,CAAC,wDAAwD,EAAEN,OAAOc,UAAU,CAAC,CAAC,CAAC;IACjH,KAAK,MAAM,CAACC,KAAK7G,MAAM,IAAIwB,OAAOsF,OAAO,CAAChB,OAAOiB,WAAW,EAAGN,MAAML,IAAI,CAAC,CAAC,EAAE,EAAES,IAAI,EAAE,EAAE7G,OAAO;IAC9F,IAAI8F,OAAOkB,QAAQ,CAACzF,MAAM,GAAG,GAAG;QAC9BkF,MAAML,IAAI,CAAC,IAAI;QACf,KAAK,MAAMa,KAAKnB,OAAOkB,QAAQ,CAAEP,MAAML,IAAI,CAAC,CAAC,EAAE,EAAE,IAAInB,MAAM,CAACgC,EAAEC,KAAK,EAAY,CAAC,EAAED,EAAEE,OAAO,CAAC,IAAI,EAAEF,EAAEG,UAAU,CAAC,CAAC,EAAEH,EAAEI,QAAQ,CAAC,GAAG,EAAEJ,EAAEN,MAAM,CAAC,EAAE,CAAC;QAC9I,IAAIb,OAAOwB,aAAa,GAAGxB,OAAOkB,QAAQ,CAACzF,MAAM,EAAEkF,MAAML,IAAI,CAAC,CAAC,IAAI,EAAEN,OAAOwB,aAAa,GAAGxB,OAAOkB,QAAQ,CAACzF,MAAM,CAAC,iDAAiD,CAAC;IACvK,OAAO,IAAIuE,OAAOT,GAAG,CAACiB,QAAQ,CAAC,aAAaG,MAAML,IAAI,CAAC,IAAI;IAC3D,IAAIN,OAAOT,GAAG,CAACiB,QAAQ,CAAC,UAAU;QAChCG,MAAML,IAAI,CAAC,IAAI;IACjB,OAAO;QACL,MAAMmB,WAAW,CAACC,OAAeC,OAAiB1D;YAChD,IAAIA,QAAQ,GAAG0C,MAAML,IAAI,CAAC,GAAGoB,MAAM,EAAE,EAAEzD,MAAM,GAAG,EAAE0D,MAAM/G,IAAI,CAAC,QAAQqD,QAAQ0D,MAAMlG,MAAM,GAAG,CAAC,GAAG,EAAEwC,QAAQ0D,MAAMlG,MAAM,CAAC,KAAK,CAAC,GAAG,IAAI;QACtI;QACA,IAAIuE,OAAO4B,aAAa,GAAG5B,OAAO6B,eAAe,GAAG7B,OAAO8B,cAAc,GAAG,GAAGnB,MAAML,IAAI,CAAC;QAC1FmB,SAAS,aAAazB,OAAO+B,QAAQ,EAAE/B,OAAO4B,aAAa;QAC3DH,SAAS,cAAczB,OAAOgC,UAAU,EAAEhC,OAAO6B,eAAe;QAChEJ,SAAS,aAAazB,OAAOiC,SAAS,EAAEjC,OAAO8B,cAAc;IAC/D;IACA,OAAOnB,MAAM/F,IAAI,CAAC;AACpB"}
@@ -0,0 +1,2 @@
1
+ export declare function segmentField(text: string): string;
2
+ export declare function segmentMatch(terms: string): string;
@@ -0,0 +1,65 @@
1
+ // Word boundaries FTS5's tokenizers cannot find on their own.
2
+ //
3
+ // Segmentation is context-free: seg(query) appears inside seg(document) whenever the query is
4
+ // a substring of the document. That is the contract -- substring semantics, what `grep` and
5
+ // `LIKE '%..%'` give -- and it is what makes index and query agree on every input, always.
6
+ // Earlier attempts used ICU word segmentation (Intl.Segmenter), which is context-dependent:
7
+ // `东京都政府` splits as `东 | 京都 | 政府` while the query `东京` splits as one word, so the
8
+ // index and the query disagree about identical text. No ICU here. No dictionary.
9
+ // Scripts written without word spaces: a closed set of writing systems.
10
+ // Script_Extensions, not Script: the katakana long-vowel mark ー is Script=Common.
11
+ const SCRIPTS = '\\p{scx=Han}\\p{scx=Hiragana}\\p{scx=Katakana}\\p{scx=Thai}\\p{scx=Khmer}\\p{scx=Lao}\\p{scx=Myanmar}';
12
+ // A run is script BASE characters with their combining marks attached; a bare mark after a
13
+ // Latin letter (decomposed é) never starts one.
14
+ const RUN = new RegExp(`((?:[${SCRIPTS}]\\p{M}*)+)`, 'gu');
15
+ const HAS_RUN = new RegExp(`[${SCRIPTS}]`, 'u');
16
+ // Grapheme = base char plus its marks. Context-free: index and query can never disagree.
17
+ const GRAPHEME = /(\P{M}\p{M}*)/gu;
18
+ // Token barrier between separate runs so their graphemes are never phrase-adjacent.
19
+ // U+A7F7: a letter (so FTS5 keeps it as a token) that no one types into a search.
20
+ const BARRIER = 'ꟷ';
21
+ function graphemes(run) {
22
+ var _run_match;
23
+ return (_run_match = run.match(GRAPHEME)) !== null && _run_match !== void 0 ? _run_match : [];
24
+ }
25
+ // Index side: '' when the field has no unspaced-script run (the common case, paid for by
26
+ // nothing). Otherwise each run explodes into its graphemes, barrier-delimited from its
27
+ // neighbors so graphemes from separate runs are never phrase-adjacent, and whitespace collapses.
28
+ export function segmentField(text) {
29
+ if (!HAS_RUN.test(text)) return '';
30
+ const out = text.replace(RUN, (run)=>` ${BARRIER} ${graphemes(run).join(' ')} ${BARRIER} `);
31
+ return out.replace(/\s+/g, ' ').trim();
32
+ }
33
+ // A `title:`/`summary:`/`text:` qualifier directly before a run that is about to become a
34
+ // quoted grapheme phrase, so the rewrite can retarget it at the matching `_seg` column.
35
+ const QUALIFIER = /(^|[\s(])(-?)(title|summary|text)\s*:\s*$/;
36
+ // Query side: each unspaced run becomes a quoted phrase of its graphemes, matching how
37
+ // segmentField indexed it. A qualifier ahead of such a run maps to its `_seg` column, since
38
+ // segmented text lives only there. An author's own quoted phrase, and everything else, pass
39
+ // through byte-identical.
40
+ export function segmentMatch(terms) {
41
+ if (!HAS_RUN.test(terms)) return terms;
42
+ let out = '';
43
+ let quoted = false;
44
+ const pieces = terms.split(RUN); // split keeps captured runs at odd indices
45
+ for(let i = 0; i < pieces.length; i++){
46
+ if (i % 2 === 0) {
47
+ for (const ch of pieces[i])if (ch === '"') quoted = !quoted;
48
+ out += pieces[i];
49
+ continue;
50
+ }
51
+ if (quoted) {
52
+ out += pieces[i]; // an author's phrase is matched as written
53
+ continue;
54
+ }
55
+ const g = graphemes(pieces[i]);
56
+ const phrase = g.length > 1 ? `"${g.join(' ')}"` : g[0];
57
+ const m = out.match(QUALIFIER);
58
+ if (m) {
59
+ out = `${out.slice(0, m.index)}${m[1]}${m[2]}${m[3]}_seg:${phrase}`;
60
+ } else {
61
+ out += phrase;
62
+ }
63
+ }
64
+ return out;
65
+ }
@@ -0,0 +1 @@
1
+ {"version":3,"sources":["/Users/kevin/Dev/OpenSource/ai/sensemaking/src/segment.ts"],"sourcesContent":["// Word boundaries FTS5's tokenizers cannot find on their own.\n//\n// Segmentation is context-free: seg(query) appears inside seg(document) whenever the query is\n// a substring of the document. That is the contract -- substring semantics, what `grep` and\n// `LIKE '%..%'` give -- and it is what makes index and query agree on every input, always.\n// Earlier attempts used ICU word segmentation (Intl.Segmenter), which is context-dependent:\n// `东京都政府` splits as `东 | 京都 | 政府` while the query `东京` splits as one word, so the\n// index and the query disagree about identical text. No ICU here. No dictionary.\n\n// Scripts written without word spaces: a closed set of writing systems.\n// Script_Extensions, not Script: the katakana long-vowel mark ー is Script=Common.\nconst SCRIPTS = '\\\\p{scx=Han}\\\\p{scx=Hiragana}\\\\p{scx=Katakana}\\\\p{scx=Thai}\\\\p{scx=Khmer}\\\\p{scx=Lao}\\\\p{scx=Myanmar}';\n// A run is script BASE characters with their combining marks attached; a bare mark after a\n// Latin letter (decomposed é) never starts one.\nconst RUN = new RegExp(`((?:[${SCRIPTS}]\\\\p{M}*)+)`, 'gu');\nconst HAS_RUN = new RegExp(`[${SCRIPTS}]`, 'u');\n// Grapheme = base char plus its marks. Context-free: index and query can never disagree.\nconst GRAPHEME = /(\\P{M}\\p{M}*)/gu;\n// Token barrier between separate runs so their graphemes are never phrase-adjacent.\n// U+A7F7: a letter (so FTS5 keeps it as a token) that no one types into a search.\nconst BARRIER = 'ꟷ';\n\nfunction graphemes(run: string): string[] {\n return run.match(GRAPHEME) ?? [];\n}\n\n// Index side: '' when the field has no unspaced-script run (the common case, paid for by\n// nothing). Otherwise each run explodes into its graphemes, barrier-delimited from its\n// neighbors so graphemes from separate runs are never phrase-adjacent, and whitespace collapses.\nexport function segmentField(text: string): string {\n if (!HAS_RUN.test(text)) return '';\n const out = text.replace(RUN, (run) => ` ${BARRIER} ${graphemes(run).join(' ')} ${BARRIER} `);\n return out.replace(/\\s+/g, ' ').trim();\n}\n\n// A `title:`/`summary:`/`text:` qualifier directly before a run that is about to become a\n// quoted grapheme phrase, so the rewrite can retarget it at the matching `_seg` column.\nconst QUALIFIER = /(^|[\\s(])(-?)(title|summary|text)\\s*:\\s*$/;\n\n// Query side: each unspaced run becomes a quoted phrase of its graphemes, matching how\n// segmentField indexed it. A qualifier ahead of such a run maps to its `_seg` column, since\n// segmented text lives only there. An author's own quoted phrase, and everything else, pass\n// through byte-identical.\nexport function segmentMatch(terms: string): string {\n if (!HAS_RUN.test(terms)) return terms;\n let out = '';\n let quoted = false;\n const pieces = terms.split(RUN); // split keeps captured runs at odd indices\n for (let i = 0; i < pieces.length; i++) {\n if (i % 2 === 0) {\n for (const ch of pieces[i]) if (ch === '\"') quoted = !quoted;\n out += pieces[i];\n continue;\n }\n if (quoted) {\n out += pieces[i]; // an author's phrase is matched as written\n continue;\n }\n const g = graphemes(pieces[i]);\n const phrase = g.length > 1 ? `\"${g.join(' ')}\"` : g[0];\n const m = out.match(QUALIFIER);\n if (m) {\n out = `${out.slice(0, m.index)}${m[1]}${m[2]}${m[3]}_seg:${phrase}`;\n } else {\n out += phrase;\n }\n }\n return out;\n}\n"],"names":["SCRIPTS","RUN","RegExp","HAS_RUN","GRAPHEME","BARRIER","graphemes","run","match","segmentField","text","test","out","replace","join","trim","QUALIFIER","segmentMatch","terms","quoted","pieces","split","i","length","ch","g","phrase","m","slice","index"],"mappings":"AAAA,8DAA8D;AAC9D,EAAE;AACF,8FAA8F;AAC9F,4FAA4F;AAC5F,2FAA2F;AAC3F,4FAA4F;AAC5F,kFAAkF;AAClF,iFAAiF;AAEjF,wEAAwE;AACxE,kFAAkF;AAClF,MAAMA,UAAU;AAChB,2FAA2F;AAC3F,gDAAgD;AAChD,MAAMC,MAAM,IAAIC,OAAO,CAAC,KAAK,EAAEF,QAAQ,WAAW,CAAC,EAAE;AACrD,MAAMG,UAAU,IAAID,OAAO,CAAC,CAAC,EAAEF,QAAQ,CAAC,CAAC,EAAE;AAC3C,yFAAyF;AACzF,MAAMI,WAAW;AACjB,oFAAoF;AACpF,kFAAkF;AAClF,MAAMC,UAAU;AAEhB,SAASC,UAAUC,GAAW;QACrBA;IAAP,QAAOA,aAAAA,IAAIC,KAAK,CAACJ,uBAAVG,wBAAAA,aAAuB,EAAE;AAClC;AAEA,yFAAyF;AACzF,uFAAuF;AACvF,iGAAiG;AACjG,OAAO,SAASE,aAAaC,IAAY;IACvC,IAAI,CAACP,QAAQQ,IAAI,CAACD,OAAO,OAAO;IAChC,MAAME,MAAMF,KAAKG,OAAO,CAACZ,KAAK,CAACM,MAAQ,CAAC,CAAC,EAAEF,QAAQ,CAAC,EAAEC,UAAUC,KAAKO,IAAI,CAAC,KAAK,CAAC,EAAET,QAAQ,CAAC,CAAC;IAC5F,OAAOO,IAAIC,OAAO,CAAC,QAAQ,KAAKE,IAAI;AACtC;AAEA,0FAA0F;AAC1F,wFAAwF;AACxF,MAAMC,YAAY;AAElB,uFAAuF;AACvF,4FAA4F;AAC5F,4FAA4F;AAC5F,0BAA0B;AAC1B,OAAO,SAASC,aAAaC,KAAa;IACxC,IAAI,CAACf,QAAQQ,IAAI,CAACO,QAAQ,OAAOA;IACjC,IAAIN,MAAM;IACV,IAAIO,SAAS;IACb,MAAMC,SAASF,MAAMG,KAAK,CAACpB,MAAM,2CAA2C;IAC5E,IAAK,IAAIqB,IAAI,GAAGA,IAAIF,OAAOG,MAAM,EAAED,IAAK;QACtC,IAAIA,IAAI,MAAM,GAAG;YACf,KAAK,MAAME,MAAMJ,MAAM,CAACE,EAAE,CAAE,IAAIE,OAAO,KAAKL,SAAS,CAACA;YACtDP,OAAOQ,MAAM,CAACE,EAAE;YAChB;QACF;QACA,IAAIH,QAAQ;YACVP,OAAOQ,MAAM,CAACE,EAAE,EAAE,2CAA2C;YAC7D;QACF;QACA,MAAMG,IAAInB,UAAUc,MAAM,CAACE,EAAE;QAC7B,MAAMI,SAASD,EAAEF,MAAM,GAAG,IAAI,CAAC,CAAC,EAAEE,EAAEX,IAAI,CAAC,KAAK,CAAC,CAAC,GAAGW,CAAC,CAAC,EAAE;QACvD,MAAME,IAAIf,IAAIJ,KAAK,CAACQ;QACpB,IAAIW,GAAG;YACLf,MAAM,GAAGA,IAAIgB,KAAK,CAAC,GAAGD,EAAEE,KAAK,IAAIF,CAAC,CAAC,EAAE,GAAGA,CAAC,CAAC,EAAE,GAAGA,CAAC,CAAC,EAAE,CAAC,KAAK,EAAED,QAAQ;QACrE,OAAO;YACLd,OAAOc;QACT;IACF;IACA,OAAOd;AACT"}
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "sensemaking",
3
- "version": "0.11.4",
3
+ "version": "0.12.0",
4
4
  "description": "Query and search your markdown notes with context-aware progressive disclosure: SQL over frontmatter, links, and text, plus semantic search and link-graph ranking. No server, no build step",
5
5
  "keywords": [
6
6
  "markdown",
@@ -8,8 +8,11 @@
8
8
  "sql",
9
9
  "sqlite",
10
10
  "fts5",
11
+ "full-text-search",
11
12
  "query",
12
13
  "search",
14
+ "multilingual",
15
+ "cjk",
13
16
  "semantic-search",
14
17
  "hybrid-search",
15
18
  "knowledge-base",
package/schema.json CHANGED
@@ -83,9 +83,20 @@
83
83
  "key": { "type": "string" }
84
84
  }
85
85
  },
86
+ "content": {
87
+ "type": "object",
88
+ "additionalProperties": false,
89
+ "description": "Settings for the `content` FTS5 table. Changing anything here rebuilds the content table: FTS5 cannot alter a tokenizer in place, and the prose is stored nowhere else, so every file's text is re-read. Vectors, links, and sections are file-derived and tokenizer-independent, so they are kept. The rebuild announces itself on stderr, naming the content tokenizer as the cause.",
90
+ "properties": {
91
+ "tokenize": {
92
+ "type": "string",
93
+ "description": "FTS5 tokenizer for word search, defaulting to `porter unicode61`. Languages written without word spaces (Chinese, Japanese, Thai, Khmer, Lao, Burmese) need no decision here: their text is indexed per grapheme and searched as an ordered grapheme phrase against machine-written sidecar columns, which is substring semantics -- what grep gives, a query matches wherever its exact text occurs, including inside a longer run, and needs no minimum length. What this setting is for is substring matching inside a Latin word, `trigram`, which indexes every three-character window, at the cost of stemming (`running` stops matching `run`), weaker bm25 relevance because trigrams are not words, a larger index, and a floor of three characters. `unicode61 tokenchars '-_'` is the smaller lever, keeping hyphenated terms whole. Naming any tokenizer turns the grapheme-phrase scheme off, on the reasoning that the tree has chosen its own scheme. Built-in choices are unicode61, ascii, porter, and trigram, each with their own options; the value is validated by building a throwaway table with it, so whatever the linked SQLite accepts is accepted here and anything else errors carrying SQLite's own message."
94
+ }
95
+ }
96
+ },
86
97
  "queries": {
87
98
  "type": "object",
88
- "description": "Saved queries runnable as `sense <name> [params...]`, each naming the verb it runs, one to one with the two commands: `{ sql }` runs like `sense sql`, `{ search }` like `sense search`. A bare string is rejected -- it silently meant SQL. `{ sql }`: `?` placeholders bind to CLI positional args in order; deterministic and enumerating, including raw FTS5 `MATCH` for word search under your own SQL -- \"0 rows = not in the tree\" lives here. Tables: `frontmatter` (one row per file, one column per discovered frontmatter key, plus `path`/`_mtime`/`_size`/`_rank`/`_parse_error`, the last being NULL when the frontmatter parsed and the YAML message when it did not, in which case no other column is populated), `content` (FTS5: `title`, `summary`, `text`, `path`), `links` (`src`, `target`, `dst`), `sections` (`path`, `idx`, `level`, `heading`, `start_line`, `end_line`, `tokens`), and `preset_files` (`preset`, `path`) -- which presets cover which files. `has(field, value)`: array membership on a JSON-array field, substring match on a string (so has(f.status, 'active') also matches 'inactive'), false on NULL. Exact matches: `=` for scalars, `EXISTS (SELECT 1 FROM json_each(f.tags) WHERE value = ?)` for array members. Canonical query: `SELECT f.path, content.title, content.summary, CASE WHEN length(content.text) <= 16384 THEN snippet(content, -1, '«', '»', '…', 10) END AS hit FROM frontmatter f JOIN content ON content.path = f.path WHERE content MATCH ? ORDER BY bm25(content, 10.0, 5.0, 1.0) LIMIT 10`. The CASE bounds snippet(), which re-tokenizes each matched document and costs seconds per query once a tree holds a megabyte-scale note; `search` applies the same bound internally. Saved search object: one text driving every engine the scoped preset has -- word match, links, vectors -- fused into one ranked list, `via` labeling which engine produced each row; `search` text must be non-empty (a saved query saves a question -- a scope without one is just flags). `preset` names one declared preset (defaults to `default`); `include` is an ad hoc glob scope that replaces the preset's include/exclude entirely, same as `search --include`; `where` and `k` behave like the `search` command's flags. `sense <name>` behaves like `sense search <search> [--preset] [--include] [--where] [--k]` with zero flags; it takes no positional parameters. Running an entry validates it: a typo'd column or unknown preset errors and exits nonzero, and a parameterised entry validates with any argument, since SQL is prepared before parameters bind. Nothing asserts on the result itself; a returned row set is the reader's judgment. Reserved frontmatter keys: `path`, `_mtime`, `_size`, `_rank`, `_parse_error`, `content`, `links`, `sections`. Reserved names (unreachable as saved queries, any shape): `init`, `sql`, `search`, `map`, `peek`, `path`, `related`, `download`, `watch`, `status`.",
99
+ "description": "Saved queries runnable as `sense <name> [params...]`, each naming the verb it runs, one to one with the two commands: `{ sql }` runs like `sense sql`, `{ search }` like `sense search`. A bare string is rejected -- it silently meant SQL. `{ sql }`: `?` placeholders bind to CLI positional args in order; deterministic and enumerating, including raw FTS5 `MATCH` for word search under your own SQL -- \"0 rows = not in the tree\" lives here. Tables: `frontmatter` (one row per file, one column per discovered frontmatter key, plus `path`/`_mtime`/`_size`/`_rank`/`_parse_error`, the last being NULL when the frontmatter parsed and the YAML message when it did not, in which case no other column is populated), `content` (FTS5: `title`, `summary`, `text`, `path`, plus machine-written `title_seg`/`summary_seg`/`text_seg` sidecars holding the grapheme phrases for Chinese, Japanese, Thai, Khmer, Lao, and Burmese text -- present for matching, not reading; a hand-written `MATCH` reaches them through `segment()`), `links` (`src`, `target`, `dst`), `sections` (`path`, `idx`, `level`, `heading`, `start_line`, `end_line`, `tokens`), and `preset_files` (`preset`, `path`) -- which presets cover which files. `has(field, value)`: array membership on a JSON-array field, substring match on a string (so has(f.status, 'active') also matches 'inactive'), false on NULL. `segment(terms)`: rewrites a run of unspaced-script text in `terms` into the ordered grapheme phrase the sidecar columns need; text with no such run passes through unchanged, so it is safe to add to any query. Exact matches: `=` for scalars, `EXISTS (SELECT 1 FROM json_each(f.tags) WHERE value = ?)` for array members. Canonical query: `SELECT f.path, content.title, content.summary, CASE WHEN length(content.text) <= 16384 THEN snippet(content, 2, '«', '»', '…', 10) END AS hit FROM frontmatter f JOIN content ON content.path = f.path WHERE content MATCH ? ORDER BY bm25(content, 10.0, 5.0, 1.0, 0, 10.0, 5.0, 1.0) LIMIT 10`. snippet() names column 2 (`text`) explicitly rather than -1 (best column), since -1 could surface a sidecar as the excerpt. bm25's full form repeats the three weights onto the sidecars so a title hit reached through `title_seg` ranks like one reached through `title`; the plain `bm25(content, 10.0, 5.0, 1.0)` still runs (FTS5 defaults unnamed columns to 1.0) but ranks a sidecar match at body weight. The CASE bounds snippet(), which re-tokenizes each matched document and costs seconds per query once a tree holds a megabyte-scale note; `search` applies the same bound internally. Saved search object: one text driving every engine the scoped preset has -- word match, links, vectors -- fused into one ranked list, `via` labeling which engine produced each row; `search` text must be non-empty (a saved query saves a question -- a scope without one is just flags). `preset` names one declared preset (defaults to `default`); `include` is an ad hoc glob scope that replaces the preset's include/exclude entirely, same as `search --include`; `where` and `k` behave like the `search` command's flags. `sense <name>` behaves like `sense search <search> [--preset] [--include] [--where] [--k]` with zero flags; it takes no positional parameters. Running an entry validates it: a typo'd column or unknown preset errors and exits nonzero, and a parameterised entry validates with any argument, since SQL is prepared before parameters bind. Nothing asserts on the result itself; a returned row set is the reader's judgment. Reserved frontmatter keys: `path`, `_mtime`, `_size`, `_rank`, `_parse_error`, `content`, `links`, `sections`. Reserved names (unreachable as saved queries, any shape): `init`, `sql`, `search`, `map`, `peek`, `path`, `related`, `download`, `watch`, `status`.",
89
100
  "additionalProperties": {
90
101
  "oneOf": [
91
102
  {
@@ -5,7 +5,7 @@ description: "Query a markdown tree with the sense CLI: filter notes by frontmat
5
5
 
6
6
  # sense
7
7
 
8
- SQL over a markdown tree, kept fresh by a filesystem check on every query. Every file becomes rows in `frontmatter` (one column per key, plus `path`/`_mtime`/`_size`/`_rank`/`_parse_error`), `content` (FTS5: `title`, `summary`, `text`), `links` (`src`, `target`, `dst`; `NULL` dst = dead link), `sections` (heading outline with line ranges and token estimates), and `preset_files` (`path`, `preset`: which presets cover which files). Features add their own storage; `map` and `status` report which are on.
8
+ SQL over a markdown tree, kept fresh by a filesystem check on every query. Every file becomes rows in `frontmatter` (one column per key, plus `path`/`_mtime`/`_size`/`_rank`/`_parse_error`), `content` (FTS5: `title`, `summary`, `text`, `path`, plus machine-written `title_seg`/`summary_seg`/`text_seg` sidecars used for matching Chinese, Japanese, Thai, Khmer, Lao, and Burmese text, not for reading), `links` (`src`, `target`, `dst`; `NULL` dst = dead link), `sections` (heading outline with line ranges and token estimates), and `preset_files` (`path`, `preset`: which presets cover which files). Features add their own storage; `map` and `status` report which are on.
9
9
 
10
10
  ## What each tool is for
11
11
 
@@ -19,7 +19,7 @@ Every result is a reference (path, metadata, excerpt), never file contents; pros
19
19
  - `related <note>` ranks notes near in meaning to one note that it does not already link to: the links it is missing. It reads the meaning-vectors, so it needs vectors on for the scope and scans them, costing about what a semantic `search` does, not what a `peek` does. Vectors need the model on disk; nothing fetches it implicitly, so `sense download` is a one-time step per machine. Without it `search` still answers on words and links (it prints a note saying vectors are off), while `related` reports the missing model instead of an empty list.
20
20
  - When you know the file and need its contents, `Read` it. sense adds nothing there. On large files peek's ranges let you read just one section; small files are often cheaper whole.
21
21
 
22
- Output defaults to a table, built for humans; `--format json` returns the same rows machine-parseable. That also makes a saved query usable as a CI/hook gate with zero added mechanism: `[ "$(sense <name> --format json)" = "[]" ]` is true exactly when it returned no rows.
22
+ Output defaults to a table, built for humans; `--format json` returns the same rows machine-parseable, and `--format csv` returns one row per line for `grep` and `awk`. csv keeps every character a value holds, embedded newlines included, but cannot express NULL versus an empty string or a value's type, which json can. The commands that emit rows (`sql`, `search`, `related`, saved queries) take all three; `map`, `peek`, `status`, and `path` render a structure rather than a row set and take table or json. That also makes a saved query usable as a CI/hook gate with zero added mechanism: `[ "$(sense <name> --format json)" = "[]" ]` is true exactly when it returned no rows.
23
23
 
24
24
  ## Commands
25
25
 
@@ -91,15 +91,16 @@ WHERE a.dst = ? AND b.dst IS NOT NULL AND b.dst <> a.dst;
91
91
  - A saved `{ sql }` written against `scope` is preset-agnostic: `sense <name> --preset raw` re-points the same statement at another layer, so one entry serves every preset instead of one copy each.
92
92
  - `content MATCH` only works against the fts5 table by its own name, never through an alias or a view: `FROM content c ... WHERE c MATCH 'x'` fails with `no such column: c`. This is why `--preset` binds a table to join rather than shadowing the tables.
93
93
  - `content MATCH` takes FTS5 syntax: `a OR b`, `"phrase"`, `pref*`, `NEAR(a b, 5)`, `summary: term`. Stemmed; markdown stripped at index time. Double-quote any term with punctuation. Bare `customer-facing` errors (`-` reads as a column filter), bare apostrophes are syntax errors: write `"customer-facing"`, `"founder's"`.
94
- - Rank with `ORDER BY bm25(content, 10.0, 5.0, 1.0)` (title > summary > body); excerpt with `snippet(content, -1, '«', '»', '…', 10)`. snippet() re-tokenizes each matched doc and its cost grows superlinearly with doc size, measured ~10 s per query on a tree holding one 1 MB note. `search` bounds this itself (docs past 16 KB get an equivalent excerpt another way); in hand-written SQL, guard it: `CASE WHEN length(text) <= 16384 THEN snippet(...) END`, or select `title`/`summary` instead of an excerpt.
94
+ - A language written without word spaces (Chinese, Japanese, Thai, Khmer, Lao, Burmese) is indexed per grapheme and searched as an ordered grapheme phrase against the `_seg` sidecar columns: substring semantics, what `grep` gives -- a query matches wherever its exact text occurs, including inside a longer run (`京都` matches `东京都政府`, correctly, because it's there at position 2), and needs no minimum length. No decision is needed for these languages. Hand-written SQL is not rewritten for you, so a raw `content MATCH '数据库'` finds nothing: write `content MATCH segment(?)` and bind the terms. `segment()` returns text with no such run unchanged, so it is safe to leave in a query whatever the tree's language.
95
+ - Rank with `ORDER BY bm25(content, 10.0, 5.0, 1.0)` (title > summary > body); the full form `bm25(content, 10.0, 5.0, 1.0, 0, 10.0, 5.0, 1.0)` mirrors the same weights onto the `_seg` sidecars, so a title hit found through `title_seg` ranks like one found through `title` (the three-weight form still runs -- FTS5 defaults unnamed columns to 1.0 -- but ranks a sidecar match at body weight). Excerpt with `snippet(content, 2, '«', '»', '…', 10)`, naming the `text` column explicitly: `-1` means best column, which can surface a machine-spaced sidecar as the excerpt (`search` itself always names column 2). snippet() re-tokenizes each matched doc and its cost grows superlinearly with doc size, measured ~10 s per query on a tree holding one 1 MB note. `search` bounds this itself (docs past 16 KB get an equivalent excerpt another way); in hand-written SQL, guard it: `CASE WHEN length(text) <= 16384 THEN snippet(...) END`, or select `title`/`summary` instead of an excerpt.
95
96
  - Select `content.title`/`content.summary` (always exist, empty when absent) rather than `f.title`/`f.summary` (discovered columns; error on trees that never declare them).
96
97
  - Frontmatter values keep their YAML type: strings are TEXT, whole numbers and booleans are INTEGER (`true` stores as 1, so `WHERE flag = 1` matches and `WHERE flag = 'true'` matches nothing), fractions are REAL, lists and maps are JSON text. `map` prints the observed type per field, and a field showing two types (`integer,text`) has drifted across notes.
97
98
  - **Dead links need the attachment filter.** `dst IS NULL` alone is not "broken link": a wikilink to anything that is not markdown (`[[Board.base]]`, `![[Pasted image.png]]`, `[[spec.pdf]]`) can never resolve, because sense indexes markdown and resolution only tries the exact path or `+.md`. Those are out of the index's universe, not broken. On a 1,400-note Obsidian vault the unfiltered query returns 143 rows where 14 are real. Exclude anything carrying a file extension, as in the recipe above, and widen the exclusion if your notes have dotted titles (`[[Node.js]]` carries one too, so a stricter list -- `'*.png'`, `'*.pdf'`, `'*.base'`, and whatever else your vault attaches -- is safer on a tree whose titles use dots). Scope it with `preset_files` as well: template and skill files are full of `[[Note Name]]` examples that are deliberately unresolved.
98
99
  - `has(field, value)`: array membership on JSON-array fields, substring on strings, false on NULL. This is the `includes()` convention. Substring means `has(f.status, 'active')` also matches `inactive`; exact scalar match is `f.status = ?`, deliberate substring is `LIKE`, exact array membership is `EXISTS (SELECT 1 FROM json_each(f.tags) WHERE value = ?)`. To aggregate per member instead, use `json_each(frontmatter.<field>)` (above) -- GROUP BY on the raw column splits `["a","b"]` and `["b","a"]` into separate buckets.
99
- - Date fields are stored as written. Compare through `datetime()`, which normalizes ISO 8601 timezone offsets to UTC: `WHERE datetime(created) >= datetime(?)`. Bare string comparison is only safe when every note uses the same offset.
100
+ - Compare dates through `datetime()`, which resolves ISO 8601 offsets to UTC: `WHERE datetime(created) >= datetime(?)`. Bare string comparison is only safe when every note uses the same offset.
100
101
  - Date spellings SQLite rejects (`-0800`, `-08`, a space separator) are normalized at index time, offset preserved. One it cannot fix is left as written and warned about by path: `datetime()` returns NULL there, so the row is invisible to a date comparison rather than excluded by it. List them with `WHERE d IS NOT NULL AND datetime(d) IS NULL`.
101
102
  - **SQLite's `now` is UTC, so any query about "today" needs `'localtime'`.** `date('now')` reads as tomorrow from mid-afternoon onward in the Americas, which silently flips "scheduled today" into "overdue" every evening: write `date('now','localtime')` and `datetime('now','start of day','localtime')`. This only matters where the boundary carries the meaning; a `'-90 day'` window is unaffected by a few hours of skew.
102
- - To bound what a query puts into context: `snippet()` excerpts just the matching text, `LIMIT` caps row counts, and selecting `path`/`title`/`summary` keeps rows small. `SELECT text FROM content` returns the tree's entire prose (sense warns past 50 KB). Aggregates (`COUNT`, `GROUP BY`) are already bounded. `SELECT * FROM frontmatter` is always safe: prose is not a frontmatter column.
103
+ - To bound what a query puts into context: `snippet()` excerpts just the matching text, `LIMIT` caps row counts, and selecting `path`/`title`/`summary` keeps rows small. A large result can also stay out of context entirely: `--format csv > file` writes it whole, and `grep`/`awk` over that file returns only the rows wanted. `SELECT text FROM content` returns the tree's entire prose (sense warns past 50 KB, after the rows have already printed, so the warning records the cost rather than preventing it). Aggregates (`COUNT`, `GROUP BY`) are already bounded. `SELECT * FROM frontmatter` is always safe: prose is not a frontmatter column.
103
104
 
104
105
  Worked traces: [EXAMPLES.md](EXAMPLES.md).
105
106
 
@@ -108,7 +109,7 @@ Worked traces: [EXAMPLES.md](EXAMPLES.md).
108
109
  - Missing CLI: `npm install -g sensemaking`. Missing config: `sense init` at the tree root. Discovery walks up from cwd; `--config <path>` overrides. Setting up or restructuring a tree (presets, frontmatter conventions, note design) is the `sense-setup` skill.
109
110
  - `map` and `status` report each preset's coverage (files matched, embedded count). Indexing derives from presets, so the coverage numbers are how you see what a config actually indexes and embeds. A scope with fewer signals just uses fewer (a semantic-off preset searches lexically); a saved search naming an unknown preset errors when run, listing the declared ones.
110
111
  - Save a query into `sense.config.json` only when it will be reused; run ad-hoc otherwise.
111
- - A one-line `summary:` per note is optional and pays twice: it appears in result rows and is a weighted search field. Date comparisons work for dates written as ISO 8601 (`2026-08-12`, or with time and offset), the only format `datetime()` parses. Field names in examples (`status`, `tags`, `created`) are illustrative; your tree defines its own.
112
+ - A one-line `summary:` per note is optional and pays twice: it appears in result rows and is a weighted search field. Date comparisons work for dates written as ISO 8601 (`2026-08-12`, or with time and offset); other formats do not compare. Field names in examples (`status`, `tags`, `created`) are illustrative; your tree defines its own.
112
113
  - Reserved frontmatter keys (dropped with a warning): `path`, `_mtime`, `_size`, `_rank`, `_parse_error`, `content`, `links`, `sections`.
113
114
  - A note whose frontmatter does not parse is indexed with **no** frontmatter columns and `_parse_error` set to the YAML message, which carries the line. Nothing is half-recovered: a non-NULL value is a value the author wrote. So a NULL column means the key was absent *or* the note did not parse, and `_parse_error` is how you tell: `WHERE status IS NULL AND _parse_error IS NULL` is "genuinely missing status". List what needs fixing with `sense sql "SELECT path, _parse_error FROM frontmatter WHERE _parse_error IS NOT NULL"`; fixing a file clears it on the next command. `sense status` reports the count.
114
115
  - Exit codes: `0` ok, `1` error (SQLite message verbatim), `2` usage (unknown query, wrong param count).
@@ -42,6 +42,7 @@ These belong to the tree's owner. sense works with any of them and reads no inst
42
42
  - **Presets are path-shaped; frontmatter is state-shaped.** A preset's coverage must be computable from the path alone (it decides indexing, baked into the cache). Volatile state (`status`, `project`, dates) lives in frontmatter and filters at query time (`where`, `has()`, `datetime()`). A state worth different *indexing* (retired memory, superseded sources) is a state worth moving the file: the archive-folder pattern in EXAMPLES.md.
43
43
  - **What a note omits is also a filter.** A layer that deliberately carries none of the fields the saved views filter on is excluded from all of them without any view naming the layer. Sparse fields cut both ways: less of the tree filters when you want breadth, and exactly this separation when layers differ in authority.
44
44
  - **Dates.** `datetime()` comparisons work for dates written as ISO 8601, the only format it parses. A tree that mixes date formats can store them, but can't compare them in SQL.
45
+ - **Language.** No decision needed. A language written without word spaces (Chinese, Japanese, Thai, Khmer, Lao, Burmese) is indexed per grapheme and searched as an ordered grapheme phrase, so `sense search "全文"` finds what it should. This is substring semantics, what `grep` gives: a query matches wherever its exact text occurs, including inside a longer run, and needs no minimum length. A language written with spaces is left exactly as it was, storing nothing extra, and its stemming is unaffected. The one place this does not reach is a hand-written `content MATCH '...'`, which cannot be rewritten for its author: pass the terms through `segment()` there. `content.tokenize` is a separate lever with a different purpose, substring matching inside a Latin word (`trigram`) or keeping hyphenated terms whole (`unicode61 tokenchars '-_'`); naming one turns the grapheme-phrase scheme off, since the tree has then chosen its own scheme. Changing it rebuilds the text index only; vectors, links, and sections are kept.
45
46
  - **Summaries.** A one-line `summary:` is optional and pays twice: it shows in every result row (often answering a question with no file read) and is a weighted search field ranked above body text. The cost is writing and maintaining the line as notes change.
46
47
  - **Folder shape.** Globs find the files, paths are queryable text, links resolve by basename at any depth, but presets make folders meaningful: a folder is the natural unit that gets its own coverage and settings.
47
48
  - **Note size.** Many small notes: precise search hits, whole-file reads stay cheap, more links to maintain. Fewer large notes: `sections`, `peek`, and the `lines` column carry the cost down to line-range reads. Both work.