sensemaking 0.21.0 → 0.22.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -1
- package/dist/cjs/chunk/group.d.cts +0 -2
- package/dist/cjs/chunk/group.d.ts +0 -2
- package/dist/cjs/chunk/group.js +25 -62
- package/dist/cjs/chunk/group.js.map +1 -1
- package/dist/cjs/chunk/index.d.cts +1 -1
- package/dist/cjs/chunk/index.d.ts +1 -1
- package/dist/cjs/chunk/index.js +3 -2
- package/dist/cjs/chunk/index.js.map +1 -1
- package/dist/cjs/chunk/parse.js +46 -29
- package/dist/cjs/chunk/parse.js.map +1 -1
- package/dist/cjs/chunk/tokens.d.cts +2 -0
- package/dist/cjs/chunk/tokens.d.ts +2 -0
- package/dist/cjs/chunk/tokens.js +50 -0
- package/dist/cjs/chunk/tokens.js.map +1 -0
- package/dist/cjs/commands/search.js +1 -1
- package/dist/cjs/commands/search.js.map +1 -1
- package/dist/cjs/config/validate.js.map +1 -1
- package/dist/cjs/index.d.cts +1 -1
- package/dist/cjs/index.d.ts +1 -1
- package/dist/cjs/index.js.map +1 -1
- package/dist/cjs/scan/pool.d.cts +1 -0
- package/dist/cjs/scan/pool.d.ts +1 -0
- package/dist/cjs/scan/pool.js +2 -1
- package/dist/cjs/scan/pool.js.map +1 -1
- package/dist/cjs/scan/reparse.d.cts +1 -0
- package/dist/cjs/scan/reparse.d.ts +1 -0
- package/dist/cjs/scan/reparse.js +6 -3
- package/dist/cjs/scan/reparse.js.map +1 -1
- package/dist/cjs/store/builder.d.cts +2 -0
- package/dist/cjs/store/builder.d.ts +2 -0
- package/dist/cjs/store/builder.js.map +1 -1
- package/dist/cjs/store/duckdb/connection.d.cts +4 -1
- package/dist/cjs/store/duckdb/connection.d.ts +4 -1
- package/dist/cjs/store/duckdb/connection.js +1 -0
- package/dist/cjs/store/duckdb/connection.js.map +1 -1
- package/dist/cjs/store/duckdb/reconcile.js +105 -1
- package/dist/cjs/store/duckdb/reconcile.js.map +1 -1
- package/dist/cjs/store/index.d.cts +1 -5
- package/dist/cjs/store/index.d.ts +1 -5
- package/dist/cjs/store/index.js +9 -46
- package/dist/cjs/store/index.js.map +1 -1
- package/dist/cjs/store/lock-wait.d.cts +2 -0
- package/dist/cjs/store/lock-wait.d.ts +2 -0
- package/dist/cjs/store/lock-wait.js +51 -0
- package/dist/cjs/store/lock-wait.js.map +1 -0
- package/dist/cjs/store/open.d.cts +2 -4
- package/dist/cjs/store/open.d.ts +2 -4
- package/dist/cjs/store/open.js +29 -66
- package/dist/cjs/store/open.js.map +1 -1
- package/dist/cjs/store/reconcile.d.cts +3 -0
- package/dist/cjs/store/reconcile.d.ts +3 -0
- package/dist/cjs/store/reconcile.js +418 -226
- package/dist/cjs/store/reconcile.js.map +1 -1
- package/dist/cjs/store/sqlite/lexical.js +1 -1
- package/dist/cjs/store/sqlite/lexical.js.map +1 -1
- package/dist/cjs/store/sqlite/store.js +0 -1
- package/dist/cjs/store/sqlite/store.js.map +1 -1
- package/dist/cjs/store/stages.d.cts +18 -0
- package/dist/cjs/store/stages.d.ts +18 -0
- package/dist/cjs/store/stages.js +372 -0
- package/dist/cjs/store/stages.js.map +1 -0
- package/dist/cjs/store/turso/store.js.map +1 -1
- package/dist/cjs/store/types.d.cts +2 -1
- package/dist/cjs/store/types.d.ts +2 -1
- package/dist/cjs/store/types.js +0 -1
- package/dist/cjs/store/types.js.map +1 -1
- package/dist/cjs/text/segment.js +7 -5
- package/dist/cjs/text/segment.js.map +1 -1
- package/dist/cjs/watch.js +162 -107
- package/dist/cjs/watch.js.map +1 -1
- package/dist/cjs/workers/parse.d.cts +1 -0
- package/dist/cjs/workers/parse.d.ts +1 -0
- package/dist/cjs/workers/parse.js +4 -1
- package/dist/cjs/workers/parse.js.map +1 -1
- package/dist/esm/chunk/group.d.ts +0 -2
- package/dist/esm/chunk/group.js +14 -24
- package/dist/esm/chunk/group.js.map +1 -1
- package/dist/esm/chunk/index.d.ts +1 -1
- package/dist/esm/chunk/index.js +1 -1
- package/dist/esm/chunk/index.js.map +1 -1
- package/dist/esm/chunk/parse.js +41 -29
- package/dist/esm/chunk/parse.js.map +1 -1
- package/dist/esm/chunk/tokens.d.ts +2 -0
- package/dist/esm/chunk/tokens.js +15 -0
- package/dist/esm/chunk/tokens.js.map +1 -0
- package/dist/esm/commands/search.js +1 -1
- package/dist/esm/commands/search.js.map +1 -1
- package/dist/esm/config/validate.js +2 -2
- package/dist/esm/config/validate.js.map +1 -1
- package/dist/esm/index.d.ts +1 -1
- package/dist/esm/index.js.map +1 -1
- package/dist/esm/scan/pool.d.ts +1 -0
- package/dist/esm/scan/pool.js +2 -1
- package/dist/esm/scan/pool.js.map +1 -1
- package/dist/esm/scan/reparse.d.ts +1 -0
- package/dist/esm/scan/reparse.js +5 -2
- package/dist/esm/scan/reparse.js.map +1 -1
- package/dist/esm/store/builder.d.ts +2 -0
- package/dist/esm/store/builder.js.map +1 -1
- package/dist/esm/store/duckdb/connection.d.ts +4 -1
- package/dist/esm/store/duckdb/connection.js +1 -0
- package/dist/esm/store/duckdb/connection.js.map +1 -1
- package/dist/esm/store/duckdb/reconcile.js +34 -1
- package/dist/esm/store/duckdb/reconcile.js.map +1 -1
- package/dist/esm/store/index.d.ts +1 -5
- package/dist/esm/store/index.js +6 -28
- package/dist/esm/store/index.js.map +1 -1
- package/dist/esm/store/lock-wait.d.ts +2 -0
- package/dist/esm/store/lock-wait.js +36 -0
- package/dist/esm/store/lock-wait.js.map +1 -0
- package/dist/esm/store/open.d.ts +2 -4
- package/dist/esm/store/open.js +17 -32
- package/dist/esm/store/open.js.map +1 -1
- package/dist/esm/store/reconcile.d.ts +3 -0
- package/dist/esm/store/reconcile.js +77 -40
- package/dist/esm/store/reconcile.js.map +1 -1
- package/dist/esm/store/sqlite/lexical.js +1 -1
- package/dist/esm/store/sqlite/lexical.js.map +1 -1
- package/dist/esm/store/sqlite/store.js +0 -1
- package/dist/esm/store/sqlite/store.js.map +1 -1
- package/dist/esm/store/stages.d.ts +18 -0
- package/dist/esm/store/stages.js +63 -0
- package/dist/esm/store/stages.js.map +1 -0
- package/dist/esm/store/turso/store.js +1 -1
- package/dist/esm/store/turso/store.js.map +1 -1
- package/dist/esm/store/types.d.ts +2 -1
- package/dist/esm/store/types.js +0 -1
- package/dist/esm/store/types.js.map +1 -1
- package/dist/esm/text/segment.js +7 -5
- package/dist/esm/text/segment.js.map +1 -1
- package/dist/esm/watch.js +63 -38
- package/dist/esm/watch.js.map +1 -1
- package/dist/esm/workers/parse.d.ts +1 -0
- package/dist/esm/workers/parse.js +4 -1
- package/dist/esm/workers/parse.js.map +1 -1
- package/package.json +3 -3
- package/skills/sense/SKILL.md +1 -1
- package/skills/sense-setup/SKILL.md +1 -1
|
@@ -27,11 +27,14 @@ var selected = _indexts.FEATURES.filter(function(feature) {
|
|
|
27
27
|
});
|
|
28
28
|
function parseTask(file) {
|
|
29
29
|
try {
|
|
30
|
+
var start = process.hrtime.bigint();
|
|
30
31
|
var _parseFile = (0, _indexts1.parseFile)(file, (0, _reparsets.featuresForFile)(selected, cfg, file), cfg), doc = _parseFile.doc, warnings = _parseFile.warnings;
|
|
32
|
+
var parseMs = Number(process.hrtime.bigint() - start) / 1e6;
|
|
31
33
|
return {
|
|
32
34
|
ok: true,
|
|
33
35
|
doc: doc,
|
|
34
|
-
warnings: warnings
|
|
36
|
+
warnings: warnings,
|
|
37
|
+
parseMs: parseMs
|
|
35
38
|
};
|
|
36
39
|
} catch (err) {
|
|
37
40
|
return {
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"sources":["/Users/kevin/Dev/OpenSource/ai/sensemaking/src/workers/parse.ts"],"sourcesContent":["import Tinypool from 'tinypool';\nimport type { Config, FeatureName } from '../config/index.ts';\nimport { FEATURES } from '../features/index.ts';\nimport type { ParsedDoc } from '../scan/index.ts';\nimport { parseFile } from '../scan/index.ts';\nimport type { FileStat } from '../scan/list.ts';\nimport { featuresForFile } from '../scan/reparse.ts';\nimport type { WorkerErrorPayload } from '../scan/worker-error.ts';\nimport { serializeError } from '../scan/worker-error.ts';\n\n// Constant for the whole dispatch, so it crosses once per worker instead of once per task. A\n// Feature carries closures and cannot cross the thread boundary; its name can, and the registry here resolves it back.\nexport interface ParseWorkerData {\n cfg: Config;\n featureNames: FeatureName[];\n}\n\n// tinypool's task, in and out. The task itself is one FileStat. Result carries only what\n// parseFile already returns -- extracted text and per-feature values, never the mdast tree.\nexport type ParseTask = FileStat;\n\nexport type ParseTaskResult = { ok: true; doc: ParsedDoc; warnings: string[] } | { ok: false; error: WorkerErrorPayload };\n\n// Read once per worker, not per task.\nconst { cfg, featureNames } = Tinypool.workerData as ParseWorkerData;\n// Filtering the registry (rather than mapping the names) keeps registry order, which is the\n// order `extracted` keys land in on the serial path.\nconst selected = FEATURES.filter((feature) => featureNames.includes(feature.name));\n\nexport default function parseTask(file: ParseTask): ParseTaskResult {\n try {\n const { doc, warnings } = parseFile(file, featuresForFile(selected, cfg, file), cfg);\n return { ok: true, doc, warnings };\n } catch (err) {\n return { ok: false, error: serializeError(err) };\n }\n}\n"],"names":["parseTask","Tinypool","workerData","cfg","featureNames","selected","FEATURES","filter","feature","includes","name","file","parseFile","featuresForFile","doc","warnings","ok","err","error","serializeError"],"mappings":";;;;+
|
|
1
|
+
{"version":3,"sources":["/Users/kevin/Dev/OpenSource/ai/sensemaking/src/workers/parse.ts"],"sourcesContent":["import Tinypool from 'tinypool';\nimport type { Config, FeatureName } from '../config/index.ts';\nimport { FEATURES } from '../features/index.ts';\nimport type { ParsedDoc } from '../scan/index.ts';\nimport { parseFile } from '../scan/index.ts';\nimport type { FileStat } from '../scan/list.ts';\nimport { featuresForFile } from '../scan/reparse.ts';\nimport type { WorkerErrorPayload } from '../scan/worker-error.ts';\nimport { serializeError } from '../scan/worker-error.ts';\n\n// Constant for the whole dispatch, so it crosses once per worker instead of once per task. A\n// Feature carries closures and cannot cross the thread boundary; its name can, and the registry here resolves it back.\nexport interface ParseWorkerData {\n cfg: Config;\n featureNames: FeatureName[];\n}\n\n// tinypool's task, in and out. The task itself is one FileStat. Result carries only what\n// parseFile already returns -- extracted text and per-feature values, never the mdast tree.\nexport type ParseTask = FileStat;\n\n// parseMs is the worker's own hrtime for parseFile, excluding dispatch and the return clone --\n// what separates it from the pool's dispatch-to-drain `parse` stage.\nexport type ParseTaskResult = { ok: true; doc: ParsedDoc; warnings: string[]; parseMs: number } | { ok: false; error: WorkerErrorPayload };\n\n// Read once per worker, not per task.\nconst { cfg, featureNames } = Tinypool.workerData as ParseWorkerData;\n// Filtering the registry (rather than mapping the names) keeps registry order, which is the\n// order `extracted` keys land in on the serial path.\nconst selected = FEATURES.filter((feature) => featureNames.includes(feature.name));\n\nexport default function parseTask(file: ParseTask): ParseTaskResult {\n try {\n const start = process.hrtime.bigint();\n const { doc, warnings } = parseFile(file, featuresForFile(selected, cfg, file), cfg);\n const parseMs = Number(process.hrtime.bigint() - start) / 1e6;\n return { ok: true, doc, warnings, parseMs };\n } catch (err) {\n return { ok: false, error: serializeError(err) };\n }\n}\n"],"names":["parseTask","Tinypool","workerData","cfg","featureNames","selected","FEATURES","filter","feature","includes","name","file","start","process","hrtime","bigint","parseFile","featuresForFile","doc","warnings","parseMs","Number","ok","err","error","serializeError"],"mappings":";;;;+BA+BA;;;eAAwBA;;;+DA/BH;uBAEI;wBAEC;yBAEM;6BAED;;;;;;AAiB/B,sCAAsC;AACtC,IAA8BC,uBAAAA,iBAAQ,CAACC,UAAU,EAAzCC,MAAsBF,qBAAtBE,KAAKC,eAAiBH,qBAAjBG;AACb,4FAA4F;AAC5F,qDAAqD;AACrD,IAAMC,WAAWC,iBAAQ,CAACC,MAAM,CAAC,SAACC;WAAYJ,aAAaK,QAAQ,CAACD,QAAQE,IAAI;;AAEjE,SAASV,UAAUW,IAAe;IAC/C,IAAI;QACF,IAAMC,QAAQC,QAAQC,MAAM,CAACC,MAAM;QACnC,IAA0BC,aAAAA,IAAAA,mBAAS,EAACL,MAAMM,IAAAA,0BAAe,EAACZ,UAAUF,KAAKQ,OAAOR,MAAxEe,MAAkBF,WAAlBE,KAAKC,WAAaH,WAAbG;QACb,IAAMC,UAAUC,OAAOR,QAAQC,MAAM,CAACC,MAAM,KAAKH,SAAS;QAC1D,OAAO;YAAEU,IAAI;YAAMJ,KAAAA;YAAKC,UAAAA;YAAUC,SAAAA;QAAQ;IAC5C,EAAE,OAAOG,KAAK;QACZ,OAAO;YAAED,IAAI;YAAOE,OAAOC,IAAAA,6BAAc,EAACF;QAAK;IACjD;AACF"}
|
package/dist/esm/chunk/group.js
CHANGED
|
@@ -1,22 +1,8 @@
|
|
|
1
|
-
import { UNSPACED_SCRIPTS } from '../text/segment.js';
|
|
2
1
|
import { extractText } from './extract.js';
|
|
3
2
|
import { parse } from './parse.js';
|
|
4
|
-
|
|
3
|
+
import { DEFAULT_TARGET_TOKENS, estimateTokens } from './tokens.js';
|
|
5
4
|
const PGC_GROUP_SIZE = 2;
|
|
6
5
|
const OVERSIZE_TRIGGER_MULTIPLE = 2;
|
|
7
|
-
// D5: segment.ts's unspaced-script set packs close to one token per character; Hangul is
|
|
8
|
-
// spaced but still token-dense and sits in this set as a conservative bound (one Unicode set, one home).
|
|
9
|
-
const DENSE_SCRIPT = new RegExp(`[${UNSPACED_SCRIPTS}\\p{scx=Hangul}]`, 'u');
|
|
10
|
-
// D5's size estimate: dense-script graphemes 1:1, everything else at 4 chars/token.
|
|
11
|
-
export function estimateTokens(text) {
|
|
12
|
-
let dense = 0;
|
|
13
|
-
let other = 0;
|
|
14
|
-
for (const ch of text){
|
|
15
|
-
if (DENSE_SCRIPT.test(ch)) dense++;
|
|
16
|
-
else other++;
|
|
17
|
-
}
|
|
18
|
-
return dense + other / 4;
|
|
19
|
-
}
|
|
20
6
|
function resolveOptions(opts) {
|
|
21
7
|
var _ref, _ref1;
|
|
22
8
|
return {
|
|
@@ -51,13 +37,17 @@ function finalize(parts) {
|
|
|
51
37
|
};
|
|
52
38
|
}
|
|
53
39
|
const NEWLINE_TOKENS = estimateTokens('\n');
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
granularity
|
|
59
|
-
|
|
60
|
-
|
|
40
|
+
// Built on first use and kept: each construction is ~3.5 ms, and only an oversize block is ever
|
|
41
|
+
// split, so no command pays for a segmenter it never reaches.
|
|
42
|
+
const SEGMENTERS = new Map();
|
|
43
|
+
function segmentsOf(text, granularity) {
|
|
44
|
+
let segmenter = SEGMENTERS.get(granularity);
|
|
45
|
+
if (!segmenter) {
|
|
46
|
+
segmenter = new Intl.Segmenter(undefined, {
|
|
47
|
+
granularity
|
|
48
|
+
});
|
|
49
|
+
SEGMENTERS.set(granularity, segmenter);
|
|
50
|
+
}
|
|
61
51
|
return Array.from(segmenter.segment(text), (s)=>s.segment);
|
|
62
52
|
}
|
|
63
53
|
// Greedily packs segments (already contiguous, tiling the source text with no gaps) into groups
|
|
@@ -82,7 +72,7 @@ function pack(segments, working) {
|
|
|
82
72
|
// Line-split alone can't shrink a lone dense line (the CJK case): falls back to sentence then
|
|
83
73
|
// word boundaries (Intl.Segmenter, the same grapheme-safe engine as segment.ts), mode-agnostic on `text`.
|
|
84
74
|
function splitLineText(text, working) {
|
|
85
|
-
const sentences = segmentsOf(text,
|
|
75
|
+
const sentences = segmentsOf(text, 'sentence');
|
|
86
76
|
const out = [];
|
|
87
77
|
let current = '';
|
|
88
78
|
let tokens = 0;
|
|
@@ -97,7 +87,7 @@ function splitLineText(text, working) {
|
|
|
97
87
|
const sentenceTokens = estimateTokens(sentence);
|
|
98
88
|
if (sentenceTokens > working) {
|
|
99
89
|
flush();
|
|
100
|
-
out.push(...pack(segmentsOf(sentence,
|
|
90
|
+
out.push(...pack(segmentsOf(sentence, 'word'), working));
|
|
101
91
|
continue;
|
|
102
92
|
}
|
|
103
93
|
if (current.length > 0 && tokens + sentenceTokens > working) flush();
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"sources":["/Users/kevin/Dev/OpenSource/ai/sensemaking/src/chunk/group.ts"],"sourcesContent":["import { UNSPACED_SCRIPTS } from '../text/segment.ts';\nimport { extractText } from './extract.ts';\nimport { parse } from './parse.ts';\nimport type { Block, BlockType, Chunk, ChunkOptions } from './types.ts';\n\nexport const DEFAULT_TARGET_TOKENS = 500;\nconst PGC_GROUP_SIZE = 2;\nconst OVERSIZE_TRIGGER_MULTIPLE = 2;\n\n// D5: segment.ts's unspaced-script set packs close to one token per character; Hangul is\n// spaced but still token-dense and sits in this set as a conservative bound (one Unicode set, one home).\nconst DENSE_SCRIPT = new RegExp(`[${UNSPACED_SCRIPTS}\\\\p{scx=Hangul}]`, 'u');\n\n// D5's size estimate: dense-script graphemes 1:1, everything else at 4 chars/token.\nexport function estimateTokens(text: string): number {\n let dense = 0;\n let other = 0;\n for (const ch of text) {\n if (DENSE_SCRIPT.test(ch)) dense++;\n else other++;\n }\n return dense + other / 4;\n}\n\ninterface ResolvedOptions {\n targetTokens: number;\n text: 'extracted' | 'raw';\n}\n\nfunction resolveOptions(opts?: ChunkOptions): ResolvedOptions {\n return {\n targetTokens: opts?.targetTokens ?? DEFAULT_TARGET_TOKENS,\n text: opts?.text ?? 'raw',\n };\n}\n\n// A heading of any depth ends the current scope and starts a new one (D1); the heading block\n// itself is carried into the new scope, where it joins that scope's first group (F7/F10).\nfunction splitScopes(blocks: Block[]): Block[][] {\n const scopes: Block[][] = [];\n let current: Block[] = [];\n for (const block of blocks) {\n if (block.type === 'heading' && current.length > 0) {\n scopes.push(current);\n current = [];\n }\n current.push(block);\n }\n if (current.length > 0) scopes.push(current);\n return scopes;\n}\n\ninterface Part {\n startLine: number;\n endLine: number;\n text: string;\n // True for a sub-line split piece: its text is already the final slice, not the whole line --\n // group()'s extent-based raw re-slice must not touch it (siblings share startLine === endLine).\n final?: boolean;\n}\n\nfunction finalize(parts: Part[]): (Chunk & { final?: boolean }) | undefined {\n if (parts.length === 0) return undefined;\n const first = parts[0];\n const last = parts[parts.length - 1];\n return { startLine: first.startLine, endLine: last.endLine, text: parts.map((p) => p.text).join('\\n'), final: parts.some((p) => p.final) };\n}\n\nconst NEWLINE_TOKENS = estimateTokens('\\n');\nconst SENTENCE_SEGMENTER = new Intl.Segmenter(undefined, { granularity: 'sentence' });\nconst WORD_SEGMENTER = new Intl.Segmenter(undefined, { granularity: 'word' });\n\nfunction segmentsOf(text: string, segmenter: Intl.Segmenter): string[] {\n return Array.from(segmenter.segment(text), (s) => s.segment);\n}\n\n// Greedily packs segments (already contiguous, tiling the source text with no gaps) into groups\n// of at most `working` estimated tokens; a lone segment over `working` still stands alone.\nfunction pack(segments: string[], working: number): string[] {\n const groups: string[] = [];\n let current = '';\n let tokens = 0;\n for (const segment of segments) {\n const segmentTokens = estimateTokens(segment);\n if (current.length > 0 && tokens + segmentTokens > working) {\n groups.push(current);\n current = '';\n tokens = 0;\n }\n current += segment;\n tokens += segmentTokens;\n }\n if (current.length > 0) groups.push(current);\n return groups;\n}\n\n// Line-split alone can't shrink a lone dense line (the CJK case): falls back to sentence then\n// word boundaries (Intl.Segmenter, the same grapheme-safe engine as segment.ts), mode-agnostic on `text`.\nfunction splitLineText(text: string, working: number): string[] {\n const sentences = segmentsOf(text, SENTENCE_SEGMENTER);\n const out: string[] = [];\n let current = '';\n let tokens = 0;\n const flush = () => {\n if (current.length > 0) {\n out.push(current);\n current = '';\n tokens = 0;\n }\n };\n for (const sentence of sentences) {\n const sentenceTokens = estimateTokens(sentence);\n if (sentenceTokens > working) {\n flush();\n out.push(...pack(segmentsOf(sentence, WORD_SEGMENTER), working));\n continue;\n }\n if (current.length > 0 && tokens + sentenceTokens > working) flush();\n current += sentence;\n tokens += sentenceTokens;\n }\n flush();\n return out;\n}\n\nconst ATOMIC_TYPES: ReadonlySet<BlockType> = new Set(['code', 'table', 'list']);\n\n// Atomic (code/table/list) pieces are always a raw line slice: re-parsing a table's later pieces\n// without their header/delimiter rows would demote them to paragraph text.\nfunction piece(pieceLines: string[], startLine: number, endLine: number, blockType: BlockType, textMode: 'extracted' | 'raw'): Part {\n const text =\n ATOMIC_TYPES.has(blockType) || textMode === 'raw'\n ? pieceLines.join('\\n')\n : parse(pieceLines.join('\\n'))\n .map((b) => extractText(b.node))\n .join('\\n');\n return { startLine, endLine, text };\n}\n\n// A one-line piece over working can't shrink via another line-boundary pass (rule 5's gap), so it\n// splits at sentence/word boundaries instead; `final` stops group() re-deriving its text by extent (F5).\nfunction finalizePiece(pieceLines: string[], startLine: number, endLine: number, working: number, blockType: BlockType, textMode: 'extracted' | 'raw'): Part[] {\n const p = piece(pieceLines, startLine, endLine, blockType, textMode);\n if (pieceLines.length === 1 && estimateTokens(p.text) > working) {\n return splitLineText(p.text, working).map((text) => ({ startLine, endLine, text, final: true }));\n }\n return [p];\n}\n\n// A block over 2x working size splits at line boundaries into pieces each <= working size, never\n// mid-line. `seed`: pending tokens (e.g. a heading) the first piece must join, checked against the limit.\nfunction splitOversizeBlock(lines: string[], startLine: number, endLine: number, working: number, blockType: BlockType, textMode: 'extracted' | 'raw', seed = 0): Part[] {\n const pieces: Part[] = [];\n let pieceLines: string[] = [];\n let pieceStart = startLine;\n let tokens = seed;\n for (let line = startLine; line <= endLine; line++) {\n const lineText = lines[line - 1];\n const sep = pieceLines.length > 0 || tokens > 0 ? NEWLINE_TOKENS : 0;\n const lineTokens = estimateTokens(lineText);\n if (pieceLines.length > 0 && tokens + sep + lineTokens > working) {\n pieces.push(...finalizePiece(pieceLines, pieceStart, line - 1, working, blockType, textMode));\n pieceLines = [];\n tokens = 0;\n pieceStart = line;\n pieceLines.push(lineText);\n tokens += lineTokens;\n continue;\n }\n pieceLines.push(lineText);\n tokens += sep + lineTokens;\n }\n if (pieceLines.length > 0) pieces.push(...finalizePiece(pieceLines, pieceStart, endLine, working, blockType, textMode));\n return pieces;\n}\n\n// One heading scope's groups (D1): a heading opens the first group, and an oversize block\n// (rule 5, including an oversize heading) splits into pieces that each close their own group.\nfunction groupScope(scopeBlocks: Block[], lines: string[], resolved: ResolvedOptions): (Chunk & { final?: boolean })[] {\n const working = resolved.targetTokens;\n const trigger = working * OVERSIZE_TRIGGER_MULTIPLE;\n const finished: (Chunk & { final?: boolean })[] = [];\n let parts: Part[] = [];\n let paragraphCount = 0;\n let tokens = 0;\n\n function close(): void {\n const group = finalize(parts);\n if (group) finished.push(group);\n parts = [];\n paragraphCount = 0;\n tokens = 0;\n }\n\n // tokens tracks the active text mode's own estimate (a newline between parts costs\n // NEWLINE_TOKENS too), so packing decisions size the text the chunk will actually ship as.\n function addPart(text: string, sizeText: string, startLine: number, endLine: number): void {\n tokens += (parts.length > 0 ? NEWLINE_TOKENS : 0) + estimateTokens(sizeText);\n parts.push({ startLine, endLine, text });\n }\n\n for (const block of scopeBlocks) {\n const raw = lines.slice(block.startLine - 1, block.endLine).join('\\n');\n const blockTokens = estimateTokens(raw);\n\n if (blockTokens > trigger) {\n const seed = parts.length > 0 ? tokens : 0;\n for (const p of splitOversizeBlock(lines, block.startLine, block.endLine, working, block.type, resolved.text, seed)) {\n parts.push(p);\n close();\n }\n continue;\n }\n\n const extracted = extractText(block.node);\n const sizeText = resolved.text === 'raw' ? raw : extracted;\n\n if (block.type === 'heading') {\n addPart(extracted, sizeText, block.startLine, block.endLine);\n continue;\n }\n\n // The 2x-working invariant holds even under pgc's paper-faithful 2-paragraph pairing --\n // close first if the pair about to form would cross it.\n const pairOversize = parts.length > 0 && tokens + NEWLINE_TOKENS + blockTokens > trigger;\n if (pairOversize) close();\n\n addPart(extracted, sizeText, block.startLine, block.endLine);\n paragraphCount++;\n\n if (paragraphCount >= PGC_GROUP_SIZE) close();\n }\n close();\n\n return finished;\n}\n\n// Groups already-parsed blocks per opts (D1/D3), against the same body the blocks were parsed\n// from (line lookups for oversize splitting).\nexport function group(blocks: Block[], body: string, opts?: ChunkOptions): Chunk[] {\n const resolved = resolveOptions(opts);\n const lines = body.split('\\n');\n const chunks: (Chunk & { final?: boolean })[] = [];\n for (const scope of splitScopes(blocks)) chunks.push(...groupScope(scope, lines, resolved));\n // 'raw': the chunk's own source lines verbatim, replacing the flavor-resolved join above (D9).\n // A `final` chunk already carries its own slice's raw text; re-slicing by extent would return the whole shared line.\n const texted =\n resolved.text === 'raw'\n ? chunks.map((c) =>\n c.final\n ? c\n : {\n ...c,\n text: lines\n .slice(c.startLine - 1, c.endLine)\n .join('\\n')\n .trim(),\n }\n )\n : chunks;\n // A group can be all-blank (flavor-stripped to nothing, or a raw slice of pure syntax); it never produces a chunk.\n return texted.filter((c) => c.text.trim().length > 0).map((c) => ({ startLine: c.startLine, endLine: c.endLine, text: c.text }));\n}\n"],"names":["UNSPACED_SCRIPTS","extractText","parse","DEFAULT_TARGET_TOKENS","PGC_GROUP_SIZE","OVERSIZE_TRIGGER_MULTIPLE","DENSE_SCRIPT","RegExp","estimateTokens","text","dense","other","ch","test","resolveOptions","opts","targetTokens","splitScopes","blocks","scopes","current","block","type","length","push","finalize","parts","undefined","first","last","startLine","endLine","map","p","join","final","some","NEWLINE_TOKENS","SENTENCE_SEGMENTER","Intl","Segmenter","granularity","WORD_SEGMENTER","segmentsOf","segmenter","Array","from","segment","s","pack","segments","working","groups","tokens","segmentTokens","splitLineText","sentences","out","flush","sentence","sentenceTokens","ATOMIC_TYPES","Set","piece","pieceLines","blockType","textMode","has","b","node","finalizePiece","splitOversizeBlock","lines","seed","pieces","pieceStart","line","lineText","sep","lineTokens","groupScope","scopeBlocks","resolved","trigger","finished","paragraphCount","close","group","addPart","sizeText","raw","slice","blockTokens","extracted","pairOversize","body","split","chunks","scope","texted","c","trim","filter"],"mappings":"AAAA,SAASA,gBAAgB,QAAQ,qBAAqB;AACtD,SAASC,WAAW,QAAQ,eAAe;AAC3C,SAASC,KAAK,QAAQ,aAAa;AAGnC,OAAO,MAAMC,wBAAwB,IAAI;AACzC,MAAMC,iBAAiB;AACvB,MAAMC,4BAA4B;AAElC,yFAAyF;AACzF,yGAAyG;AACzG,MAAMC,eAAe,IAAIC,OAAO,CAAC,CAAC,EAAEP,iBAAiB,gBAAgB,CAAC,EAAE;AAExE,oFAAoF;AACpF,OAAO,SAASQ,eAAeC,IAAY;IACzC,IAAIC,QAAQ;IACZ,IAAIC,QAAQ;IACZ,KAAK,MAAMC,MAAMH,KAAM;QACrB,IAAIH,aAAaO,IAAI,CAACD,KAAKF;aACtBC;IACP;IACA,OAAOD,QAAQC,QAAQ;AACzB;AAOA,SAASG,eAAeC,IAAmB;;IACzC,OAAO;QACLC,YAAY,UAAED,iBAAAA,2BAAAA,KAAMC,YAAY,uCAAIb;QACpCM,IAAI,WAAEM,iBAAAA,2BAAAA,KAAMN,IAAI,yCAAI;IACtB;AACF;AAEA,6FAA6F;AAC7F,0FAA0F;AAC1F,SAASQ,YAAYC,MAAe;IAClC,MAAMC,SAAoB,EAAE;IAC5B,IAAIC,UAAmB,EAAE;IACzB,KAAK,MAAMC,SAASH,OAAQ;QAC1B,IAAIG,MAAMC,IAAI,KAAK,aAAaF,QAAQG,MAAM,GAAG,GAAG;YAClDJ,OAAOK,IAAI,CAACJ;YACZA,UAAU,EAAE;QACd;QACAA,QAAQI,IAAI,CAACH;IACf;IACA,IAAID,QAAQG,MAAM,GAAG,GAAGJ,OAAOK,IAAI,CAACJ;IACpC,OAAOD;AACT;AAWA,SAASM,SAASC,KAAa;IAC7B,IAAIA,MAAMH,MAAM,KAAK,GAAG,OAAOI;IAC/B,MAAMC,QAAQF,KAAK,CAAC,EAAE;IACtB,MAAMG,OAAOH,KAAK,CAACA,MAAMH,MAAM,GAAG,EAAE;IACpC,OAAO;QAAEO,WAAWF,MAAME,SAAS;QAAEC,SAASF,KAAKE,OAAO;QAAEtB,MAAMiB,MAAMM,GAAG,CAAC,CAACC,IAAMA,EAAExB,IAAI,EAAEyB,IAAI,CAAC;QAAOC,OAAOT,MAAMU,IAAI,CAAC,CAACH,IAAMA,EAAEE,KAAK;IAAE;AAC3I;AAEA,MAAME,iBAAiB7B,eAAe;AACtC,MAAM8B,qBAAqB,IAAIC,KAAKC,SAAS,CAACb,WAAW;IAAEc,aAAa;AAAW;AACnF,MAAMC,iBAAiB,IAAIH,KAAKC,SAAS,CAACb,WAAW;IAAEc,aAAa;AAAO;AAE3E,SAASE,WAAWlC,IAAY,EAAEmC,SAAyB;IACzD,OAAOC,MAAMC,IAAI,CAACF,UAAUG,OAAO,CAACtC,OAAO,CAACuC,IAAMA,EAAED,OAAO;AAC7D;AAEA,gGAAgG;AAChG,2FAA2F;AAC3F,SAASE,KAAKC,QAAkB,EAAEC,OAAe;IAC/C,MAAMC,SAAmB,EAAE;IAC3B,IAAIhC,UAAU;IACd,IAAIiC,SAAS;IACb,KAAK,MAAMN,WAAWG,SAAU;QAC9B,MAAMI,gBAAgB9C,eAAeuC;QACrC,IAAI3B,QAAQG,MAAM,GAAG,KAAK8B,SAASC,gBAAgBH,SAAS;YAC1DC,OAAO5B,IAAI,CAACJ;YACZA,UAAU;YACViC,SAAS;QACX;QACAjC,WAAW2B;QACXM,UAAUC;IACZ;IACA,IAAIlC,QAAQG,MAAM,GAAG,GAAG6B,OAAO5B,IAAI,CAACJ;IACpC,OAAOgC;AACT;AAEA,8FAA8F;AAC9F,0GAA0G;AAC1G,SAASG,cAAc9C,IAAY,EAAE0C,OAAe;IAClD,MAAMK,YAAYb,WAAWlC,MAAM6B;IACnC,MAAMmB,MAAgB,EAAE;IACxB,IAAIrC,UAAU;IACd,IAAIiC,SAAS;IACb,MAAMK,QAAQ;QACZ,IAAItC,QAAQG,MAAM,GAAG,GAAG;YACtBkC,IAAIjC,IAAI,CAACJ;YACTA,UAAU;YACViC,SAAS;QACX;IACF;IACA,KAAK,MAAMM,YAAYH,UAAW;QAChC,MAAMI,iBAAiBpD,eAAemD;QACtC,IAAIC,iBAAiBT,SAAS;YAC5BO;YACAD,IAAIjC,IAAI,IAAIyB,KAAKN,WAAWgB,UAAUjB,iBAAiBS;YACvD;QACF;QACA,IAAI/B,QAAQG,MAAM,GAAG,KAAK8B,SAASO,iBAAiBT,SAASO;QAC7DtC,WAAWuC;QACXN,UAAUO;IACZ;IACAF;IACA,OAAOD;AACT;AAEA,MAAMI,eAAuC,IAAIC,IAAI;IAAC;IAAQ;IAAS;CAAO;AAE9E,iGAAiG;AACjG,2EAA2E;AAC3E,SAASC,MAAMC,UAAoB,EAAElC,SAAiB,EAAEC,OAAe,EAAEkC,SAAoB,EAAEC,QAA6B;IAC1H,MAAMzD,OACJoD,aAAaM,GAAG,CAACF,cAAcC,aAAa,QACxCF,WAAW9B,IAAI,CAAC,QAChBhC,MAAM8D,WAAW9B,IAAI,CAAC,OACnBF,GAAG,CAAC,CAACoC,IAAMnE,YAAYmE,EAAEC,IAAI,GAC7BnC,IAAI,CAAC;IACd,OAAO;QAAEJ;QAAWC;QAAStB;IAAK;AACpC;AAEA,kGAAkG;AAClG,yGAAyG;AACzG,SAAS6D,cAAcN,UAAoB,EAAElC,SAAiB,EAAEC,OAAe,EAAEoB,OAAe,EAAEc,SAAoB,EAAEC,QAA6B;IACnJ,MAAMjC,IAAI8B,MAAMC,YAAYlC,WAAWC,SAASkC,WAAWC;IAC3D,IAAIF,WAAWzC,MAAM,KAAK,KAAKf,eAAeyB,EAAExB,IAAI,IAAI0C,SAAS;QAC/D,OAAOI,cAActB,EAAExB,IAAI,EAAE0C,SAASnB,GAAG,CAAC,CAACvB,OAAU,CAAA;gBAAEqB;gBAAWC;gBAAStB;gBAAM0B,OAAO;YAAK,CAAA;IAC/F;IACA,OAAO;QAACF;KAAE;AACZ;AAEA,iGAAiG;AACjG,0GAA0G;AAC1G,SAASsC,mBAAmBC,KAAe,EAAE1C,SAAiB,EAAEC,OAAe,EAAEoB,OAAe,EAAEc,SAAoB,EAAEC,QAA6B,EAAEO,OAAO,CAAC;IAC7J,MAAMC,SAAiB,EAAE;IACzB,IAAIV,aAAuB,EAAE;IAC7B,IAAIW,aAAa7C;IACjB,IAAIuB,SAASoB;IACb,IAAK,IAAIG,OAAO9C,WAAW8C,QAAQ7C,SAAS6C,OAAQ;QAClD,MAAMC,WAAWL,KAAK,CAACI,OAAO,EAAE;QAChC,MAAME,MAAMd,WAAWzC,MAAM,GAAG,KAAK8B,SAAS,IAAIhB,iBAAiB;QACnE,MAAM0C,aAAavE,eAAeqE;QAClC,IAAIb,WAAWzC,MAAM,GAAG,KAAK8B,SAASyB,MAAMC,aAAa5B,SAAS;YAChEuB,OAAOlD,IAAI,IAAI8C,cAAcN,YAAYW,YAAYC,OAAO,GAAGzB,SAASc,WAAWC;YACnFF,aAAa,EAAE;YACfX,SAAS;YACTsB,aAAaC;YACbZ,WAAWxC,IAAI,CAACqD;YAChBxB,UAAU0B;YACV;QACF;QACAf,WAAWxC,IAAI,CAACqD;QAChBxB,UAAUyB,MAAMC;IAClB;IACA,IAAIf,WAAWzC,MAAM,GAAG,GAAGmD,OAAOlD,IAAI,IAAI8C,cAAcN,YAAYW,YAAY5C,SAASoB,SAASc,WAAWC;IAC7G,OAAOQ;AACT;AAEA,0FAA0F;AAC1F,8FAA8F;AAC9F,SAASM,WAAWC,WAAoB,EAAET,KAAe,EAAEU,QAAyB;IAClF,MAAM/B,UAAU+B,SAASlE,YAAY;IACrC,MAAMmE,UAAUhC,UAAU9C;IAC1B,MAAM+E,WAA4C,EAAE;IACpD,IAAI1D,QAAgB,EAAE;IACtB,IAAI2D,iBAAiB;IACrB,IAAIhC,SAAS;IAEb,SAASiC;QACP,MAAMC,QAAQ9D,SAASC;QACvB,IAAI6D,OAAOH,SAAS5D,IAAI,CAAC+D;QACzB7D,QAAQ,EAAE;QACV2D,iBAAiB;QACjBhC,SAAS;IACX;IAEA,mFAAmF;IACnF,2FAA2F;IAC3F,SAASmC,QAAQ/E,IAAY,EAAEgF,QAAgB,EAAE3D,SAAiB,EAAEC,OAAe;QACjFsB,UAAU,AAAC3B,CAAAA,MAAMH,MAAM,GAAG,IAAIc,iBAAiB,CAAA,IAAK7B,eAAeiF;QACnE/D,MAAMF,IAAI,CAAC;YAAEM;YAAWC;YAAStB;QAAK;IACxC;IAEA,KAAK,MAAMY,SAAS4D,YAAa;QAC/B,MAAMS,MAAMlB,MAAMmB,KAAK,CAACtE,MAAMS,SAAS,GAAG,GAAGT,MAAMU,OAAO,EAAEG,IAAI,CAAC;QACjE,MAAM0D,cAAcpF,eAAekF;QAEnC,IAAIE,cAAcT,SAAS;YACzB,MAAMV,OAAO/C,MAAMH,MAAM,GAAG,IAAI8B,SAAS;YACzC,KAAK,MAAMpB,KAAKsC,mBAAmBC,OAAOnD,MAAMS,SAAS,EAAET,MAAMU,OAAO,EAAEoB,SAAS9B,MAAMC,IAAI,EAAE4D,SAASzE,IAAI,EAAEgE,MAAO;gBACnH/C,MAAMF,IAAI,CAACS;gBACXqD;YACF;YACA;QACF;QAEA,MAAMO,YAAY5F,YAAYoB,MAAMgD,IAAI;QACxC,MAAMoB,WAAWP,SAASzE,IAAI,KAAK,QAAQiF,MAAMG;QAEjD,IAAIxE,MAAMC,IAAI,KAAK,WAAW;YAC5BkE,QAAQK,WAAWJ,UAAUpE,MAAMS,SAAS,EAAET,MAAMU,OAAO;YAC3D;QACF;QAEA,wFAAwF;QACxF,wDAAwD;QACxD,MAAM+D,eAAepE,MAAMH,MAAM,GAAG,KAAK8B,SAAShB,iBAAiBuD,cAAcT;QACjF,IAAIW,cAAcR;QAElBE,QAAQK,WAAWJ,UAAUpE,MAAMS,SAAS,EAAET,MAAMU,OAAO;QAC3DsD;QAEA,IAAIA,kBAAkBjF,gBAAgBkF;IACxC;IACAA;IAEA,OAAOF;AACT;AAEA,8FAA8F;AAC9F,8CAA8C;AAC9C,OAAO,SAASG,MAAMrE,MAAe,EAAE6E,IAAY,EAAEhF,IAAmB;IACtE,MAAMmE,WAAWpE,eAAeC;IAChC,MAAMyD,QAAQuB,KAAKC,KAAK,CAAC;IACzB,MAAMC,SAA0C,EAAE;IAClD,KAAK,MAAMC,SAASjF,YAAYC,QAAS+E,OAAOzE,IAAI,IAAIwD,WAAWkB,OAAO1B,OAAOU;IACjF,+FAA+F;IAC/F,qHAAqH;IACrH,MAAMiB,SACJjB,SAASzE,IAAI,KAAK,QACdwF,OAAOjE,GAAG,CAAC,CAACoE,IACVA,EAAEjE,KAAK,GACHiE,IACA;YACE,GAAGA,CAAC;YACJ3F,MAAM+D,MACHmB,KAAK,CAACS,EAAEtE,SAAS,GAAG,GAAGsE,EAAErE,OAAO,EAChCG,IAAI,CAAC,MACLmE,IAAI;QACT,KAENJ;IACN,mHAAmH;IACnH,OAAOE,OAAOG,MAAM,CAAC,CAACF,IAAMA,EAAE3F,IAAI,CAAC4F,IAAI,GAAG9E,MAAM,GAAG,GAAGS,GAAG,CAAC,CAACoE,IAAO,CAAA;YAAEtE,WAAWsE,EAAEtE,SAAS;YAAEC,SAASqE,EAAErE,OAAO;YAAEtB,MAAM2F,EAAE3F,IAAI;QAAC,CAAA;AAC/H"}
|
|
1
|
+
{"version":3,"sources":["/Users/kevin/Dev/OpenSource/ai/sensemaking/src/chunk/group.ts"],"sourcesContent":["import { extractText } from './extract.ts';\nimport { parse } from './parse.ts';\nimport { DEFAULT_TARGET_TOKENS, estimateTokens } from './tokens.ts';\nimport type { Block, BlockType, Chunk, ChunkOptions } from './types.ts';\n\nconst PGC_GROUP_SIZE = 2;\nconst OVERSIZE_TRIGGER_MULTIPLE = 2;\n\ninterface ResolvedOptions {\n targetTokens: number;\n text: 'extracted' | 'raw';\n}\n\nfunction resolveOptions(opts?: ChunkOptions): ResolvedOptions {\n return {\n targetTokens: opts?.targetTokens ?? DEFAULT_TARGET_TOKENS,\n text: opts?.text ?? 'raw',\n };\n}\n\n// A heading of any depth ends the current scope and starts a new one (D1); the heading block\n// itself is carried into the new scope, where it joins that scope's first group (F7/F10).\nfunction splitScopes(blocks: Block[]): Block[][] {\n const scopes: Block[][] = [];\n let current: Block[] = [];\n for (const block of blocks) {\n if (block.type === 'heading' && current.length > 0) {\n scopes.push(current);\n current = [];\n }\n current.push(block);\n }\n if (current.length > 0) scopes.push(current);\n return scopes;\n}\n\ninterface Part {\n startLine: number;\n endLine: number;\n text: string;\n // True for a sub-line split piece: its text is already the final slice, not the whole line --\n // group()'s extent-based raw re-slice must not touch it (siblings share startLine === endLine).\n final?: boolean;\n}\n\nfunction finalize(parts: Part[]): (Chunk & { final?: boolean }) | undefined {\n if (parts.length === 0) return undefined;\n const first = parts[0];\n const last = parts[parts.length - 1];\n return { startLine: first.startLine, endLine: last.endLine, text: parts.map((p) => p.text).join('\\n'), final: parts.some((p) => p.final) };\n}\n\nconst NEWLINE_TOKENS = estimateTokens('\\n');\n// Built on first use and kept: each construction is ~3.5 ms, and only an oversize block is ever\n// split, so no command pays for a segmenter it never reaches.\nconst SEGMENTERS = new Map<string, Intl.Segmenter>();\n\nfunction segmentsOf(text: string, granularity: 'sentence' | 'word'): string[] {\n let segmenter = SEGMENTERS.get(granularity);\n if (!segmenter) {\n segmenter = new Intl.Segmenter(undefined, { granularity });\n SEGMENTERS.set(granularity, segmenter);\n }\n return Array.from(segmenter.segment(text), (s) => s.segment);\n}\n\n// Greedily packs segments (already contiguous, tiling the source text with no gaps) into groups\n// of at most `working` estimated tokens; a lone segment over `working` still stands alone.\nfunction pack(segments: string[], working: number): string[] {\n const groups: string[] = [];\n let current = '';\n let tokens = 0;\n for (const segment of segments) {\n const segmentTokens = estimateTokens(segment);\n if (current.length > 0 && tokens + segmentTokens > working) {\n groups.push(current);\n current = '';\n tokens = 0;\n }\n current += segment;\n tokens += segmentTokens;\n }\n if (current.length > 0) groups.push(current);\n return groups;\n}\n\n// Line-split alone can't shrink a lone dense line (the CJK case): falls back to sentence then\n// word boundaries (Intl.Segmenter, the same grapheme-safe engine as segment.ts), mode-agnostic on `text`.\nfunction splitLineText(text: string, working: number): string[] {\n const sentences = segmentsOf(text, 'sentence');\n const out: string[] = [];\n let current = '';\n let tokens = 0;\n const flush = () => {\n if (current.length > 0) {\n out.push(current);\n current = '';\n tokens = 0;\n }\n };\n for (const sentence of sentences) {\n const sentenceTokens = estimateTokens(sentence);\n if (sentenceTokens > working) {\n flush();\n out.push(...pack(segmentsOf(sentence, 'word'), working));\n continue;\n }\n if (current.length > 0 && tokens + sentenceTokens > working) flush();\n current += sentence;\n tokens += sentenceTokens;\n }\n flush();\n return out;\n}\n\nconst ATOMIC_TYPES: ReadonlySet<BlockType> = new Set(['code', 'table', 'list']);\n\n// Atomic (code/table/list) pieces are always a raw line slice: re-parsing a table's later pieces\n// without their header/delimiter rows would demote them to paragraph text.\nfunction piece(pieceLines: string[], startLine: number, endLine: number, blockType: BlockType, textMode: 'extracted' | 'raw'): Part {\n const text =\n ATOMIC_TYPES.has(blockType) || textMode === 'raw'\n ? pieceLines.join('\\n')\n : parse(pieceLines.join('\\n'))\n .map((b) => extractText(b.node))\n .join('\\n');\n return { startLine, endLine, text };\n}\n\n// A one-line piece over working can't shrink via another line-boundary pass (rule 5's gap), so it\n// splits at sentence/word boundaries instead; `final` stops group() re-deriving its text by extent (F5).\nfunction finalizePiece(pieceLines: string[], startLine: number, endLine: number, working: number, blockType: BlockType, textMode: 'extracted' | 'raw'): Part[] {\n const p = piece(pieceLines, startLine, endLine, blockType, textMode);\n if (pieceLines.length === 1 && estimateTokens(p.text) > working) {\n return splitLineText(p.text, working).map((text) => ({ startLine, endLine, text, final: true }));\n }\n return [p];\n}\n\n// A block over 2x working size splits at line boundaries into pieces each <= working size, never\n// mid-line. `seed`: pending tokens (e.g. a heading) the first piece must join, checked against the limit.\nfunction splitOversizeBlock(lines: string[], startLine: number, endLine: number, working: number, blockType: BlockType, textMode: 'extracted' | 'raw', seed = 0): Part[] {\n const pieces: Part[] = [];\n let pieceLines: string[] = [];\n let pieceStart = startLine;\n let tokens = seed;\n for (let line = startLine; line <= endLine; line++) {\n const lineText = lines[line - 1];\n const sep = pieceLines.length > 0 || tokens > 0 ? NEWLINE_TOKENS : 0;\n const lineTokens = estimateTokens(lineText);\n if (pieceLines.length > 0 && tokens + sep + lineTokens > working) {\n pieces.push(...finalizePiece(pieceLines, pieceStart, line - 1, working, blockType, textMode));\n pieceLines = [];\n tokens = 0;\n pieceStart = line;\n pieceLines.push(lineText);\n tokens += lineTokens;\n continue;\n }\n pieceLines.push(lineText);\n tokens += sep + lineTokens;\n }\n if (pieceLines.length > 0) pieces.push(...finalizePiece(pieceLines, pieceStart, endLine, working, blockType, textMode));\n return pieces;\n}\n\n// One heading scope's groups (D1): a heading opens the first group, and an oversize block\n// (rule 5, including an oversize heading) splits into pieces that each close their own group.\nfunction groupScope(scopeBlocks: Block[], lines: string[], resolved: ResolvedOptions): (Chunk & { final?: boolean })[] {\n const working = resolved.targetTokens;\n const trigger = working * OVERSIZE_TRIGGER_MULTIPLE;\n const finished: (Chunk & { final?: boolean })[] = [];\n let parts: Part[] = [];\n let paragraphCount = 0;\n let tokens = 0;\n\n function close(): void {\n const group = finalize(parts);\n if (group) finished.push(group);\n parts = [];\n paragraphCount = 0;\n tokens = 0;\n }\n\n // tokens tracks the active text mode's own estimate (a newline between parts costs\n // NEWLINE_TOKENS too), so packing decisions size the text the chunk will actually ship as.\n function addPart(text: string, sizeText: string, startLine: number, endLine: number): void {\n tokens += (parts.length > 0 ? NEWLINE_TOKENS : 0) + estimateTokens(sizeText);\n parts.push({ startLine, endLine, text });\n }\n\n for (const block of scopeBlocks) {\n const raw = lines.slice(block.startLine - 1, block.endLine).join('\\n');\n const blockTokens = estimateTokens(raw);\n\n if (blockTokens > trigger) {\n const seed = parts.length > 0 ? tokens : 0;\n for (const p of splitOversizeBlock(lines, block.startLine, block.endLine, working, block.type, resolved.text, seed)) {\n parts.push(p);\n close();\n }\n continue;\n }\n\n const extracted = extractText(block.node);\n const sizeText = resolved.text === 'raw' ? raw : extracted;\n\n if (block.type === 'heading') {\n addPart(extracted, sizeText, block.startLine, block.endLine);\n continue;\n }\n\n // The 2x-working invariant holds even under pgc's paper-faithful 2-paragraph pairing --\n // close first if the pair about to form would cross it.\n const pairOversize = parts.length > 0 && tokens + NEWLINE_TOKENS + blockTokens > trigger;\n if (pairOversize) close();\n\n addPart(extracted, sizeText, block.startLine, block.endLine);\n paragraphCount++;\n\n if (paragraphCount >= PGC_GROUP_SIZE) close();\n }\n close();\n\n return finished;\n}\n\n// Groups already-parsed blocks per opts (D1/D3), against the same body the blocks were parsed\n// from (line lookups for oversize splitting).\nexport function group(blocks: Block[], body: string, opts?: ChunkOptions): Chunk[] {\n const resolved = resolveOptions(opts);\n const lines = body.split('\\n');\n const chunks: (Chunk & { final?: boolean })[] = [];\n for (const scope of splitScopes(blocks)) chunks.push(...groupScope(scope, lines, resolved));\n // 'raw': the chunk's own source lines verbatim, replacing the flavor-resolved join above (D9).\n // A `final` chunk already carries its own slice's raw text; re-slicing by extent would return the whole shared line.\n const texted =\n resolved.text === 'raw'\n ? chunks.map((c) =>\n c.final\n ? c\n : {\n ...c,\n text: lines\n .slice(c.startLine - 1, c.endLine)\n .join('\\n')\n .trim(),\n }\n )\n : chunks;\n // A group can be all-blank (flavor-stripped to nothing, or a raw slice of pure syntax); it never produces a chunk.\n return texted.filter((c) => c.text.trim().length > 0).map((c) => ({ startLine: c.startLine, endLine: c.endLine, text: c.text }));\n}\n"],"names":["extractText","parse","DEFAULT_TARGET_TOKENS","estimateTokens","PGC_GROUP_SIZE","OVERSIZE_TRIGGER_MULTIPLE","resolveOptions","opts","targetTokens","text","splitScopes","blocks","scopes","current","block","type","length","push","finalize","parts","undefined","first","last","startLine","endLine","map","p","join","final","some","NEWLINE_TOKENS","SEGMENTERS","Map","segmentsOf","granularity","segmenter","get","Intl","Segmenter","set","Array","from","segment","s","pack","segments","working","groups","tokens","segmentTokens","splitLineText","sentences","out","flush","sentence","sentenceTokens","ATOMIC_TYPES","Set","piece","pieceLines","blockType","textMode","has","b","node","finalizePiece","splitOversizeBlock","lines","seed","pieces","pieceStart","line","lineText","sep","lineTokens","groupScope","scopeBlocks","resolved","trigger","finished","paragraphCount","close","group","addPart","sizeText","raw","slice","blockTokens","extracted","pairOversize","body","split","chunks","scope","texted","c","trim","filter"],"mappings":"AAAA,SAASA,WAAW,QAAQ,eAAe;AAC3C,SAASC,KAAK,QAAQ,aAAa;AACnC,SAASC,qBAAqB,EAAEC,cAAc,QAAQ,cAAc;AAGpE,MAAMC,iBAAiB;AACvB,MAAMC,4BAA4B;AAOlC,SAASC,eAAeC,IAAmB;;IACzC,OAAO;QACLC,YAAY,UAAED,iBAAAA,2BAAAA,KAAMC,YAAY,uCAAIN;QACpCO,IAAI,WAAEF,iBAAAA,2BAAAA,KAAME,IAAI,yCAAI;IACtB;AACF;AAEA,6FAA6F;AAC7F,0FAA0F;AAC1F,SAASC,YAAYC,MAAe;IAClC,MAAMC,SAAoB,EAAE;IAC5B,IAAIC,UAAmB,EAAE;IACzB,KAAK,MAAMC,SAASH,OAAQ;QAC1B,IAAIG,MAAMC,IAAI,KAAK,aAAaF,QAAQG,MAAM,GAAG,GAAG;YAClDJ,OAAOK,IAAI,CAACJ;YACZA,UAAU,EAAE;QACd;QACAA,QAAQI,IAAI,CAACH;IACf;IACA,IAAID,QAAQG,MAAM,GAAG,GAAGJ,OAAOK,IAAI,CAACJ;IACpC,OAAOD;AACT;AAWA,SAASM,SAASC,KAAa;IAC7B,IAAIA,MAAMH,MAAM,KAAK,GAAG,OAAOI;IAC/B,MAAMC,QAAQF,KAAK,CAAC,EAAE;IACtB,MAAMG,OAAOH,KAAK,CAACA,MAAMH,MAAM,GAAG,EAAE;IACpC,OAAO;QAAEO,WAAWF,MAAME,SAAS;QAAEC,SAASF,KAAKE,OAAO;QAAEf,MAAMU,MAAMM,GAAG,CAAC,CAACC,IAAMA,EAAEjB,IAAI,EAAEkB,IAAI,CAAC;QAAOC,OAAOT,MAAMU,IAAI,CAAC,CAACH,IAAMA,EAAEE,KAAK;IAAE;AAC3I;AAEA,MAAME,iBAAiB3B,eAAe;AACtC,gGAAgG;AAChG,8DAA8D;AAC9D,MAAM4B,aAAa,IAAIC;AAEvB,SAASC,WAAWxB,IAAY,EAAEyB,WAAgC;IAChE,IAAIC,YAAYJ,WAAWK,GAAG,CAACF;IAC/B,IAAI,CAACC,WAAW;QACdA,YAAY,IAAIE,KAAKC,SAAS,CAAClB,WAAW;YAAEc;QAAY;QACxDH,WAAWQ,GAAG,CAACL,aAAaC;IAC9B;IACA,OAAOK,MAAMC,IAAI,CAACN,UAAUO,OAAO,CAACjC,OAAO,CAACkC,IAAMA,EAAED,OAAO;AAC7D;AAEA,gGAAgG;AAChG,2FAA2F;AAC3F,SAASE,KAAKC,QAAkB,EAAEC,OAAe;IAC/C,MAAMC,SAAmB,EAAE;IAC3B,IAAIlC,UAAU;IACd,IAAImC,SAAS;IACb,KAAK,MAAMN,WAAWG,SAAU;QAC9B,MAAMI,gBAAgB9C,eAAeuC;QACrC,IAAI7B,QAAQG,MAAM,GAAG,KAAKgC,SAASC,gBAAgBH,SAAS;YAC1DC,OAAO9B,IAAI,CAACJ;YACZA,UAAU;YACVmC,SAAS;QACX;QACAnC,WAAW6B;QACXM,UAAUC;IACZ;IACA,IAAIpC,QAAQG,MAAM,GAAG,GAAG+B,OAAO9B,IAAI,CAACJ;IACpC,OAAOkC;AACT;AAEA,8FAA8F;AAC9F,0GAA0G;AAC1G,SAASG,cAAczC,IAAY,EAAEqC,OAAe;IAClD,MAAMK,YAAYlB,WAAWxB,MAAM;IACnC,MAAM2C,MAAgB,EAAE;IACxB,IAAIvC,UAAU;IACd,IAAImC,SAAS;IACb,MAAMK,QAAQ;QACZ,IAAIxC,QAAQG,MAAM,GAAG,GAAG;YACtBoC,IAAInC,IAAI,CAACJ;YACTA,UAAU;YACVmC,SAAS;QACX;IACF;IACA,KAAK,MAAMM,YAAYH,UAAW;QAChC,MAAMI,iBAAiBpD,eAAemD;QACtC,IAAIC,iBAAiBT,SAAS;YAC5BO;YACAD,IAAInC,IAAI,IAAI2B,KAAKX,WAAWqB,UAAU,SAASR;YAC/C;QACF;QACA,IAAIjC,QAAQG,MAAM,GAAG,KAAKgC,SAASO,iBAAiBT,SAASO;QAC7DxC,WAAWyC;QACXN,UAAUO;IACZ;IACAF;IACA,OAAOD;AACT;AAEA,MAAMI,eAAuC,IAAIC,IAAI;IAAC;IAAQ;IAAS;CAAO;AAE9E,iGAAiG;AACjG,2EAA2E;AAC3E,SAASC,MAAMC,UAAoB,EAAEpC,SAAiB,EAAEC,OAAe,EAAEoC,SAAoB,EAAEC,QAA6B;IAC1H,MAAMpD,OACJ+C,aAAaM,GAAG,CAACF,cAAcC,aAAa,QACxCF,WAAWhC,IAAI,CAAC,QAChB1B,MAAM0D,WAAWhC,IAAI,CAAC,OACnBF,GAAG,CAAC,CAACsC,IAAM/D,YAAY+D,EAAEC,IAAI,GAC7BrC,IAAI,CAAC;IACd,OAAO;QAAEJ;QAAWC;QAASf;IAAK;AACpC;AAEA,kGAAkG;AAClG,yGAAyG;AACzG,SAASwD,cAAcN,UAAoB,EAAEpC,SAAiB,EAAEC,OAAe,EAAEsB,OAAe,EAAEc,SAAoB,EAAEC,QAA6B;IACnJ,MAAMnC,IAAIgC,MAAMC,YAAYpC,WAAWC,SAASoC,WAAWC;IAC3D,IAAIF,WAAW3C,MAAM,KAAK,KAAKb,eAAeuB,EAAEjB,IAAI,IAAIqC,SAAS;QAC/D,OAAOI,cAAcxB,EAAEjB,IAAI,EAAEqC,SAASrB,GAAG,CAAC,CAAChB,OAAU,CAAA;gBAAEc;gBAAWC;gBAASf;gBAAMmB,OAAO;YAAK,CAAA;IAC/F;IACA,OAAO;QAACF;KAAE;AACZ;AAEA,iGAAiG;AACjG,0GAA0G;AAC1G,SAASwC,mBAAmBC,KAAe,EAAE5C,SAAiB,EAAEC,OAAe,EAAEsB,OAAe,EAAEc,SAAoB,EAAEC,QAA6B,EAAEO,OAAO,CAAC;IAC7J,MAAMC,SAAiB,EAAE;IACzB,IAAIV,aAAuB,EAAE;IAC7B,IAAIW,aAAa/C;IACjB,IAAIyB,SAASoB;IACb,IAAK,IAAIG,OAAOhD,WAAWgD,QAAQ/C,SAAS+C,OAAQ;QAClD,MAAMC,WAAWL,KAAK,CAACI,OAAO,EAAE;QAChC,MAAME,MAAMd,WAAW3C,MAAM,GAAG,KAAKgC,SAAS,IAAIlB,iBAAiB;QACnE,MAAM4C,aAAavE,eAAeqE;QAClC,IAAIb,WAAW3C,MAAM,GAAG,KAAKgC,SAASyB,MAAMC,aAAa5B,SAAS;YAChEuB,OAAOpD,IAAI,IAAIgD,cAAcN,YAAYW,YAAYC,OAAO,GAAGzB,SAASc,WAAWC;YACnFF,aAAa,EAAE;YACfX,SAAS;YACTsB,aAAaC;YACbZ,WAAW1C,IAAI,CAACuD;YAChBxB,UAAU0B;YACV;QACF;QACAf,WAAW1C,IAAI,CAACuD;QAChBxB,UAAUyB,MAAMC;IAClB;IACA,IAAIf,WAAW3C,MAAM,GAAG,GAAGqD,OAAOpD,IAAI,IAAIgD,cAAcN,YAAYW,YAAY9C,SAASsB,SAASc,WAAWC;IAC7G,OAAOQ;AACT;AAEA,0FAA0F;AAC1F,8FAA8F;AAC9F,SAASM,WAAWC,WAAoB,EAAET,KAAe,EAAEU,QAAyB;IAClF,MAAM/B,UAAU+B,SAASrE,YAAY;IACrC,MAAMsE,UAAUhC,UAAUzC;IAC1B,MAAM0E,WAA4C,EAAE;IACpD,IAAI5D,QAAgB,EAAE;IACtB,IAAI6D,iBAAiB;IACrB,IAAIhC,SAAS;IAEb,SAASiC;QACP,MAAMC,QAAQhE,SAASC;QACvB,IAAI+D,OAAOH,SAAS9D,IAAI,CAACiE;QACzB/D,QAAQ,EAAE;QACV6D,iBAAiB;QACjBhC,SAAS;IACX;IAEA,mFAAmF;IACnF,2FAA2F;IAC3F,SAASmC,QAAQ1E,IAAY,EAAE2E,QAAgB,EAAE7D,SAAiB,EAAEC,OAAe;QACjFwB,UAAU,AAAC7B,CAAAA,MAAMH,MAAM,GAAG,IAAIc,iBAAiB,CAAA,IAAK3B,eAAeiF;QACnEjE,MAAMF,IAAI,CAAC;YAAEM;YAAWC;YAASf;QAAK;IACxC;IAEA,KAAK,MAAMK,SAAS8D,YAAa;QAC/B,MAAMS,MAAMlB,MAAMmB,KAAK,CAACxE,MAAMS,SAAS,GAAG,GAAGT,MAAMU,OAAO,EAAEG,IAAI,CAAC;QACjE,MAAM4D,cAAcpF,eAAekF;QAEnC,IAAIE,cAAcT,SAAS;YACzB,MAAMV,OAAOjD,MAAMH,MAAM,GAAG,IAAIgC,SAAS;YACzC,KAAK,MAAMtB,KAAKwC,mBAAmBC,OAAOrD,MAAMS,SAAS,EAAET,MAAMU,OAAO,EAAEsB,SAAShC,MAAMC,IAAI,EAAE8D,SAASpE,IAAI,EAAE2D,MAAO;gBACnHjD,MAAMF,IAAI,CAACS;gBACXuD;YACF;YACA;QACF;QAEA,MAAMO,YAAYxF,YAAYc,MAAMkD,IAAI;QACxC,MAAMoB,WAAWP,SAASpE,IAAI,KAAK,QAAQ4E,MAAMG;QAEjD,IAAI1E,MAAMC,IAAI,KAAK,WAAW;YAC5BoE,QAAQK,WAAWJ,UAAUtE,MAAMS,SAAS,EAAET,MAAMU,OAAO;YAC3D;QACF;QAEA,wFAAwF;QACxF,wDAAwD;QACxD,MAAMiE,eAAetE,MAAMH,MAAM,GAAG,KAAKgC,SAASlB,iBAAiByD,cAAcT;QACjF,IAAIW,cAAcR;QAElBE,QAAQK,WAAWJ,UAAUtE,MAAMS,SAAS,EAAET,MAAMU,OAAO;QAC3DwD;QAEA,IAAIA,kBAAkB5E,gBAAgB6E;IACxC;IACAA;IAEA,OAAOF;AACT;AAEA,8FAA8F;AAC9F,8CAA8C;AAC9C,OAAO,SAASG,MAAMvE,MAAe,EAAE+E,IAAY,EAAEnF,IAAmB;IACtE,MAAMsE,WAAWvE,eAAeC;IAChC,MAAM4D,QAAQuB,KAAKC,KAAK,CAAC;IACzB,MAAMC,SAA0C,EAAE;IAClD,KAAK,MAAMC,SAASnF,YAAYC,QAASiF,OAAO3E,IAAI,IAAI0D,WAAWkB,OAAO1B,OAAOU;IACjF,+FAA+F;IAC/F,qHAAqH;IACrH,MAAMiB,SACJjB,SAASpE,IAAI,KAAK,QACdmF,OAAOnE,GAAG,CAAC,CAACsE,IACVA,EAAEnE,KAAK,GACHmE,IACA;YACE,GAAGA,CAAC;YACJtF,MAAM0D,MACHmB,KAAK,CAACS,EAAExE,SAAS,GAAG,GAAGwE,EAAEvE,OAAO,EAChCG,IAAI,CAAC,MACLqE,IAAI;QACT,KAENJ;IACN,mHAAmH;IACnH,OAAOE,OAAOG,MAAM,CAAC,CAACF,IAAMA,EAAEtF,IAAI,CAACuF,IAAI,GAAGhF,MAAM,GAAG,GAAGS,GAAG,CAAC,CAACsE,IAAO,CAAA;YAAExE,WAAWwE,EAAExE,SAAS;YAAEC,SAASuE,EAAEvE,OAAO;YAAEf,MAAMsF,EAAEtF,IAAI;QAAC,CAAA;AAC/H"}
|
|
@@ -2,7 +2,7 @@ import type { Block, Chunk, ChunkOptions } from './types.js';
|
|
|
2
2
|
export declare function chunkFromBlocks(blocks: Block[], body: string, opts?: ChunkOptions): Chunk[];
|
|
3
3
|
export declare function chunk(body: string, opts?: ChunkOptions): Chunk[];
|
|
4
4
|
export { extractText, extractTexts } from './extract.js';
|
|
5
|
-
export { DEFAULT_TARGET_TOKENS, estimateTokens } from './group.js';
|
|
6
5
|
export { parse } from './parse.js';
|
|
6
|
+
export { DEFAULT_TARGET_TOKENS, estimateTokens } from './tokens.js';
|
|
7
7
|
export type { Block, BlockType, Chunk, ChunkOptions } from './types.js';
|
|
8
8
|
export { CHUNK_VERSION } from './version.js';
|
package/dist/esm/chunk/index.js
CHANGED
|
@@ -11,6 +11,6 @@ export function chunk(body, opts) {
|
|
|
11
11
|
return chunkFromBlocks(parse(body), body, opts);
|
|
12
12
|
}
|
|
13
13
|
export { extractText, extractTexts } from './extract.js';
|
|
14
|
-
export { DEFAULT_TARGET_TOKENS, estimateTokens } from './group.js';
|
|
15
14
|
export { parse } from './parse.js';
|
|
15
|
+
export { DEFAULT_TARGET_TOKENS, estimateTokens } from './tokens.js';
|
|
16
16
|
export { CHUNK_VERSION } from './version.js';
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"sources":["/Users/kevin/Dev/OpenSource/ai/sensemaking/src/chunk/index.ts"],"sourcesContent":["import { group } from './group.ts';\nimport { parse } from './parse.ts';\nimport type { Block, Chunk, ChunkOptions } from './types.ts';\n\n// Groups blocks a caller already parsed (e.g. scan/index.ts, sharing one parse with the FTS\n// text path) against the same body they came from, per opts (D1/D3).\nexport function chunkFromBlocks(blocks: Block[], body: string, opts?: ChunkOptions): Chunk[] {\n return group(blocks, body, opts);\n}\n\n// Pure, deterministic function of file content; chunk semantics are version-stamped via\n// CHUNK_VERSION (./version.ts). Algorithm and evidence: BENCHMARKING.md, \"The chunking algorithm\".\nexport function chunk(body: string, opts?: ChunkOptions): Chunk[] {\n return chunkFromBlocks(parse(body), body, opts);\n}\n\nexport { extractText, extractTexts } from './extract.ts';\nexport {
|
|
1
|
+
{"version":3,"sources":["/Users/kevin/Dev/OpenSource/ai/sensemaking/src/chunk/index.ts"],"sourcesContent":["import { group } from './group.ts';\nimport { parse } from './parse.ts';\nimport type { Block, Chunk, ChunkOptions } from './types.ts';\n\n// Groups blocks a caller already parsed (e.g. scan/index.ts, sharing one parse with the FTS\n// text path) against the same body they came from, per opts (D1/D3).\nexport function chunkFromBlocks(blocks: Block[], body: string, opts?: ChunkOptions): Chunk[] {\n return group(blocks, body, opts);\n}\n\n// Pure, deterministic function of file content; chunk semantics are version-stamped via\n// CHUNK_VERSION (./version.ts). Algorithm and evidence: BENCHMARKING.md, \"The chunking algorithm\".\nexport function chunk(body: string, opts?: ChunkOptions): Chunk[] {\n return chunkFromBlocks(parse(body), body, opts);\n}\n\nexport { extractText, extractTexts } from './extract.ts';\nexport { parse } from './parse.ts';\nexport { DEFAULT_TARGET_TOKENS, estimateTokens } from './tokens.ts';\nexport type { Block, BlockType, Chunk, ChunkOptions } from './types.ts';\nexport { CHUNK_VERSION } from './version.ts';\n"],"names":["group","parse","chunkFromBlocks","blocks","body","opts","chunk","extractText","extractTexts","DEFAULT_TARGET_TOKENS","estimateTokens","CHUNK_VERSION"],"mappings":"AAAA,SAASA,KAAK,QAAQ,aAAa;AACnC,SAASC,KAAK,QAAQ,aAAa;AAGnC,4FAA4F;AAC5F,qEAAqE;AACrE,OAAO,SAASC,gBAAgBC,MAAe,EAAEC,IAAY,EAAEC,IAAmB;IAChF,OAAOL,MAAMG,QAAQC,MAAMC;AAC7B;AAEA,wFAAwF;AACxF,mGAAmG;AACnG,OAAO,SAASC,MAAMF,IAAY,EAAEC,IAAmB;IACrD,OAAOH,gBAAgBD,MAAMG,OAAOA,MAAMC;AAC5C;AAEA,SAASE,WAAW,EAAEC,YAAY,QAAQ,eAAe;AACzD,SAASP,KAAK,QAAQ,aAAa;AACnC,SAASQ,qBAAqB,EAAEC,cAAc,QAAQ,cAAc;AAEpE,SAASC,aAAa,QAAQ,eAAe"}
|
package/dist/esm/chunk/parse.js
CHANGED
|
@@ -1,15 +1,8 @@
|
|
|
1
|
-
import
|
|
2
|
-
import { gfmAutolinkLiteralFromMarkdown } from 'mdast-util-gfm-autolink-literal';
|
|
3
|
-
import { gfmFootnoteFromMarkdown } from 'mdast-util-gfm-footnote';
|
|
4
|
-
import { gfmStrikethroughFromMarkdown } from 'mdast-util-gfm-strikethrough';
|
|
5
|
-
import { gfmTableFromMarkdown } from 'mdast-util-gfm-table';
|
|
6
|
-
import { gfmTaskListItemFromMarkdown } from 'mdast-util-gfm-task-list-item';
|
|
7
|
-
import { gfmAutolinkLiteral } from 'micromark-extension-gfm-autolink-literal';
|
|
8
|
-
import { gfmFootnote } from 'micromark-extension-gfm-footnote';
|
|
9
|
-
import { gfmStrikethrough } from 'micromark-extension-gfm-strikethrough';
|
|
10
|
-
import { gfmTable } from 'micromark-extension-gfm-table';
|
|
11
|
-
import { gfmTaskListItem } from 'micromark-extension-gfm-task-list-item';
|
|
1
|
+
import Module from 'node:module';
|
|
12
2
|
import { extractText } from './extract.js';
|
|
3
|
+
// Tier-2, as embed/static.ts: the parser's packages cost ~19 ms to load and a warm tree never
|
|
4
|
+
// parses, so every store-opening command paid for them until a file actually changed.
|
|
5
|
+
const _require = typeof require === 'undefined' ? Module.createRequire(import.meta.url) : require;
|
|
13
6
|
const BLOCK_TYPES = {
|
|
14
7
|
heading: 'heading',
|
|
15
8
|
paragraph: 'paragraph',
|
|
@@ -18,29 +11,48 @@ const BLOCK_TYPES = {
|
|
|
18
11
|
list: 'list',
|
|
19
12
|
blockquote: 'blockquote'
|
|
20
13
|
};
|
|
14
|
+
let cached;
|
|
21
15
|
// Imported individually, not via micromark-extension-gfm/mdast-util-gfm: those bundles also pull
|
|
22
16
|
// in gfm-tagfilter, an HTML sanitizer this library never uses (no htmlExtensions call anywhere).
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
const
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
17
|
+
function parser() {
|
|
18
|
+
if (cached) return cached;
|
|
19
|
+
const { fromMarkdown } = _require('mdast-util-from-markdown');
|
|
20
|
+
const { gfmAutolinkLiteralFromMarkdown } = _require('mdast-util-gfm-autolink-literal');
|
|
21
|
+
const { gfmFootnoteFromMarkdown } = _require('mdast-util-gfm-footnote');
|
|
22
|
+
const { gfmStrikethroughFromMarkdown } = _require('mdast-util-gfm-strikethrough');
|
|
23
|
+
const { gfmTableFromMarkdown } = _require('mdast-util-gfm-table');
|
|
24
|
+
const { gfmTaskListItemFromMarkdown } = _require('mdast-util-gfm-task-list-item');
|
|
25
|
+
const { gfmAutolinkLiteral } = _require('micromark-extension-gfm-autolink-literal');
|
|
26
|
+
const { gfmFootnote } = _require('micromark-extension-gfm-footnote');
|
|
27
|
+
const { gfmStrikethrough } = _require('micromark-extension-gfm-strikethrough');
|
|
28
|
+
const { gfmTable } = _require('micromark-extension-gfm-table');
|
|
29
|
+
const { gfmTaskListItem } = _require('micromark-extension-gfm-task-list-item');
|
|
30
|
+
cached = {
|
|
31
|
+
fromMarkdown,
|
|
32
|
+
options: {
|
|
33
|
+
extensions: [
|
|
34
|
+
gfmAutolinkLiteral(),
|
|
35
|
+
gfmFootnote(),
|
|
36
|
+
gfmStrikethrough(),
|
|
37
|
+
gfmTable(),
|
|
38
|
+
gfmTaskListItem()
|
|
39
|
+
],
|
|
40
|
+
mdastExtensions: [
|
|
41
|
+
gfmAutolinkLiteralFromMarkdown(),
|
|
42
|
+
gfmFootnoteFromMarkdown(),
|
|
43
|
+
gfmStrikethroughFromMarkdown(),
|
|
44
|
+
gfmTableFromMarkdown(),
|
|
45
|
+
gfmTaskListItemFromMarkdown()
|
|
46
|
+
]
|
|
47
|
+
}
|
|
48
|
+
};
|
|
49
|
+
return cached;
|
|
50
|
+
}
|
|
37
51
|
// Top-level blocks of a markdown body, typed and line-extent bounded from mdast's own
|
|
38
52
|
// node.position (never a regex guess). GFM extensions add tables, task lists, footnotes, strikethrough.
|
|
39
53
|
export function parse(body) {
|
|
40
|
-
const
|
|
41
|
-
|
|
42
|
-
mdastExtensions: MDAST_EXTENSIONS
|
|
43
|
-
});
|
|
54
|
+
const { fromMarkdown, options } = parser();
|
|
55
|
+
const tree = fromMarkdown(body, options);
|
|
44
56
|
return tree.children.map((node)=>{
|
|
45
57
|
var _BLOCK_TYPES_node_type;
|
|
46
58
|
const position = node.position;
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"sources":["/Users/kevin/Dev/OpenSource/ai/sensemaking/src/chunk/parse.ts"],"sourcesContent":["import type { RootContent } from 'mdast';\nimport { fromMarkdown } from 'mdast-util-from-markdown';\
|
|
1
|
+
{"version":3,"sources":["/Users/kevin/Dev/OpenSource/ai/sensemaking/src/chunk/parse.ts"],"sourcesContent":["import Module from 'node:module';\nimport type { RootContent } from 'mdast';\nimport { extractText } from './extract.ts';\nimport type { Block, BlockType } from './types.ts';\n\n// Tier-2, as embed/static.ts: the parser's packages cost ~19 ms to load and a warm tree never\n// parses, so every store-opening command paid for them until a file actually changed.\nconst _require = typeof require === 'undefined' ? Module.createRequire(import.meta.url) : require;\n\nconst BLOCK_TYPES: Partial<Record<RootContent['type'], BlockType>> = {\n heading: 'heading',\n paragraph: 'paragraph',\n code: 'code',\n table: 'table',\n list: 'list',\n blockquote: 'blockquote',\n};\n\ntype FromMarkdown = typeof import('mdast-util-from-markdown').fromMarkdown;\ntype Parser = { fromMarkdown: FromMarkdown; options: NonNullable<Parameters<FromMarkdown>[1]> };\nlet cached: Parser | undefined;\n\n// Imported individually, not via micromark-extension-gfm/mdast-util-gfm: those bundles also pull\n// in gfm-tagfilter, an HTML sanitizer this library never uses (no htmlExtensions call anywhere).\nfunction parser(): Parser {\n if (cached) return cached;\n const { fromMarkdown } = _require('mdast-util-from-markdown') as typeof import('mdast-util-from-markdown');\n const { gfmAutolinkLiteralFromMarkdown } = _require('mdast-util-gfm-autolink-literal') as typeof import('mdast-util-gfm-autolink-literal');\n const { gfmFootnoteFromMarkdown } = _require('mdast-util-gfm-footnote') as typeof import('mdast-util-gfm-footnote');\n const { gfmStrikethroughFromMarkdown } = _require('mdast-util-gfm-strikethrough') as typeof import('mdast-util-gfm-strikethrough');\n const { gfmTableFromMarkdown } = _require('mdast-util-gfm-table') as typeof import('mdast-util-gfm-table');\n const { gfmTaskListItemFromMarkdown } = _require('mdast-util-gfm-task-list-item') as typeof import('mdast-util-gfm-task-list-item');\n const { gfmAutolinkLiteral } = _require('micromark-extension-gfm-autolink-literal') as typeof import('micromark-extension-gfm-autolink-literal');\n const { gfmFootnote } = _require('micromark-extension-gfm-footnote') as typeof import('micromark-extension-gfm-footnote');\n const { gfmStrikethrough } = _require('micromark-extension-gfm-strikethrough') as typeof import('micromark-extension-gfm-strikethrough');\n const { gfmTable } = _require('micromark-extension-gfm-table') as typeof import('micromark-extension-gfm-table');\n const { gfmTaskListItem } = _require('micromark-extension-gfm-task-list-item') as typeof import('micromark-extension-gfm-task-list-item');\n cached = {\n fromMarkdown,\n options: {\n extensions: [gfmAutolinkLiteral(), gfmFootnote(), gfmStrikethrough(), gfmTable(), gfmTaskListItem()],\n mdastExtensions: [gfmAutolinkLiteralFromMarkdown(), gfmFootnoteFromMarkdown(), gfmStrikethroughFromMarkdown(), gfmTableFromMarkdown(), gfmTaskListItemFromMarkdown()],\n },\n };\n return cached;\n}\n\n// Top-level blocks of a markdown body, typed and line-extent bounded from mdast's own\n// node.position (never a regex guess). GFM extensions add tables, task lists, footnotes, strikethrough.\nexport function parse(body: string): Block[] {\n const { fromMarkdown, options } = parser();\n const tree = fromMarkdown(body, options);\n return tree.children.map((node) => {\n const position = node.position;\n const block: Block = {\n type: BLOCK_TYPES[node.type] ?? 'other',\n startLine: position ? position.start.line : 1,\n endLine: position ? position.end.line : 1,\n node,\n };\n if (node.type === 'heading') {\n block.depth = node.depth;\n block.text = extractText(node);\n }\n return block;\n });\n}\n"],"names":["Module","extractText","_require","require","createRequire","url","BLOCK_TYPES","heading","paragraph","code","table","list","blockquote","cached","parser","fromMarkdown","gfmAutolinkLiteralFromMarkdown","gfmFootnoteFromMarkdown","gfmStrikethroughFromMarkdown","gfmTableFromMarkdown","gfmTaskListItemFromMarkdown","gfmAutolinkLiteral","gfmFootnote","gfmStrikethrough","gfmTable","gfmTaskListItem","options","extensions","mdastExtensions","parse","body","tree","children","map","node","position","block","type","startLine","start","line","endLine","end","depth","text"],"mappings":"AAAA,OAAOA,YAAY,cAAc;AAEjC,SAASC,WAAW,QAAQ,eAAe;AAG3C,8FAA8F;AAC9F,sFAAsF;AACtF,MAAMC,WAAW,OAAOC,YAAY,cAAcH,OAAOI,aAAa,CAAC,YAAYC,GAAG,IAAIF;AAE1F,MAAMG,cAA+D;IACnEC,SAAS;IACTC,WAAW;IACXC,MAAM;IACNC,OAAO;IACPC,MAAM;IACNC,YAAY;AACd;AAIA,IAAIC;AAEJ,iGAAiG;AACjG,iGAAiG;AACjG,SAASC;IACP,IAAID,QAAQ,OAAOA;IACnB,MAAM,EAAEE,YAAY,EAAE,GAAGb,SAAS;IAClC,MAAM,EAAEc,8BAA8B,EAAE,GAAGd,SAAS;IACpD,MAAM,EAAEe,uBAAuB,EAAE,GAAGf,SAAS;IAC7C,MAAM,EAAEgB,4BAA4B,EAAE,GAAGhB,SAAS;IAClD,MAAM,EAAEiB,oBAAoB,EAAE,GAAGjB,SAAS;IAC1C,MAAM,EAAEkB,2BAA2B,EAAE,GAAGlB,SAAS;IACjD,MAAM,EAAEmB,kBAAkB,EAAE,GAAGnB,SAAS;IACxC,MAAM,EAAEoB,WAAW,EAAE,GAAGpB,SAAS;IACjC,MAAM,EAAEqB,gBAAgB,EAAE,GAAGrB,SAAS;IACtC,MAAM,EAAEsB,QAAQ,EAAE,GAAGtB,SAAS;IAC9B,MAAM,EAAEuB,eAAe,EAAE,GAAGvB,SAAS;IACrCW,SAAS;QACPE;QACAW,SAAS;YACPC,YAAY;gBAACN;gBAAsBC;gBAAeC;gBAAoBC;gBAAYC;aAAkB;YACpGG,iBAAiB;gBAACZ;gBAAkCC;gBAA2BC;gBAAgCC;gBAAwBC;aAA8B;QACvK;IACF;IACA,OAAOP;AACT;AAEA,sFAAsF;AACtF,wGAAwG;AACxG,OAAO,SAASgB,MAAMC,IAAY;IAChC,MAAM,EAAEf,YAAY,EAAEW,OAAO,EAAE,GAAGZ;IAClC,MAAMiB,OAAOhB,aAAae,MAAMJ;IAChC,OAAOK,KAAKC,QAAQ,CAACC,GAAG,CAAC,CAACC;YAGhB5B;QAFR,MAAM6B,WAAWD,KAAKC,QAAQ;QAC9B,MAAMC,QAAe;YACnBC,IAAI,GAAE/B,yBAAAA,WAAW,CAAC4B,KAAKG,IAAI,CAAC,cAAtB/B,oCAAAA,yBAA0B;YAChCgC,WAAWH,WAAWA,SAASI,KAAK,CAACC,IAAI,GAAG;YAC5CC,SAASN,WAAWA,SAASO,GAAG,CAACF,IAAI,GAAG;YACxCN;QACF;QACA,IAAIA,KAAKG,IAAI,KAAK,WAAW;YAC3BD,MAAMO,KAAK,GAAGT,KAAKS,KAAK;YACxBP,MAAMQ,IAAI,GAAG3C,YAAYiC;QAC3B;QACA,OAAOE;IACT;AACF"}
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
import { UNSPACED_SCRIPTS } from '../text/segment.js';
|
|
2
|
+
export const DEFAULT_TARGET_TOKENS = 500;
|
|
3
|
+
// D5: segment.ts's unspaced-script set packs close to one token per character; Hangul is
|
|
4
|
+
// spaced but still token-dense and sits in this set as a conservative bound (one Unicode set, one home).
|
|
5
|
+
const DENSE_SCRIPT = new RegExp(`[${UNSPACED_SCRIPTS}\\p{scx=Hangul}]`, 'u');
|
|
6
|
+
// D5's size estimate: dense-script graphemes 1:1, everything else at 4 chars/token.
|
|
7
|
+
export function estimateTokens(text) {
|
|
8
|
+
let dense = 0;
|
|
9
|
+
let other = 0;
|
|
10
|
+
for (const ch of text){
|
|
11
|
+
if (DENSE_SCRIPT.test(ch)) dense++;
|
|
12
|
+
else other++;
|
|
13
|
+
}
|
|
14
|
+
return dense + other / 4;
|
|
15
|
+
}
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"sources":["/Users/kevin/Dev/OpenSource/ai/sensemaking/src/chunk/tokens.ts"],"sourcesContent":["import { UNSPACED_SCRIPTS } from '../text/segment.ts';\n\nexport const DEFAULT_TARGET_TOKENS = 500;\n\n// D5: segment.ts's unspaced-script set packs close to one token per character; Hangul is\n// spaced but still token-dense and sits in this set as a conservative bound (one Unicode set, one home).\nconst DENSE_SCRIPT = new RegExp(`[${UNSPACED_SCRIPTS}\\\\p{scx=Hangul}]`, 'u');\n\n// D5's size estimate: dense-script graphemes 1:1, everything else at 4 chars/token.\nexport function estimateTokens(text: string): number {\n let dense = 0;\n let other = 0;\n for (const ch of text) {\n if (DENSE_SCRIPT.test(ch)) dense++;\n else other++;\n }\n return dense + other / 4;\n}\n"],"names":["UNSPACED_SCRIPTS","DEFAULT_TARGET_TOKENS","DENSE_SCRIPT","RegExp","estimateTokens","text","dense","other","ch","test"],"mappings":"AAAA,SAASA,gBAAgB,QAAQ,qBAAqB;AAEtD,OAAO,MAAMC,wBAAwB,IAAI;AAEzC,yFAAyF;AACzF,yGAAyG;AACzG,MAAMC,eAAe,IAAIC,OAAO,CAAC,CAAC,EAAEH,iBAAiB,gBAAgB,CAAC,EAAE;AAExE,oFAAoF;AACpF,OAAO,SAASI,eAAeC,IAAY;IACzC,IAAIC,QAAQ;IACZ,IAAIC,QAAQ;IACZ,KAAK,MAAMC,MAAMH,KAAM;QACrB,IAAIH,aAAaO,IAAI,CAACD,KAAKF;aACtBC;IACP;IACA,OAAOD,QAAQC,QAAQ;AACzB"}
|
|
@@ -7,7 +7,7 @@ import { searchError } from '../output/search-error.js';
|
|
|
7
7
|
import { materializeScope, narrowByWhere, rawScope, scopeHasEmbeddings } from './scope.js';
|
|
8
8
|
import { linksCandidates, vectorsCandidates, wordsCandidates } from './signals.js';
|
|
9
9
|
// snippet() re-tokenizes each candidate doc, superlinearly: ~10s for one 1MB doc
|
|
10
|
-
// (benchmark/reports/2026-08-23-hub-release-battery.md). Past this bound, rows get the JS excerpt.
|
|
10
|
+
// (benchmark/reports/2026-08-23-0.13.2-hub-release-battery.md). Past this bound, rows get the JS excerpt.
|
|
11
11
|
const EXCERPT_WINDOW = 160;
|
|
12
12
|
// Bare terms from an FTS5 query string: strips operators/quoting so the oversized-doc excerpt
|
|
13
13
|
// scan matches the same words the query matched on, not FTS5 syntax.
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"sources":["/Users/kevin/Dev/OpenSource/ai/sensemaking/src/commands/search.ts"],"sourcesContent":["import { readFileSync } from 'node:fs';\nimport { join } from 'node:path';\nimport type { ResolvedConfig } from '../config/index.ts';\nimport { embedConfig, featureEnabled, resolveSearch } from '../config/index.ts';\nimport { localModelMissing, MODEL_FILENAMES } from '../embed/store.ts';\nimport { SenseError } from '../errors.ts';\nimport type { Row } from '../output/output.ts';\nimport { searchError } from '../output/search-error.ts';\nimport type { LexicalHit, Store } from '../store/types.ts';\nimport { materializeScope, narrowByWhere, rawScope, scopeHasEmbeddings } from './scope.ts';\nimport { linksCandidates, vectorsCandidates, wordsCandidates } from './signals.ts';\n\n// snippet() re-tokenizes each candidate doc, superlinearly: ~10s for one 1MB doc\n// (benchmark/reports/2026-08-23-hub-release-battery.md). Past this bound, rows get the JS excerpt.\nconst EXCERPT_WINDOW = 160;\n\n// Bare terms from an FTS5 query string: strips operators/quoting so the oversized-doc excerpt\n// scan matches the same words the query matched on, not FTS5 syntax.\nfunction extractBareTerms(query: string): string[] {\n const cleaned = query\n .replace(/\"/g, ' ')\n .replace(/[()*]/g, ' ')\n .replace(/\\b(AND|OR|NOT|NEAR)\\b(\\/\\d+)?/gi, ' ');\n return cleaned\n .split(/\\s+/)\n .map((tok) => tok.replace(/^[A-Za-z_]\\w*:/, '')) // column filter, e.g. title:term\n .map((tok) => tok.toLowerCase().trim())\n .filter((tok) => tok.length > 0);\n}\n\nfunction findOccurrences(haystackLower: string, terms: string[]): Array<{ start: number; end: number; term: string }> {\n const occ: Array<{ start: number; end: number; term: string }> = [];\n for (const term of terms) {\n let idx = 0;\n for (;;) {\n const found = haystackLower.indexOf(term, idx);\n if (found === -1) break;\n occ.push({ start: found, end: found + term.length, term });\n idx = found + term.length;\n }\n }\n // Longest span first at equal start, so \"tests\" beats its substring \"test\" and the\n // whole word gets highlighted; the emit loop then absorbs the shorter overlap.\n return occ.sort((a, b) => a.start - b.start || b.end - a.end);\n}\n\n// Slides a window over the sorted occurrences, same semantics as FTS5's own best-window\n// pick: most distinct terms, ties broken by most total hits.\nfunction bestWindowStart(occ: Array<{ start: number; end: number; term: string }>, windowSize: number): number {\n let best = occ[0].start;\n let bestDistinct = 0;\n let bestCount = 0;\n for (let i = 0; i < occ.length; i++) {\n const winEnd = occ[i].start + windowSize;\n const seen = new Set<string>();\n let count = 0;\n for (let j = i; j < occ.length && occ[j].start < winEnd; j++) {\n seen.add(occ[j].term);\n count++;\n }\n if (seen.size > bestDistinct || (seen.size === bestDistinct && count > bestCount)) {\n best = occ[i].start;\n bestDistinct = seen.size;\n bestCount = count;\n }\n }\n return best;\n}\n\n// snippet()-shaped excerpt: matched terms wrapped in «», … at cut edges. Linear in doc\n// length, only run for rows actually returned (<= k), unlike snippet()'s per-row cost.\nfunction computeExcerpt(text: string, terms: string[]): { excerpt: string; offset: number } {\n const occ = findOccurrences(text.toLowerCase(), terms);\n if (occ.length === 0) {\n // Raw substring scan, not porter-stemmed: a doc matched only through a stemmed variant\n // finds no occurrence here, so fall back to the doc's start, unmarked.\n const end = Math.min(text.length, EXCERPT_WINDOW);\n return { excerpt: `${text.slice(0, end).replace(/\\s+/g, ' ').trim()}${end < text.length ? '…' : ''}`, offset: 0 };\n }\n const start = bestWindowStart(occ, EXCERPT_WINDOW);\n const end = Math.min(text.length, start + EXCERPT_WINDOW);\n let out = '';\n let cursor = start;\n for (const o of occ) {\n if (o.start < start || o.end > end) continue;\n // Occurrences of duplicate or substring-overlapping terms (\"test tests\") can overlap;\n // emitting each would duplicate document text. Keep the first, absorb the rest.\n if (o.start < cursor) continue;\n out += `${text.slice(cursor, o.start)}«${text.slice(o.start, o.end)}»`;\n cursor = o.end;\n }\n out += text.slice(cursor, end);\n const prefix = start > 0 ? '…' : '';\n const suffix = end < text.length ? '…' : '';\n return { excerpt: `${prefix}${out.replace(/\\s+/g, ' ')}${suffix}`, offset: start };\n}\n\nfunction lineNumberAt(text: string, offset: number): number {\n let line = 1;\n for (let i = 0; i < offset; i++) if (text.charCodeAt(i) === 10) line++;\n return line;\n}\n\nasync function lineRangeFor(store: Store, path: string, line: number): Promise<string | null> {\n const stmt = await store.prepare('SELECT start_line, end_line FROM sections WHERE \"path\" = ? AND start_line <= ? AND end_line >= ? ORDER BY start_line DESC LIMIT 1');\n const row = (await stmt.get(path, line, line)) as { start_line: number; end_line: number } | undefined;\n return row ? `L${row.start_line}-${row.end_line}` : null;\n}\n\nexport interface SearchOptions {\n k?: number;\n where?: string; // SQL fragment against frontmatter alias `f`, e.g. \"f.status = 'active'\"\n preset?: string; // named preset; unknown name throws listing declared presets, undefined -> \"default\"\n include?: string[]; // ad hoc scope override (repeatable --include); independent of exclude\n exclude?: string[]; // ad hoc scope override (repeatable --exclude); independent of include\n noExclude?: boolean; // --no-exclude: drop the preset's exclude for this command\n}\n\n// The declared or defaulted signals compose via RRF; `via` names which ones produced each row.\n// `opts` arrives already resolved (config.ts:resolveSearch).\nexport async function search(store: Store, cfg: ResolvedConfig, terms: string, opts: SearchOptions = {}): Promise<Row[]> {\n const effective = resolveSearch(cfg, opts);\n const { k, signals } = effective;\n\n const allPaths = ((await (await store.prepare('SELECT \"path\" FROM frontmatter')).all()) as Array<{ path: string }>).map((r) => r.path);\n const scopePaths = await rawScope(store, cfg, opts, allPaths);\n const scopeActive = scopePaths.size < allPaths.length;\n // The set every candidate pool must be filtered to before truncation: scope narrowed by\n // --where, the same composition scopedPaths() gives the other commands.\n let allowedPaths: Set<string>;\n try {\n allowedPaths = await narrowByWhere(store, scopePaths, effective.where);\n } catch (err) {\n throw searchError(err as Error, terms, effective.where);\n }\n const fetch = Math.max(k * 3, 30);\n\n // A downloadable HF id proceeds -- getProvider fetches it lazily on consent. Only a\n // local path with missing files errors here, since nothing will ever fetch it for itself.\n const wantsVectors = signals.vectors !== undefined;\n if (wantsVectors) {\n const e = embedConfig(cfg); // validate.ts guarantees this is set whenever \"vectors\" is declared\n if (localModelMissing(e)) {\n throw new SenseError('EMBED_MODEL_MISSING', `preset \"${effective.presetName}\" searches with vectors, but the local model path \"${e.model}\" is missing ${MODEL_FILENAMES}; point embed.model at a directory containing them, or drop \"vectors\" from that preset's signals to search without them`);\n }\n }\n const semanticEnabled = wantsVectors && (await scopeHasEmbeddings(store, cfg, allowedPaths));\n\n // --where applies inside the candidate query (a post-filter would drop matches ranked past\n // the pool) and again on the final select, for link-derived rows.\n const scope = effective.where;\n const whereJoin = scope ? `JOIN frontmatter f ON f.\"path\" = content.path` : '';\n const whereCond = scope ? `AND (${scope})` : '';\n // Filtered before LIMIT, not after, or scoped notes ranking below the global top-`fetch`\n // never reach the filter; joined against a temp table since real scopes exceed SQLITE_MAX_VARIABLE_NUMBER.\n if (scopeActive) await materializeScope(store, '_search_scope', scopePaths);\n const scopeCond = scopeActive ? `AND content.path IN (SELECT \"path\" FROM _search_scope)` : '';\n\n const candidates = new Map<string, { score: number; via: string }>();\n let matchRows: LexicalHit[] = [];\n // A query that is only whitespace has no searchable words: zero lexical rows, and the other\n // signals compose normally.\n if (signals.words !== undefined && terms.trim() !== '') {\n try {\n matchRows = await wordsCandidates(store, candidates, terms, whereJoin, whereCond, scopeCond, fetch, signals.words);\n } catch (err) {\n throw searchError(err as Error, terms, scope);\n }\n }\n const hits = new Map(matchRows.map((r) => [r.path, r.hit]));\n // hit === null here means the bound suppressed snippet(), not \"no match\" -- distinguish\n // from via='link' rows (never in matchRows, so absent from this set) below.\n const oversized = new Set(matchRows.filter((r) => r.hit === null).map((r) => r.path));\n\n if (signals.links !== undefined) await linksCandidates(store, candidates, matchRows, allPaths, allowedPaths, fetch, signals.links);\n\n let chunkLines = new Map<string, string>();\n let chunkSimilarity = new Map<string, number>();\n if (semanticEnabled) {\n ({ chunkLines, chunkSimilarity } = await vectorsCandidates(store, cfg, candidates, terms, fetch, allowedPaths, signals.vectors as number));\n }\n\n await store.exec('DROP TABLE IF EXISTS _search');\n // DOUBLE, not REAL: sqlite's REAL is an 8-byte double but duckdb's is a 4-byte float, and the\n // rounded score/similarity are printed at full precision in rows.\n await store.exec('CREATE TEMP TABLE _search (\"path\" TEXT PRIMARY KEY, score DOUBLE, via TEXT, hit TEXT, lines TEXT, similarity DOUBLE)');\n if (candidates.size > 0) {\n await store.runBatch(\n 'INSERT INTO _search (\"path\", score, via, hit, lines, similarity) VALUES (?, ?, ?, ?, ?, ?)',\n [...candidates].map(([path, c]) => [path, c.score, c.via, hits.get(path) ?? null, chunkLines.get(path) ?? null, chunkSimilarity.get(path) ?? null])\n );\n }\n\n // Reapplies --where even though _search is already scope+where filtered, since the join to\n // frontmatter is already needed for the path column.\n const where = scope ? `WHERE (${scope})` : '';\n // lines: semantic rows carry their chunk's range, oversized-doc lexical rows gain one\n // below, everything else stays null; similarity stays semantic-only.\n const similarityCol = semanticEnabled ? ', _search.similarity' : '';\n const selectStmt = await store.prepare(\n `SELECT f.\"path\" AS path, content.title, content.summary, _search.hit, _search.via, round(_search.score, 4) AS score, _search.lines${similarityCol}\n FROM _search JOIN frontmatter f ON f.\"path\" = _search.\"path\" JOIN content ON content.path = _search.\"path\"\n ${where} ORDER BY _search.score DESC LIMIT ?`\n );\n const rows = (await selectStmt.all(k)) as Row[];\n\n if (oversized.size > 0) {\n const bareTerms = extractBareTerms(terms);\n for (const row of rows) {\n if (row.hit !== null || !oversized.has(row.path as string)) continue;\n let text: string;\n try {\n text = readFileSync(join(cfg.baseDir, row.path as string), 'utf8');\n } catch {\n continue; // vanished since the match; leave hit/lines null rather than throw\n }\n const { excerpt, offset } = computeExcerpt(text, bareTerms);\n row.hit = excerpt;\n if (row.lines == null) row.lines = featureEnabled(cfg, 'sections') ? await lineRangeFor(store, row.path as string, lineNumberAt(text, offset)) : null;\n }\n }\n\n return rows;\n}\n"],"names":["readFileSync","join","embedConfig","featureEnabled","resolveSearch","localModelMissing","MODEL_FILENAMES","SenseError","searchError","materializeScope","narrowByWhere","rawScope","scopeHasEmbeddings","linksCandidates","vectorsCandidates","wordsCandidates","EXCERPT_WINDOW","extractBareTerms","query","cleaned","replace","split","map","tok","toLowerCase","trim","filter","length","findOccurrences","haystackLower","terms","occ","term","idx","found","indexOf","push","start","end","sort","a","b","bestWindowStart","windowSize","best","bestDistinct","bestCount","i","winEnd","seen","Set","count","j","add","size","computeExcerpt","text","Math","min","excerpt","slice","offset","out","cursor","o","prefix","suffix","lineNumberAt","line","charCodeAt","lineRangeFor","store","path","stmt","prepare","row","get","start_line","end_line","search","cfg","opts","effective","k","signals","allPaths","all","r","scopePaths","scopeActive","allowedPaths","where","err","fetch","max","wantsVectors","vectors","undefined","e","presetName","model","semanticEnabled","scope","whereJoin","whereCond","scopeCond","candidates","Map","matchRows","words","hits","hit","oversized","links","chunkLines","chunkSimilarity","exec","runBatch","c","score","via","similarityCol","selectStmt","rows","bareTerms","has","baseDir","lines"],"mappings":"AAAA,SAASA,YAAY,QAAQ,UAAU;AACvC,SAASC,IAAI,QAAQ,YAAY;AAEjC,SAASC,WAAW,EAAEC,cAAc,EAAEC,aAAa,QAAQ,qBAAqB;AAChF,SAASC,iBAAiB,EAAEC,eAAe,QAAQ,oBAAoB;AACvE,SAASC,UAAU,QAAQ,eAAe;AAE1C,SAASC,WAAW,QAAQ,4BAA4B;AAExD,SAASC,gBAAgB,EAAEC,aAAa,EAAEC,QAAQ,EAAEC,kBAAkB,QAAQ,aAAa;AAC3F,SAASC,eAAe,EAAEC,iBAAiB,EAAEC,eAAe,QAAQ,eAAe;AAEnF,iFAAiF;AACjF,mGAAmG;AACnG,MAAMC,iBAAiB;AAEvB,8FAA8F;AAC9F,qEAAqE;AACrE,SAASC,iBAAiBC,KAAa;IACrC,MAAMC,UAAUD,MACbE,OAAO,CAAC,MAAM,KACdA,OAAO,CAAC,UAAU,KAClBA,OAAO,CAAC,mCAAmC;IAC9C,OAAOD,QACJE,KAAK,CAAC,OACNC,GAAG,CAAC,CAACC,MAAQA,IAAIH,OAAO,CAAC,kBAAkB,KAAK,iCAAiC;KACjFE,GAAG,CAAC,CAACC,MAAQA,IAAIC,WAAW,GAAGC,IAAI,IACnCC,MAAM,CAAC,CAACH,MAAQA,IAAII,MAAM,GAAG;AAClC;AAEA,SAASC,gBAAgBC,aAAqB,EAAEC,KAAe;IAC7D,MAAMC,MAA2D,EAAE;IACnE,KAAK,MAAMC,QAAQF,MAAO;QACxB,IAAIG,MAAM;QACV,OAAS;YACP,MAAMC,QAAQL,cAAcM,OAAO,CAACH,MAAMC;YAC1C,IAAIC,UAAU,CAAC,GAAG;YAClBH,IAAIK,IAAI,CAAC;gBAAEC,OAAOH;gBAAOI,KAAKJ,QAAQF,KAAKL,MAAM;gBAAEK;YAAK;YACxDC,MAAMC,QAAQF,KAAKL,MAAM;QAC3B;IACF;IACA,mFAAmF;IACnF,+EAA+E;IAC/E,OAAOI,IAAIQ,IAAI,CAAC,CAACC,GAAGC,IAAMD,EAAEH,KAAK,GAAGI,EAAEJ,KAAK,IAAII,EAAEH,GAAG,GAAGE,EAAEF,GAAG;AAC9D;AAEA,wFAAwF;AACxF,6DAA6D;AAC7D,SAASI,gBAAgBX,GAAwD,EAAEY,UAAkB;IACnG,IAAIC,OAAOb,GAAG,CAAC,EAAE,CAACM,KAAK;IACvB,IAAIQ,eAAe;IACnB,IAAIC,YAAY;IAChB,IAAK,IAAIC,IAAI,GAAGA,IAAIhB,IAAIJ,MAAM,EAAEoB,IAAK;QACnC,MAAMC,SAASjB,GAAG,CAACgB,EAAE,CAACV,KAAK,GAAGM;QAC9B,MAAMM,OAAO,IAAIC;QACjB,IAAIC,QAAQ;QACZ,IAAK,IAAIC,IAAIL,GAAGK,IAAIrB,IAAIJ,MAAM,IAAII,GAAG,CAACqB,EAAE,CAACf,KAAK,GAAGW,QAAQI,IAAK;YAC5DH,KAAKI,GAAG,CAACtB,GAAG,CAACqB,EAAE,CAACpB,IAAI;YACpBmB;QACF;QACA,IAAIF,KAAKK,IAAI,GAAGT,gBAAiBI,KAAKK,IAAI,KAAKT,gBAAgBM,QAAQL,WAAY;YACjFF,OAAOb,GAAG,CAACgB,EAAE,CAACV,KAAK;YACnBQ,eAAeI,KAAKK,IAAI;YACxBR,YAAYK;QACd;IACF;IACA,OAAOP;AACT;AAEA,uFAAuF;AACvF,uFAAuF;AACvF,SAASW,eAAeC,IAAY,EAAE1B,KAAe;IACnD,MAAMC,MAAMH,gBAAgB4B,KAAKhC,WAAW,IAAIM;IAChD,IAAIC,IAAIJ,MAAM,KAAK,GAAG;QACpB,uFAAuF;QACvF,uEAAuE;QACvE,MAAMW,MAAMmB,KAAKC,GAAG,CAACF,KAAK7B,MAAM,EAAEX;QAClC,OAAO;YAAE2C,SAAS,GAAGH,KAAKI,KAAK,CAAC,GAAGtB,KAAKlB,OAAO,CAAC,QAAQ,KAAKK,IAAI,KAAKa,MAAMkB,KAAK7B,MAAM,GAAG,MAAM,IAAI;YAAEkC,QAAQ;QAAE;IAClH;IACA,MAAMxB,QAAQK,gBAAgBX,KAAKf;IACnC,MAAMsB,MAAMmB,KAAKC,GAAG,CAACF,KAAK7B,MAAM,EAAEU,QAAQrB;IAC1C,IAAI8C,MAAM;IACV,IAAIC,SAAS1B;IACb,KAAK,MAAM2B,KAAKjC,IAAK;QACnB,IAAIiC,EAAE3B,KAAK,GAAGA,SAAS2B,EAAE1B,GAAG,GAAGA,KAAK;QACpC,sFAAsF;QACtF,gFAAgF;QAChF,IAAI0B,EAAE3B,KAAK,GAAG0B,QAAQ;QACtBD,OAAO,GAAGN,KAAKI,KAAK,CAACG,QAAQC,EAAE3B,KAAK,EAAE,CAAC,EAAEmB,KAAKI,KAAK,CAACI,EAAE3B,KAAK,EAAE2B,EAAE1B,GAAG,EAAE,CAAC,CAAC;QACtEyB,SAASC,EAAE1B,GAAG;IAChB;IACAwB,OAAON,KAAKI,KAAK,CAACG,QAAQzB;IAC1B,MAAM2B,SAAS5B,QAAQ,IAAI,MAAM;IACjC,MAAM6B,SAAS5B,MAAMkB,KAAK7B,MAAM,GAAG,MAAM;IACzC,OAAO;QAAEgC,SAAS,GAAGM,SAASH,IAAI1C,OAAO,CAAC,QAAQ,OAAO8C,QAAQ;QAAEL,QAAQxB;IAAM;AACnF;AAEA,SAAS8B,aAAaX,IAAY,EAAEK,MAAc;IAChD,IAAIO,OAAO;IACX,IAAK,IAAIrB,IAAI,GAAGA,IAAIc,QAAQd,IAAK,IAAIS,KAAKa,UAAU,CAACtB,OAAO,IAAIqB;IAChE,OAAOA;AACT;AAEA,eAAeE,aAAaC,KAAY,EAAEC,IAAY,EAAEJ,IAAY;IAClE,MAAMK,OAAO,MAAMF,MAAMG,OAAO,CAAC;IACjC,MAAMC,MAAO,MAAMF,KAAKG,GAAG,CAACJ,MAAMJ,MAAMA;IACxC,OAAOO,MAAM,CAAC,CAAC,EAAEA,IAAIE,UAAU,CAAC,CAAC,EAAEF,IAAIG,QAAQ,EAAE,GAAG;AACtD;AAWA,+FAA+F;AAC/F,6DAA6D;AAC7D,OAAO,eAAeC,OAAOR,KAAY,EAAES,GAAmB,EAAElD,KAAa,EAAEmD,OAAsB,CAAC,CAAC;IACrG,MAAMC,YAAY9E,cAAc4E,KAAKC;IACrC,MAAM,EAAEE,CAAC,EAAEC,OAAO,EAAE,GAAGF;IAEvB,MAAMG,WAAW,AAAE,CAAA,MAAM,AAAC,CAAA,MAAMd,MAAMG,OAAO,CAAC,iCAAgC,EAAGY,GAAG,EAAC,EAA+BhE,GAAG,CAAC,CAACiE,IAAMA,EAAEf,IAAI;IACrI,MAAMgB,aAAa,MAAM7E,SAAS4D,OAAOS,KAAKC,MAAMI;IACpD,MAAMI,cAAcD,WAAWlC,IAAI,GAAG+B,SAAS1D,MAAM;IACrD,wFAAwF;IACxF,wEAAwE;IACxE,IAAI+D;IACJ,IAAI;QACFA,eAAe,MAAMhF,cAAc6D,OAAOiB,YAAYN,UAAUS,KAAK;IACvE,EAAE,OAAOC,KAAK;QACZ,MAAMpF,YAAYoF,KAAc9D,OAAOoD,UAAUS,KAAK;IACxD;IACA,MAAME,QAAQpC,KAAKqC,GAAG,CAACX,IAAI,GAAG;IAE9B,oFAAoF;IACpF,0FAA0F;IAC1F,MAAMY,eAAeX,QAAQY,OAAO,KAAKC;IACzC,IAAIF,cAAc;QAChB,MAAMG,IAAIhG,YAAY8E,MAAM,oEAAoE;QAChG,IAAI3E,kBAAkB6F,IAAI;YACxB,MAAM,IAAI3F,WAAW,uBAAuB,CAAC,QAAQ,EAAE2E,UAAUiB,UAAU,CAAC,mDAAmD,EAAED,EAAEE,KAAK,CAAC,aAAa,EAAE9F,gBAAgB,uHAAuH,CAAC;QAClS;IACF;IACA,MAAM+F,kBAAkBN,gBAAiB,MAAMnF,mBAAmB2D,OAAOS,KAAKU;IAE9E,2FAA2F;IAC3F,kEAAkE;IAClE,MAAMY,QAAQpB,UAAUS,KAAK;IAC7B,MAAMY,YAAYD,QAAQ,CAAC,6CAA6C,CAAC,GAAG;IAC5E,MAAME,YAAYF,QAAQ,CAAC,KAAK,EAAEA,MAAM,CAAC,CAAC,GAAG;IAC7C,yFAAyF;IACzF,2GAA2G;IAC3G,IAAIb,aAAa,MAAMhF,iBAAiB8D,OAAO,iBAAiBiB;IAChE,MAAMiB,YAAYhB,cAAc,CAAC,sDAAsD,CAAC,GAAG;IAE3F,MAAMiB,aAAa,IAAIC;IACvB,IAAIC,YAA0B,EAAE;IAChC,4FAA4F;IAC5F,4BAA4B;IAC5B,IAAIxB,QAAQyB,KAAK,KAAKZ,aAAanE,MAAML,IAAI,OAAO,IAAI;QACtD,IAAI;YACFmF,YAAY,MAAM7F,gBAAgBwD,OAAOmC,YAAY5E,OAAOyE,WAAWC,WAAWC,WAAWZ,OAAOT,QAAQyB,KAAK;QACnH,EAAE,OAAOjB,KAAK;YACZ,MAAMpF,YAAYoF,KAAc9D,OAAOwE;QACzC;IACF;IACA,MAAMQ,OAAO,IAAIH,IAAIC,UAAUtF,GAAG,CAAC,CAACiE,IAAM;YAACA,EAAEf,IAAI;YAAEe,EAAEwB,GAAG;SAAC;IACzD,wFAAwF;IACxF,4EAA4E;IAC5E,MAAMC,YAAY,IAAI9D,IAAI0D,UAAUlF,MAAM,CAAC,CAAC6D,IAAMA,EAAEwB,GAAG,KAAK,MAAMzF,GAAG,CAAC,CAACiE,IAAMA,EAAEf,IAAI;IAEnF,IAAIY,QAAQ6B,KAAK,KAAKhB,WAAW,MAAMpF,gBAAgB0D,OAAOmC,YAAYE,WAAWvB,UAAUK,cAAcG,OAAOT,QAAQ6B,KAAK;IAEjI,IAAIC,aAAa,IAAIP;IACrB,IAAIQ,kBAAkB,IAAIR;IAC1B,IAAIN,iBAAiB;QAClB,CAAA,EAAEa,UAAU,EAAEC,eAAe,EAAE,GAAG,MAAMrG,kBAAkByD,OAAOS,KAAK0B,YAAY5E,OAAO+D,OAAOH,cAAcN,QAAQY,OAAO,CAAU;IAC1I;IAEA,MAAMzB,MAAM6C,IAAI,CAAC;IACjB,8FAA8F;IAC9F,kEAAkE;IAClE,MAAM7C,MAAM6C,IAAI,CAAC;IACjB,IAAIV,WAAWpD,IAAI,GAAG,GAAG;QACvB,MAAMiB,MAAM8C,QAAQ,CAClB,8FACA;eAAIX;SAAW,CAACpF,GAAG,CAAC,CAAC,CAACkD,MAAM8C,EAAE;gBAA4BR,WAAwBI,iBAA8BC;mBAA7E;gBAAC3C;gBAAM8C,EAAEC,KAAK;gBAAED,EAAEE,GAAG;iBAAEV,YAAAA,KAAKlC,GAAG,CAACJ,mBAATsC,uBAAAA,YAAkB;iBAAMI,kBAAAA,WAAWtC,GAAG,CAACJ,mBAAf0C,6BAAAA,kBAAwB;iBAAMC,uBAAAA,gBAAgBvC,GAAG,CAACJ,mBAApB2C,kCAAAA,uBAA6B;aAAK;;IAEtJ;IAEA,2FAA2F;IAC3F,qDAAqD;IACrD,MAAMxB,QAAQW,QAAQ,CAAC,OAAO,EAAEA,MAAM,CAAC,CAAC,GAAG;IAC3C,sFAAsF;IACtF,qEAAqE;IACrE,MAAMmB,gBAAgBpB,kBAAkB,yBAAyB;IACjE,MAAMqB,aAAa,MAAMnD,MAAMG,OAAO,CACpC,CAAC,kIAAkI,EAAE+C,cAAc;;OAEhJ,EAAE9B,MAAM,oCAAoC,CAAC;IAElD,MAAMgC,OAAQ,MAAMD,WAAWpC,GAAG,CAACH;IAEnC,IAAI6B,UAAU1D,IAAI,GAAG,GAAG;QACtB,MAAMsE,YAAY3G,iBAAiBa;QACnC,KAAK,MAAM6C,OAAOgD,KAAM;YACtB,IAAIhD,IAAIoC,GAAG,KAAK,QAAQ,CAACC,UAAUa,GAAG,CAAClD,IAAIH,IAAI,GAAa;YAC5D,IAAIhB;YACJ,IAAI;gBACFA,OAAOxD,aAAaC,KAAK+E,IAAI8C,OAAO,EAAEnD,IAAIH,IAAI,GAAa;YAC7D,EAAE,OAAM;gBACN,UAAU,mEAAmE;YAC/E;YACA,MAAM,EAAEb,OAAO,EAAEE,MAAM,EAAE,GAAGN,eAAeC,MAAMoE;YACjDjD,IAAIoC,GAAG,GAAGpD;YACV,IAAIgB,IAAIoD,KAAK,IAAI,MAAMpD,IAAIoD,KAAK,GAAG5H,eAAe6E,KAAK,cAAc,MAAMV,aAAaC,OAAOI,IAAIH,IAAI,EAAYL,aAAaX,MAAMK,WAAW;QACnJ;IACF;IAEA,OAAO8D;AACT"}
|
|
1
|
+
{"version":3,"sources":["/Users/kevin/Dev/OpenSource/ai/sensemaking/src/commands/search.ts"],"sourcesContent":["import { readFileSync } from 'node:fs';\nimport { join } from 'node:path';\nimport type { ResolvedConfig } from '../config/index.ts';\nimport { embedConfig, featureEnabled, resolveSearch } from '../config/index.ts';\nimport { localModelMissing, MODEL_FILENAMES } from '../embed/store.ts';\nimport { SenseError } from '../errors.ts';\nimport type { Row } from '../output/output.ts';\nimport { searchError } from '../output/search-error.ts';\nimport type { LexicalHit, Store } from '../store/types.ts';\nimport { materializeScope, narrowByWhere, rawScope, scopeHasEmbeddings } from './scope.ts';\nimport { linksCandidates, vectorsCandidates, wordsCandidates } from './signals.ts';\n\n// snippet() re-tokenizes each candidate doc, superlinearly: ~10s for one 1MB doc\n// (benchmark/reports/2026-08-23-0.13.2-hub-release-battery.md). Past this bound, rows get the JS excerpt.\nconst EXCERPT_WINDOW = 160;\n\n// Bare terms from an FTS5 query string: strips operators/quoting so the oversized-doc excerpt\n// scan matches the same words the query matched on, not FTS5 syntax.\nfunction extractBareTerms(query: string): string[] {\n const cleaned = query\n .replace(/\"/g, ' ')\n .replace(/[()*]/g, ' ')\n .replace(/\\b(AND|OR|NOT|NEAR)\\b(\\/\\d+)?/gi, ' ');\n return cleaned\n .split(/\\s+/)\n .map((tok) => tok.replace(/^[A-Za-z_]\\w*:/, '')) // column filter, e.g. title:term\n .map((tok) => tok.toLowerCase().trim())\n .filter((tok) => tok.length > 0);\n}\n\nfunction findOccurrences(haystackLower: string, terms: string[]): Array<{ start: number; end: number; term: string }> {\n const occ: Array<{ start: number; end: number; term: string }> = [];\n for (const term of terms) {\n let idx = 0;\n for (;;) {\n const found = haystackLower.indexOf(term, idx);\n if (found === -1) break;\n occ.push({ start: found, end: found + term.length, term });\n idx = found + term.length;\n }\n }\n // Longest span first at equal start, so \"tests\" beats its substring \"test\" and the\n // whole word gets highlighted; the emit loop then absorbs the shorter overlap.\n return occ.sort((a, b) => a.start - b.start || b.end - a.end);\n}\n\n// Slides a window over the sorted occurrences, same semantics as FTS5's own best-window\n// pick: most distinct terms, ties broken by most total hits.\nfunction bestWindowStart(occ: Array<{ start: number; end: number; term: string }>, windowSize: number): number {\n let best = occ[0].start;\n let bestDistinct = 0;\n let bestCount = 0;\n for (let i = 0; i < occ.length; i++) {\n const winEnd = occ[i].start + windowSize;\n const seen = new Set<string>();\n let count = 0;\n for (let j = i; j < occ.length && occ[j].start < winEnd; j++) {\n seen.add(occ[j].term);\n count++;\n }\n if (seen.size > bestDistinct || (seen.size === bestDistinct && count > bestCount)) {\n best = occ[i].start;\n bestDistinct = seen.size;\n bestCount = count;\n }\n }\n return best;\n}\n\n// snippet()-shaped excerpt: matched terms wrapped in «», … at cut edges. Linear in doc\n// length, only run for rows actually returned (<= k), unlike snippet()'s per-row cost.\nfunction computeExcerpt(text: string, terms: string[]): { excerpt: string; offset: number } {\n const occ = findOccurrences(text.toLowerCase(), terms);\n if (occ.length === 0) {\n // Raw substring scan, not porter-stemmed: a doc matched only through a stemmed variant\n // finds no occurrence here, so fall back to the doc's start, unmarked.\n const end = Math.min(text.length, EXCERPT_WINDOW);\n return { excerpt: `${text.slice(0, end).replace(/\\s+/g, ' ').trim()}${end < text.length ? '…' : ''}`, offset: 0 };\n }\n const start = bestWindowStart(occ, EXCERPT_WINDOW);\n const end = Math.min(text.length, start + EXCERPT_WINDOW);\n let out = '';\n let cursor = start;\n for (const o of occ) {\n if (o.start < start || o.end > end) continue;\n // Occurrences of duplicate or substring-overlapping terms (\"test tests\") can overlap;\n // emitting each would duplicate document text. Keep the first, absorb the rest.\n if (o.start < cursor) continue;\n out += `${text.slice(cursor, o.start)}«${text.slice(o.start, o.end)}»`;\n cursor = o.end;\n }\n out += text.slice(cursor, end);\n const prefix = start > 0 ? '…' : '';\n const suffix = end < text.length ? '…' : '';\n return { excerpt: `${prefix}${out.replace(/\\s+/g, ' ')}${suffix}`, offset: start };\n}\n\nfunction lineNumberAt(text: string, offset: number): number {\n let line = 1;\n for (let i = 0; i < offset; i++) if (text.charCodeAt(i) === 10) line++;\n return line;\n}\n\nasync function lineRangeFor(store: Store, path: string, line: number): Promise<string | null> {\n const stmt = await store.prepare('SELECT start_line, end_line FROM sections WHERE \"path\" = ? AND start_line <= ? AND end_line >= ? ORDER BY start_line DESC LIMIT 1');\n const row = (await stmt.get(path, line, line)) as { start_line: number; end_line: number } | undefined;\n return row ? `L${row.start_line}-${row.end_line}` : null;\n}\n\nexport interface SearchOptions {\n k?: number;\n where?: string; // SQL fragment against frontmatter alias `f`, e.g. \"f.status = 'active'\"\n preset?: string; // named preset; unknown name throws listing declared presets, undefined -> \"default\"\n include?: string[]; // ad hoc scope override (repeatable --include); independent of exclude\n exclude?: string[]; // ad hoc scope override (repeatable --exclude); independent of include\n noExclude?: boolean; // --no-exclude: drop the preset's exclude for this command\n}\n\n// The declared or defaulted signals compose via RRF; `via` names which ones produced each row.\n// `opts` arrives already resolved (config.ts:resolveSearch).\nexport async function search(store: Store, cfg: ResolvedConfig, terms: string, opts: SearchOptions = {}): Promise<Row[]> {\n const effective = resolveSearch(cfg, opts);\n const { k, signals } = effective;\n\n const allPaths = ((await (await store.prepare('SELECT \"path\" FROM frontmatter')).all()) as Array<{ path: string }>).map((r) => r.path);\n const scopePaths = await rawScope(store, cfg, opts, allPaths);\n const scopeActive = scopePaths.size < allPaths.length;\n // The set every candidate pool must be filtered to before truncation: scope narrowed by\n // --where, the same composition scopedPaths() gives the other commands.\n let allowedPaths: Set<string>;\n try {\n allowedPaths = await narrowByWhere(store, scopePaths, effective.where);\n } catch (err) {\n throw searchError(err as Error, terms, effective.where);\n }\n const fetch = Math.max(k * 3, 30);\n\n // A downloadable HF id proceeds -- getProvider fetches it lazily on consent. Only a\n // local path with missing files errors here, since nothing will ever fetch it for itself.\n const wantsVectors = signals.vectors !== undefined;\n if (wantsVectors) {\n const e = embedConfig(cfg); // validate.ts guarantees this is set whenever \"vectors\" is declared\n if (localModelMissing(e)) {\n throw new SenseError('EMBED_MODEL_MISSING', `preset \"${effective.presetName}\" searches with vectors, but the local model path \"${e.model}\" is missing ${MODEL_FILENAMES}; point embed.model at a directory containing them, or drop \"vectors\" from that preset's signals to search without them`);\n }\n }\n const semanticEnabled = wantsVectors && (await scopeHasEmbeddings(store, cfg, allowedPaths));\n\n // --where applies inside the candidate query (a post-filter would drop matches ranked past\n // the pool) and again on the final select, for link-derived rows.\n const scope = effective.where;\n const whereJoin = scope ? `JOIN frontmatter f ON f.\"path\" = content.path` : '';\n const whereCond = scope ? `AND (${scope})` : '';\n // Filtered before LIMIT, not after, or scoped notes ranking below the global top-`fetch`\n // never reach the filter; joined against a temp table since real scopes exceed SQLITE_MAX_VARIABLE_NUMBER.\n if (scopeActive) await materializeScope(store, '_search_scope', scopePaths);\n const scopeCond = scopeActive ? `AND content.path IN (SELECT \"path\" FROM _search_scope)` : '';\n\n const candidates = new Map<string, { score: number; via: string }>();\n let matchRows: LexicalHit[] = [];\n // A query that is only whitespace has no searchable words: zero lexical rows, and the other\n // signals compose normally.\n if (signals.words !== undefined && terms.trim() !== '') {\n try {\n matchRows = await wordsCandidates(store, candidates, terms, whereJoin, whereCond, scopeCond, fetch, signals.words);\n } catch (err) {\n throw searchError(err as Error, terms, scope);\n }\n }\n const hits = new Map(matchRows.map((r) => [r.path, r.hit]));\n // hit === null here means the bound suppressed snippet(), not \"no match\" -- distinguish\n // from via='link' rows (never in matchRows, so absent from this set) below.\n const oversized = new Set(matchRows.filter((r) => r.hit === null).map((r) => r.path));\n\n if (signals.links !== undefined) await linksCandidates(store, candidates, matchRows, allPaths, allowedPaths, fetch, signals.links);\n\n let chunkLines = new Map<string, string>();\n let chunkSimilarity = new Map<string, number>();\n if (semanticEnabled) {\n ({ chunkLines, chunkSimilarity } = await vectorsCandidates(store, cfg, candidates, terms, fetch, allowedPaths, signals.vectors as number));\n }\n\n await store.exec('DROP TABLE IF EXISTS _search');\n // DOUBLE, not REAL: sqlite's REAL is an 8-byte double but duckdb's is a 4-byte float, and the\n // rounded score/similarity are printed at full precision in rows.\n await store.exec('CREATE TEMP TABLE _search (\"path\" TEXT PRIMARY KEY, score DOUBLE, via TEXT, hit TEXT, lines TEXT, similarity DOUBLE)');\n if (candidates.size > 0) {\n await store.runBatch(\n 'INSERT INTO _search (\"path\", score, via, hit, lines, similarity) VALUES (?, ?, ?, ?, ?, ?)',\n [...candidates].map(([path, c]) => [path, c.score, c.via, hits.get(path) ?? null, chunkLines.get(path) ?? null, chunkSimilarity.get(path) ?? null])\n );\n }\n\n // Reapplies --where even though _search is already scope+where filtered, since the join to\n // frontmatter is already needed for the path column.\n const where = scope ? `WHERE (${scope})` : '';\n // lines: semantic rows carry their chunk's range, oversized-doc lexical rows gain one\n // below, everything else stays null; similarity stays semantic-only.\n const similarityCol = semanticEnabled ? ', _search.similarity' : '';\n const selectStmt = await store.prepare(\n `SELECT f.\"path\" AS path, content.title, content.summary, _search.hit, _search.via, round(_search.score, 4) AS score, _search.lines${similarityCol}\n FROM _search JOIN frontmatter f ON f.\"path\" = _search.\"path\" JOIN content ON content.path = _search.\"path\"\n ${where} ORDER BY _search.score DESC LIMIT ?`\n );\n const rows = (await selectStmt.all(k)) as Row[];\n\n if (oversized.size > 0) {\n const bareTerms = extractBareTerms(terms);\n for (const row of rows) {\n if (row.hit !== null || !oversized.has(row.path as string)) continue;\n let text: string;\n try {\n text = readFileSync(join(cfg.baseDir, row.path as string), 'utf8');\n } catch {\n continue; // vanished since the match; leave hit/lines null rather than throw\n }\n const { excerpt, offset } = computeExcerpt(text, bareTerms);\n row.hit = excerpt;\n if (row.lines == null) row.lines = featureEnabled(cfg, 'sections') ? await lineRangeFor(store, row.path as string, lineNumberAt(text, offset)) : null;\n }\n }\n\n return rows;\n}\n"],"names":["readFileSync","join","embedConfig","featureEnabled","resolveSearch","localModelMissing","MODEL_FILENAMES","SenseError","searchError","materializeScope","narrowByWhere","rawScope","scopeHasEmbeddings","linksCandidates","vectorsCandidates","wordsCandidates","EXCERPT_WINDOW","extractBareTerms","query","cleaned","replace","split","map","tok","toLowerCase","trim","filter","length","findOccurrences","haystackLower","terms","occ","term","idx","found","indexOf","push","start","end","sort","a","b","bestWindowStart","windowSize","best","bestDistinct","bestCount","i","winEnd","seen","Set","count","j","add","size","computeExcerpt","text","Math","min","excerpt","slice","offset","out","cursor","o","prefix","suffix","lineNumberAt","line","charCodeAt","lineRangeFor","store","path","stmt","prepare","row","get","start_line","end_line","search","cfg","opts","effective","k","signals","allPaths","all","r","scopePaths","scopeActive","allowedPaths","where","err","fetch","max","wantsVectors","vectors","undefined","e","presetName","model","semanticEnabled","scope","whereJoin","whereCond","scopeCond","candidates","Map","matchRows","words","hits","hit","oversized","links","chunkLines","chunkSimilarity","exec","runBatch","c","score","via","similarityCol","selectStmt","rows","bareTerms","has","baseDir","lines"],"mappings":"AAAA,SAASA,YAAY,QAAQ,UAAU;AACvC,SAASC,IAAI,QAAQ,YAAY;AAEjC,SAASC,WAAW,EAAEC,cAAc,EAAEC,aAAa,QAAQ,qBAAqB;AAChF,SAASC,iBAAiB,EAAEC,eAAe,QAAQ,oBAAoB;AACvE,SAASC,UAAU,QAAQ,eAAe;AAE1C,SAASC,WAAW,QAAQ,4BAA4B;AAExD,SAASC,gBAAgB,EAAEC,aAAa,EAAEC,QAAQ,EAAEC,kBAAkB,QAAQ,aAAa;AAC3F,SAASC,eAAe,EAAEC,iBAAiB,EAAEC,eAAe,QAAQ,eAAe;AAEnF,iFAAiF;AACjF,0GAA0G;AAC1G,MAAMC,iBAAiB;AAEvB,8FAA8F;AAC9F,qEAAqE;AACrE,SAASC,iBAAiBC,KAAa;IACrC,MAAMC,UAAUD,MACbE,OAAO,CAAC,MAAM,KACdA,OAAO,CAAC,UAAU,KAClBA,OAAO,CAAC,mCAAmC;IAC9C,OAAOD,QACJE,KAAK,CAAC,OACNC,GAAG,CAAC,CAACC,MAAQA,IAAIH,OAAO,CAAC,kBAAkB,KAAK,iCAAiC;KACjFE,GAAG,CAAC,CAACC,MAAQA,IAAIC,WAAW,GAAGC,IAAI,IACnCC,MAAM,CAAC,CAACH,MAAQA,IAAII,MAAM,GAAG;AAClC;AAEA,SAASC,gBAAgBC,aAAqB,EAAEC,KAAe;IAC7D,MAAMC,MAA2D,EAAE;IACnE,KAAK,MAAMC,QAAQF,MAAO;QACxB,IAAIG,MAAM;QACV,OAAS;YACP,MAAMC,QAAQL,cAAcM,OAAO,CAACH,MAAMC;YAC1C,IAAIC,UAAU,CAAC,GAAG;YAClBH,IAAIK,IAAI,CAAC;gBAAEC,OAAOH;gBAAOI,KAAKJ,QAAQF,KAAKL,MAAM;gBAAEK;YAAK;YACxDC,MAAMC,QAAQF,KAAKL,MAAM;QAC3B;IACF;IACA,mFAAmF;IACnF,+EAA+E;IAC/E,OAAOI,IAAIQ,IAAI,CAAC,CAACC,GAAGC,IAAMD,EAAEH,KAAK,GAAGI,EAAEJ,KAAK,IAAII,EAAEH,GAAG,GAAGE,EAAEF,GAAG;AAC9D;AAEA,wFAAwF;AACxF,6DAA6D;AAC7D,SAASI,gBAAgBX,GAAwD,EAAEY,UAAkB;IACnG,IAAIC,OAAOb,GAAG,CAAC,EAAE,CAACM,KAAK;IACvB,IAAIQ,eAAe;IACnB,IAAIC,YAAY;IAChB,IAAK,IAAIC,IAAI,GAAGA,IAAIhB,IAAIJ,MAAM,EAAEoB,IAAK;QACnC,MAAMC,SAASjB,GAAG,CAACgB,EAAE,CAACV,KAAK,GAAGM;QAC9B,MAAMM,OAAO,IAAIC;QACjB,IAAIC,QAAQ;QACZ,IAAK,IAAIC,IAAIL,GAAGK,IAAIrB,IAAIJ,MAAM,IAAII,GAAG,CAACqB,EAAE,CAACf,KAAK,GAAGW,QAAQI,IAAK;YAC5DH,KAAKI,GAAG,CAACtB,GAAG,CAACqB,EAAE,CAACpB,IAAI;YACpBmB;QACF;QACA,IAAIF,KAAKK,IAAI,GAAGT,gBAAiBI,KAAKK,IAAI,KAAKT,gBAAgBM,QAAQL,WAAY;YACjFF,OAAOb,GAAG,CAACgB,EAAE,CAACV,KAAK;YACnBQ,eAAeI,KAAKK,IAAI;YACxBR,YAAYK;QACd;IACF;IACA,OAAOP;AACT;AAEA,uFAAuF;AACvF,uFAAuF;AACvF,SAASW,eAAeC,IAAY,EAAE1B,KAAe;IACnD,MAAMC,MAAMH,gBAAgB4B,KAAKhC,WAAW,IAAIM;IAChD,IAAIC,IAAIJ,MAAM,KAAK,GAAG;QACpB,uFAAuF;QACvF,uEAAuE;QACvE,MAAMW,MAAMmB,KAAKC,GAAG,CAACF,KAAK7B,MAAM,EAAEX;QAClC,OAAO;YAAE2C,SAAS,GAAGH,KAAKI,KAAK,CAAC,GAAGtB,KAAKlB,OAAO,CAAC,QAAQ,KAAKK,IAAI,KAAKa,MAAMkB,KAAK7B,MAAM,GAAG,MAAM,IAAI;YAAEkC,QAAQ;QAAE;IAClH;IACA,MAAMxB,QAAQK,gBAAgBX,KAAKf;IACnC,MAAMsB,MAAMmB,KAAKC,GAAG,CAACF,KAAK7B,MAAM,EAAEU,QAAQrB;IAC1C,IAAI8C,MAAM;IACV,IAAIC,SAAS1B;IACb,KAAK,MAAM2B,KAAKjC,IAAK;QACnB,IAAIiC,EAAE3B,KAAK,GAAGA,SAAS2B,EAAE1B,GAAG,GAAGA,KAAK;QACpC,sFAAsF;QACtF,gFAAgF;QAChF,IAAI0B,EAAE3B,KAAK,GAAG0B,QAAQ;QACtBD,OAAO,GAAGN,KAAKI,KAAK,CAACG,QAAQC,EAAE3B,KAAK,EAAE,CAAC,EAAEmB,KAAKI,KAAK,CAACI,EAAE3B,KAAK,EAAE2B,EAAE1B,GAAG,EAAE,CAAC,CAAC;QACtEyB,SAASC,EAAE1B,GAAG;IAChB;IACAwB,OAAON,KAAKI,KAAK,CAACG,QAAQzB;IAC1B,MAAM2B,SAAS5B,QAAQ,IAAI,MAAM;IACjC,MAAM6B,SAAS5B,MAAMkB,KAAK7B,MAAM,GAAG,MAAM;IACzC,OAAO;QAAEgC,SAAS,GAAGM,SAASH,IAAI1C,OAAO,CAAC,QAAQ,OAAO8C,QAAQ;QAAEL,QAAQxB;IAAM;AACnF;AAEA,SAAS8B,aAAaX,IAAY,EAAEK,MAAc;IAChD,IAAIO,OAAO;IACX,IAAK,IAAIrB,IAAI,GAAGA,IAAIc,QAAQd,IAAK,IAAIS,KAAKa,UAAU,CAACtB,OAAO,IAAIqB;IAChE,OAAOA;AACT;AAEA,eAAeE,aAAaC,KAAY,EAAEC,IAAY,EAAEJ,IAAY;IAClE,MAAMK,OAAO,MAAMF,MAAMG,OAAO,CAAC;IACjC,MAAMC,MAAO,MAAMF,KAAKG,GAAG,CAACJ,MAAMJ,MAAMA;IACxC,OAAOO,MAAM,CAAC,CAAC,EAAEA,IAAIE,UAAU,CAAC,CAAC,EAAEF,IAAIG,QAAQ,EAAE,GAAG;AACtD;AAWA,+FAA+F;AAC/F,6DAA6D;AAC7D,OAAO,eAAeC,OAAOR,KAAY,EAAES,GAAmB,EAAElD,KAAa,EAAEmD,OAAsB,CAAC,CAAC;IACrG,MAAMC,YAAY9E,cAAc4E,KAAKC;IACrC,MAAM,EAAEE,CAAC,EAAEC,OAAO,EAAE,GAAGF;IAEvB,MAAMG,WAAW,AAAE,CAAA,MAAM,AAAC,CAAA,MAAMd,MAAMG,OAAO,CAAC,iCAAgC,EAAGY,GAAG,EAAC,EAA+BhE,GAAG,CAAC,CAACiE,IAAMA,EAAEf,IAAI;IACrI,MAAMgB,aAAa,MAAM7E,SAAS4D,OAAOS,KAAKC,MAAMI;IACpD,MAAMI,cAAcD,WAAWlC,IAAI,GAAG+B,SAAS1D,MAAM;IACrD,wFAAwF;IACxF,wEAAwE;IACxE,IAAI+D;IACJ,IAAI;QACFA,eAAe,MAAMhF,cAAc6D,OAAOiB,YAAYN,UAAUS,KAAK;IACvE,EAAE,OAAOC,KAAK;QACZ,MAAMpF,YAAYoF,KAAc9D,OAAOoD,UAAUS,KAAK;IACxD;IACA,MAAME,QAAQpC,KAAKqC,GAAG,CAACX,IAAI,GAAG;IAE9B,oFAAoF;IACpF,0FAA0F;IAC1F,MAAMY,eAAeX,QAAQY,OAAO,KAAKC;IACzC,IAAIF,cAAc;QAChB,MAAMG,IAAIhG,YAAY8E,MAAM,oEAAoE;QAChG,IAAI3E,kBAAkB6F,IAAI;YACxB,MAAM,IAAI3F,WAAW,uBAAuB,CAAC,QAAQ,EAAE2E,UAAUiB,UAAU,CAAC,mDAAmD,EAAED,EAAEE,KAAK,CAAC,aAAa,EAAE9F,gBAAgB,uHAAuH,CAAC;QAClS;IACF;IACA,MAAM+F,kBAAkBN,gBAAiB,MAAMnF,mBAAmB2D,OAAOS,KAAKU;IAE9E,2FAA2F;IAC3F,kEAAkE;IAClE,MAAMY,QAAQpB,UAAUS,KAAK;IAC7B,MAAMY,YAAYD,QAAQ,CAAC,6CAA6C,CAAC,GAAG;IAC5E,MAAME,YAAYF,QAAQ,CAAC,KAAK,EAAEA,MAAM,CAAC,CAAC,GAAG;IAC7C,yFAAyF;IACzF,2GAA2G;IAC3G,IAAIb,aAAa,MAAMhF,iBAAiB8D,OAAO,iBAAiBiB;IAChE,MAAMiB,YAAYhB,cAAc,CAAC,sDAAsD,CAAC,GAAG;IAE3F,MAAMiB,aAAa,IAAIC;IACvB,IAAIC,YAA0B,EAAE;IAChC,4FAA4F;IAC5F,4BAA4B;IAC5B,IAAIxB,QAAQyB,KAAK,KAAKZ,aAAanE,MAAML,IAAI,OAAO,IAAI;QACtD,IAAI;YACFmF,YAAY,MAAM7F,gBAAgBwD,OAAOmC,YAAY5E,OAAOyE,WAAWC,WAAWC,WAAWZ,OAAOT,QAAQyB,KAAK;QACnH,EAAE,OAAOjB,KAAK;YACZ,MAAMpF,YAAYoF,KAAc9D,OAAOwE;QACzC;IACF;IACA,MAAMQ,OAAO,IAAIH,IAAIC,UAAUtF,GAAG,CAAC,CAACiE,IAAM;YAACA,EAAEf,IAAI;YAAEe,EAAEwB,GAAG;SAAC;IACzD,wFAAwF;IACxF,4EAA4E;IAC5E,MAAMC,YAAY,IAAI9D,IAAI0D,UAAUlF,MAAM,CAAC,CAAC6D,IAAMA,EAAEwB,GAAG,KAAK,MAAMzF,GAAG,CAAC,CAACiE,IAAMA,EAAEf,IAAI;IAEnF,IAAIY,QAAQ6B,KAAK,KAAKhB,WAAW,MAAMpF,gBAAgB0D,OAAOmC,YAAYE,WAAWvB,UAAUK,cAAcG,OAAOT,QAAQ6B,KAAK;IAEjI,IAAIC,aAAa,IAAIP;IACrB,IAAIQ,kBAAkB,IAAIR;IAC1B,IAAIN,iBAAiB;QAClB,CAAA,EAAEa,UAAU,EAAEC,eAAe,EAAE,GAAG,MAAMrG,kBAAkByD,OAAOS,KAAK0B,YAAY5E,OAAO+D,OAAOH,cAAcN,QAAQY,OAAO,CAAU;IAC1I;IAEA,MAAMzB,MAAM6C,IAAI,CAAC;IACjB,8FAA8F;IAC9F,kEAAkE;IAClE,MAAM7C,MAAM6C,IAAI,CAAC;IACjB,IAAIV,WAAWpD,IAAI,GAAG,GAAG;QACvB,MAAMiB,MAAM8C,QAAQ,CAClB,8FACA;eAAIX;SAAW,CAACpF,GAAG,CAAC,CAAC,CAACkD,MAAM8C,EAAE;gBAA4BR,WAAwBI,iBAA8BC;mBAA7E;gBAAC3C;gBAAM8C,EAAEC,KAAK;gBAAED,EAAEE,GAAG;iBAAEV,YAAAA,KAAKlC,GAAG,CAACJ,mBAATsC,uBAAAA,YAAkB;iBAAMI,kBAAAA,WAAWtC,GAAG,CAACJ,mBAAf0C,6BAAAA,kBAAwB;iBAAMC,uBAAAA,gBAAgBvC,GAAG,CAACJ,mBAApB2C,kCAAAA,uBAA6B;aAAK;;IAEtJ;IAEA,2FAA2F;IAC3F,qDAAqD;IACrD,MAAMxB,QAAQW,QAAQ,CAAC,OAAO,EAAEA,MAAM,CAAC,CAAC,GAAG;IAC3C,sFAAsF;IACtF,qEAAqE;IACrE,MAAMmB,gBAAgBpB,kBAAkB,yBAAyB;IACjE,MAAMqB,aAAa,MAAMnD,MAAMG,OAAO,CACpC,CAAC,kIAAkI,EAAE+C,cAAc;;OAEhJ,EAAE9B,MAAM,oCAAoC,CAAC;IAElD,MAAMgC,OAAQ,MAAMD,WAAWpC,GAAG,CAACH;IAEnC,IAAI6B,UAAU1D,IAAI,GAAG,GAAG;QACtB,MAAMsE,YAAY3G,iBAAiBa;QACnC,KAAK,MAAM6C,OAAOgD,KAAM;YACtB,IAAIhD,IAAIoC,GAAG,KAAK,QAAQ,CAACC,UAAUa,GAAG,CAAClD,IAAIH,IAAI,GAAa;YAC5D,IAAIhB;YACJ,IAAI;gBACFA,OAAOxD,aAAaC,KAAK+E,IAAI8C,OAAO,EAAEnD,IAAIH,IAAI,GAAa;YAC7D,EAAE,OAAM;gBACN,UAAU,mEAAmE;YAC/E;YACA,MAAM,EAAEb,OAAO,EAAEE,MAAM,EAAE,GAAGN,eAAeC,MAAMoE;YACjDjD,IAAIoC,GAAG,GAAGpD;YACV,IAAIgB,IAAIoD,KAAK,IAAI,MAAMpD,IAAIoD,KAAK,GAAG5H,eAAe6E,KAAK,cAAc,MAAMV,aAAaC,OAAOI,IAAIH,IAAI,EAAYL,aAAaX,MAAMK,WAAW;QACnJ;IACF;IAEA,OAAO8D;AACT"}
|
|
@@ -25,8 +25,8 @@ const KNOWN_FEATURE_KEYS = new Set([
|
|
|
25
25
|
'tags',
|
|
26
26
|
'rank'
|
|
27
27
|
]);
|
|
28
|
-
//
|
|
29
|
-
//
|
|
28
|
+
// The one source of truth for what the embed block accepts. schema.json carries an editor-facing
|
|
29
|
+
// copy for autocomplete; nothing at runtime reads it.
|
|
30
30
|
export const KNOWN_EMBED_KEYS = new Set([
|
|
31
31
|
'model',
|
|
32
32
|
'provider',
|