sensemaking 0.17.0 → 0.17.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (47) hide show
  1. package/dist/cjs/cli/map.js +4 -4
  2. package/dist/cjs/cli/map.js.map +1 -1
  3. package/dist/cjs/cli/named.js +2 -2
  4. package/dist/cjs/cli/named.js.map +1 -1
  5. package/dist/cjs/cli/path.js +7 -6
  6. package/dist/cjs/cli/path.js.map +1 -1
  7. package/dist/cjs/cli/peek.js +4 -4
  8. package/dist/cjs/cli/peek.js.map +1 -1
  9. package/dist/cjs/cli/related.js +4 -4
  10. package/dist/cjs/cli/related.js.map +1 -1
  11. package/dist/cjs/cli/search.js +4 -4
  12. package/dist/cjs/cli/search.js.map +1 -1
  13. package/dist/cjs/cli/status.js +18 -18
  14. package/dist/cjs/cli/status.js.map +1 -1
  15. package/dist/cjs/embed/distribution.d.cts +4 -0
  16. package/dist/cjs/embed/distribution.d.ts +4 -0
  17. package/dist/cjs/embed/distribution.js +30 -0
  18. package/dist/cjs/embed/distribution.js.map +1 -0
  19. package/dist/cjs/embed/langfit.d.cts +0 -3
  20. package/dist/cjs/embed/langfit.d.ts +0 -3
  21. package/dist/cjs/embed/langfit.js +21 -24
  22. package/dist/cjs/embed/langfit.js.map +1 -1
  23. package/dist/cjs/embed/static.js +13 -3
  24. package/dist/cjs/embed/static.js.map +1 -1
  25. package/dist/esm/cli/map.js +1 -1
  26. package/dist/esm/cli/map.js.map +1 -1
  27. package/dist/esm/cli/named.js +1 -1
  28. package/dist/esm/cli/named.js.map +1 -1
  29. package/dist/esm/cli/path.js +2 -1
  30. package/dist/esm/cli/path.js.map +1 -1
  31. package/dist/esm/cli/peek.js +1 -1
  32. package/dist/esm/cli/peek.js.map +1 -1
  33. package/dist/esm/cli/related.js +1 -1
  34. package/dist/esm/cli/related.js.map +1 -1
  35. package/dist/esm/cli/search.js +1 -1
  36. package/dist/esm/cli/search.js.map +1 -1
  37. package/dist/esm/cli/status.js +2 -2
  38. package/dist/esm/cli/status.js.map +1 -1
  39. package/dist/esm/embed/distribution.d.ts +4 -0
  40. package/dist/esm/embed/distribution.js +11 -0
  41. package/dist/esm/embed/distribution.js.map +1 -0
  42. package/dist/esm/embed/langfit.d.ts +0 -3
  43. package/dist/esm/embed/langfit.js +10 -10
  44. package/dist/esm/embed/langfit.js.map +1 -1
  45. package/dist/esm/embed/static.js +6 -1
  46. package/dist/esm/embed/static.js.map +1 -1
  47. package/package.json +1 -1
@@ -1 +1 @@
1
- {"version":3,"sources":["/Users/kevin/Dev/OpenSource/ai/sensemaking/src/embed/langfit.ts"],"sourcesContent":["import type { DatabaseSync } from 'node:sqlite';\nimport { franc } from 'franc-min';\nimport { getMeta, setMeta } from '../db/shared.ts';\nimport { SenseError } from '../errors.ts';\nimport { toIso3 } from './languages.ts';\nimport type { EmbedProvider } from './types.ts';\n\nconst META_KEY = 'embed_languages';\n// franc's own reliability floor is a handful of characters; this is a cheap upper bound, not a\n// per-chunk minimum -- a short chunk is simply more likely to come back 'und' (unclassified).\nconst SAMPLE_CHARS = 300;\n// A majority computed over a handful of chunks is noise, not a corpus signal; the decision rule\n// never fires below this many classified chunks, declared languages or not.\nconst MIN_CLASSIFIED = 10;\n\ntype LangCounts = Record<string, number>;\n\nexport function languageDistribution(db: DatabaseSync): LangCounts | undefined {\n const raw = getMeta(db, META_KEY);\n return raw ? JSON.parse(raw) : undefined;\n}\n\nfunction classify(text: string): string | undefined {\n const code = franc(text.slice(0, SAMPLE_CHARS));\n return code === 'und' ? undefined : code;\n}\n\n// Merges chunk texts into the persisted distribution; throws (without persisting) if the model\n// declares languages and a clear majority classified so far is not among them.\nexport async function checkLanguageFit(db: DatabaseSync, provider: EmbedProvider, texts: string[]): Promise<void> {\n if (texts.length === 0) return;\n const persisted = languageDistribution(db) ?? {};\n const merged = { ...persisted };\n for (const text of texts) {\n const code = classify(text);\n if (code) merged[code] = (merged[code] ?? 0) + 1;\n }\n\n const total = Object.values(merged).reduce((a, b) => a + b, 0);\n if (provider.languages && provider.languages.length > 0 && total >= MIN_CLASSIFIED) {\n const declared = new Set(provider.languages.map(toIso3));\n const [majorityLang, majorityCount] = Object.entries(merged).sort((a, b) => b[1] - a[1])[0];\n if (majorityCount / total >= 0.5 && !declared.has(majorityLang)) {\n const pct = Math.round((majorityCount / total) * 100);\n throw new SenseError(\n 'EMBED_MODEL_MISMATCH',\n `embed model ${provider.id} declares languages [${provider.languages.join(', ')}], but ${pct}% of this tree's classified text is \"${majorityLang}\" (ISO 639-3), which is not among them; choose a model for this tree's languages: \\`sense init --model ...\\`, see the sense-setup skill or INTEGRATIONS.md`\n );\n }\n }\n setMeta(db, META_KEY, JSON.stringify(merged));\n}\n"],"names":["franc","getMeta","setMeta","SenseError","toIso3","META_KEY","SAMPLE_CHARS","MIN_CLASSIFIED","languageDistribution","db","raw","JSON","parse","undefined","classify","text","code","slice","checkLanguageFit","provider","texts","length","persisted","merged","total","Object","values","reduce","a","b","languages","declared","Set","map","majorityLang","majorityCount","entries","sort","has","pct","Math","round","id","join","stringify"],"mappings":"AACA,SAASA,KAAK,QAAQ,YAAY;AAClC,SAASC,OAAO,EAAEC,OAAO,QAAQ,kBAAkB;AACnD,SAASC,UAAU,QAAQ,eAAe;AAC1C,SAASC,MAAM,QAAQ,iBAAiB;AAGxC,MAAMC,WAAW;AACjB,+FAA+F;AAC/F,8FAA8F;AAC9F,MAAMC,eAAe;AACrB,gGAAgG;AAChG,4EAA4E;AAC5E,MAAMC,iBAAiB;AAIvB,OAAO,SAASC,qBAAqBC,EAAgB;IACnD,MAAMC,MAAMT,QAAQQ,IAAIJ;IACxB,OAAOK,MAAMC,KAAKC,KAAK,CAACF,OAAOG;AACjC;AAEA,SAASC,SAASC,IAAY;IAC5B,MAAMC,OAAOhB,MAAMe,KAAKE,KAAK,CAAC,GAAGX;IACjC,OAAOU,SAAS,QAAQH,YAAYG;AACtC;AAEA,+FAA+F;AAC/F,+EAA+E;AAC/E,OAAO,eAAeE,iBAAiBT,EAAgB,EAAEU,QAAuB,EAAEC,KAAe;QAE7EZ;IADlB,IAAIY,MAAMC,MAAM,KAAK,GAAG;IACxB,MAAMC,aAAYd,wBAAAA,qBAAqBC,iBAArBD,mCAAAA,wBAA4B,CAAC;IAC/C,MAAMe,SAAS;QAAE,GAAGD,SAAS;IAAC;IAC9B,KAAK,MAAMP,QAAQK,MAAO;YAEEG;QAD1B,MAAMP,OAAOF,SAASC;QACtB,IAAIC,MAAMO,MAAM,CAACP,KAAK,GAAG,EAACO,eAAAA,MAAM,CAACP,KAAK,cAAZO,0BAAAA,eAAgB,KAAK;IACjD;IAEA,MAAMC,QAAQC,OAAOC,MAAM,CAACH,QAAQI,MAAM,CAAC,CAACC,GAAGC,IAAMD,IAAIC,GAAG;IAC5D,IAAIV,SAASW,SAAS,IAAIX,SAASW,SAAS,CAACT,MAAM,GAAG,KAAKG,SAASjB,gBAAgB;QAClF,MAAMwB,WAAW,IAAIC,IAAIb,SAASW,SAAS,CAACG,GAAG,CAAC7B;QAChD,MAAM,CAAC8B,cAAcC,cAAc,GAAGV,OAAOW,OAAO,CAACb,QAAQc,IAAI,CAAC,CAACT,GAAGC,IAAMA,CAAC,CAAC,EAAE,GAAGD,CAAC,CAAC,EAAE,CAAC,CAAC,EAAE;QAC3F,IAAIO,gBAAgBX,SAAS,OAAO,CAACO,SAASO,GAAG,CAACJ,eAAe;YAC/D,MAAMK,MAAMC,KAAKC,KAAK,CAAC,AAACN,gBAAgBX,QAAS;YACjD,MAAM,IAAIrB,WACR,wBACA,CAAC,YAAY,EAAEgB,SAASuB,EAAE,CAAC,qBAAqB,EAAEvB,SAASW,SAAS,CAACa,IAAI,CAAC,MAAM,OAAO,EAAEJ,IAAI,qCAAqC,EAAEL,aAAa,0JAA0J,CAAC;QAEhT;IACF;IACAhC,QAAQO,IAAIJ,UAAUM,KAAKiC,SAAS,CAACrB;AACvC"}
1
+ {"version":3,"sources":["/Users/kevin/Dev/OpenSource/ai/sensemaking/src/embed/langfit.ts"],"sourcesContent":["import Module from 'node:module';\nimport type { DatabaseSync } from 'node:sqlite';\nimport { SenseError } from '../errors.ts';\nimport { languageDistribution, saveLanguageDistribution } from './distribution.ts';\nimport { toIso3 } from './languages.ts';\nimport type { EmbedProvider } from './types.ts';\n\n// franc's own reliability floor is a handful of characters; this is a cheap upper bound, not a\n// per-chunk minimum -- a short chunk is simply more likely to come back 'und' (unclassified).\nconst SAMPLE_CHARS = 300;\n// A majority computed over a handful of chunks is noise, not a corpus signal; the decision rule\n// never fires below this many classified chunks, declared languages or not.\nconst MIN_CLASSIFIED = 10;\n\n// franc-min is a 148K static import; a top-level import put it on every command's module graph\n// regardless of whether the semantic path ran, shipped as the 0.17.0 regression. Deferred to a\n// synchronous require inside checkLanguageFit so only the embed path pays for it.\nconst _require = typeof require === 'undefined' ? Module.createRequire(import.meta.url) : require;\n\nfunction classify(franc: typeof import('franc-min')['franc'], text: string): string | undefined {\n const code = franc(text.slice(0, SAMPLE_CHARS));\n return code === 'und' ? undefined : code;\n}\n\n// Merges chunk texts into the persisted distribution; throws (without persisting) if the model\n// declares languages and a clear majority classified so far is not among them.\nexport async function checkLanguageFit(db: DatabaseSync, provider: EmbedProvider, texts: string[]): Promise<void> {\n if (texts.length === 0) return;\n const { franc } = _require('franc-min') as typeof import('franc-min');\n const persisted = languageDistribution(db) ?? {};\n const merged = { ...persisted };\n for (const text of texts) {\n const code = classify(franc, text);\n if (code) merged[code] = (merged[code] ?? 0) + 1;\n }\n\n const total = Object.values(merged).reduce((a, b) => a + b, 0);\n if (provider.languages && provider.languages.length > 0 && total >= MIN_CLASSIFIED) {\n const declared = new Set(provider.languages.map(toIso3));\n const [majorityLang, majorityCount] = Object.entries(merged).sort((a, b) => b[1] - a[1])[0];\n if (majorityCount / total >= 0.5 && !declared.has(majorityLang)) {\n const pct = Math.round((majorityCount / total) * 100);\n throw new SenseError(\n 'EMBED_MODEL_MISMATCH',\n `embed model ${provider.id} declares languages [${provider.languages.join(', ')}], but ${pct}% of this tree's classified text is \"${majorityLang}\" (ISO 639-3), which is not among them; choose a model for this tree's languages: \\`sense init --model ...\\`, see the sense-setup skill or INTEGRATIONS.md`\n );\n }\n }\n saveLanguageDistribution(db, merged);\n}\n"],"names":["Module","SenseError","languageDistribution","saveLanguageDistribution","toIso3","SAMPLE_CHARS","MIN_CLASSIFIED","_require","require","createRequire","url","classify","franc","text","code","slice","undefined","checkLanguageFit","db","provider","texts","length","persisted","merged","total","Object","values","reduce","a","b","languages","declared","Set","map","majorityLang","majorityCount","entries","sort","has","pct","Math","round","id","join"],"mappings":"AAAA,OAAOA,YAAY,cAAc;AAEjC,SAASC,UAAU,QAAQ,eAAe;AAC1C,SAASC,oBAAoB,EAAEC,wBAAwB,QAAQ,oBAAoB;AACnF,SAASC,MAAM,QAAQ,iBAAiB;AAGxC,+FAA+F;AAC/F,8FAA8F;AAC9F,MAAMC,eAAe;AACrB,gGAAgG;AAChG,4EAA4E;AAC5E,MAAMC,iBAAiB;AAEvB,+FAA+F;AAC/F,+FAA+F;AAC/F,kFAAkF;AAClF,MAAMC,WAAW,OAAOC,YAAY,cAAcR,OAAOS,aAAa,CAAC,YAAYC,GAAG,IAAIF;AAE1F,SAASG,SAASC,KAA0C,EAAEC,IAAY;IACxE,MAAMC,OAAOF,MAAMC,KAAKE,KAAK,CAAC,GAAGV;IACjC,OAAOS,SAAS,QAAQE,YAAYF;AACtC;AAEA,+FAA+F;AAC/F,+EAA+E;AAC/E,OAAO,eAAeG,iBAAiBC,EAAgB,EAAEC,QAAuB,EAAEC,KAAe;QAG7ElB;IAFlB,IAAIkB,MAAMC,MAAM,KAAK,GAAG;IACxB,MAAM,EAAET,KAAK,EAAE,GAAGL,SAAS;IAC3B,MAAMe,aAAYpB,wBAAAA,qBAAqBgB,iBAArBhB,mCAAAA,wBAA4B,CAAC;IAC/C,MAAMqB,SAAS;QAAE,GAAGD,SAAS;IAAC;IAC9B,KAAK,MAAMT,QAAQO,MAAO;YAEEG;QAD1B,MAAMT,OAAOH,SAASC,OAAOC;QAC7B,IAAIC,MAAMS,MAAM,CAACT,KAAK,GAAG,EAACS,eAAAA,MAAM,CAACT,KAAK,cAAZS,0BAAAA,eAAgB,KAAK;IACjD;IAEA,MAAMC,QAAQC,OAAOC,MAAM,CAACH,QAAQI,MAAM,CAAC,CAACC,GAAGC,IAAMD,IAAIC,GAAG;IAC5D,IAAIV,SAASW,SAAS,IAAIX,SAASW,SAAS,CAACT,MAAM,GAAG,KAAKG,SAASlB,gBAAgB;QAClF,MAAMyB,WAAW,IAAIC,IAAIb,SAASW,SAAS,CAACG,GAAG,CAAC7B;QAChD,MAAM,CAAC8B,cAAcC,cAAc,GAAGV,OAAOW,OAAO,CAACb,QAAQc,IAAI,CAAC,CAACT,GAAGC,IAAMA,CAAC,CAAC,EAAE,GAAGD,CAAC,CAAC,EAAE,CAAC,CAAC,EAAE;QAC3F,IAAIO,gBAAgBX,SAAS,OAAO,CAACO,SAASO,GAAG,CAACJ,eAAe;YAC/D,MAAMK,MAAMC,KAAKC,KAAK,CAAC,AAACN,gBAAgBX,QAAS;YACjD,MAAM,IAAIvB,WACR,wBACA,CAAC,YAAY,EAAEkB,SAASuB,EAAE,CAAC,qBAAqB,EAAEvB,SAASW,SAAS,CAACa,IAAI,CAAC,MAAM,OAAO,EAAEJ,IAAI,qCAAqC,EAAEL,aAAa,0JAA0J,CAAC;QAEhT;IACF;IACA/B,yBAAyBe,IAAIK;AAC/B"}
@@ -1,9 +1,13 @@
1
1
  import { existsSync, readFileSync } from 'node:fs';
2
+ import Module from 'node:module';
2
3
  import { join } from 'node:path';
3
- import { Tokenizer } from '@huggingface/tokenizers';
4
4
  import { SenseError } from '../errors.js';
5
5
  import { downloadModel, isDownloadable, MODEL_FILENAMES, MODEL_FILES, modelDir, readLanguages } from './store.js';
6
6
  const BATCH_CAP = 64;
7
+ // @huggingface/tokenizers is a 540K static import; a top-level import put it on every command's
8
+ // module graph regardless of whether the semantic path ran, shipped as the 0.17.0 regression.
9
+ // Deferred to a synchronous require inside staticProvider so only the embed path pays for it.
10
+ const _require = typeof require === 'undefined' ? Module.createRequire(import.meta.url) : require;
7
11
  // Model2Vec safetensors + pure-JS tokenizer. Encode convention from model2vec/model.py: no
8
12
  // special tokens, drop unk ids, mean-pool, L2-normalize.
9
13
  export async function staticProvider(model) {
@@ -35,6 +39,7 @@ export async function staticProvider(model) {
35
39
  const dataStart = raw.byteOffset + 8 + headerLen + spec.data_offsets[0];
36
40
  const matrix = dataStart % 4 === 0 ? new Float32Array(raw.buffer, dataStart, spec.shape[0] * dims) : new Float32Array(raw.buffer.slice(dataStart, dataStart + spec.shape[0] * dims * 4));
37
41
  const tokenizerJson = JSON.parse(readFileSync(join(dir, 'tokenizer.json'), 'utf8'));
42
+ const { Tokenizer } = _require('@huggingface/tokenizers');
38
43
  const tok = new Tokenizer(tokenizerJson, {});
39
44
  const unkId = (_ref = (_tokenizerJson_model1 = tokenizerJson.model) === null || _tokenizerJson_model1 === void 0 ? void 0 : (_tokenizerJson_model_vocab = _tokenizerJson_model1.vocab) === null || _tokenizerJson_model_vocab === void 0 ? void 0 : _tokenizerJson_model_vocab[(_tokenizerJson_model = tokenizerJson.model) === null || _tokenizerJson_model === void 0 ? void 0 : _tokenizerJson_model.unk_token]) !== null && _ref !== void 0 ? _ref : -1;
40
45
  function one(text) {
@@ -1 +1 @@
1
- {"version":3,"sources":["/Users/kevin/Dev/OpenSource/ai/sensemaking/src/embed/static.ts"],"sourcesContent":["import { existsSync, readFileSync } from 'node:fs';\nimport { join } from 'node:path';\nimport { Tokenizer } from '@huggingface/tokenizers';\nimport { SenseError } from '../errors.ts';\nimport { downloadModel, isDownloadable, MODEL_FILENAMES, MODEL_FILES, modelDir, readLanguages } from './store.ts';\nimport type { EmbedProvider } from './types.ts';\n\nconst BATCH_CAP = 64;\n\n// Model2Vec safetensors + pure-JS tokenizer. Encode convention from model2vec/model.py: no\n// special tokens, drop unk ids, mean-pool, L2-normalize.\nexport async function staticProvider(model: string): Promise<EmbedProvider> {\n let dir: string;\n let languages: string[] | undefined;\n if (isDownloadable(model)) {\n // Naming a HF id is consent: fetch whatever is missing now, announcing it on stderr.\n dir = await downloadModel(model, (file, into) => console.error(`sense: fetching ${model}/${file} into ${into}`));\n // Cached beside the repo; a local model directory has no card to read, so it has none.\n const cached = readLanguages(model);\n languages = cached && cached.length > 0 ? cached : undefined;\n } else {\n dir = modelDir(model);\n for (const file of MODEL_FILES) {\n if (!existsSync(join(dir, file))) {\n throw new SenseError('EMBED_MODEL_MISSING', `embed model ${model} is not available (looked in ${dir}); expected ${MODEL_FILENAMES} in that directory`);\n }\n }\n }\n\n const raw = readFileSync(join(dir, 'model.safetensors'));\n const headerLen = Number(raw.readBigUInt64LE(0));\n const header = JSON.parse(raw.subarray(8, 8 + headerLen).toString('utf8')) as Record<string, { dtype: string; shape: number[]; data_offsets: number[] }>;\n const entry = Object.entries(header).find(([k]) => k !== '__metadata__');\n if (!entry || entry[1].dtype !== 'F32') throw new SenseError('EMBED_MODEL', `${model}: expected an F32 safetensors matrix`);\n const spec = entry[1];\n const dims = spec.shape[1];\n const dataStart = raw.byteOffset + 8 + headerLen + spec.data_offsets[0];\n const matrix = dataStart % 4 === 0 ? new Float32Array(raw.buffer, dataStart, spec.shape[0] * dims) : new Float32Array(raw.buffer.slice(dataStart, dataStart + spec.shape[0] * dims * 4));\n\n const tokenizerJson = JSON.parse(readFileSync(join(dir, 'tokenizer.json'), 'utf8'));\n const tok = new Tokenizer(tokenizerJson, {});\n const unkId = tokenizerJson.model?.vocab?.[tokenizerJson.model?.unk_token] ?? -1;\n\n function one(text: string): Float32Array {\n // The tokenizer yields undefined (not the unk id) for tokens outside the vocab; an\n // undefined id would index the matrix at NaN and poison the whole mean-pool -- and the\n // int8 conversion then stores the NaN vector as all zeros, silently. Keep integers only.\n const ids = (tok.encode(text, { add_special_tokens: false }).ids as number[]).filter((id) => Number.isInteger(id) && id !== unkId);\n const v = new Float32Array(dims);\n if (ids.length === 0) return v;\n for (const id of ids) {\n const off = id * dims;\n for (let d = 0; d < dims; d++) v[d] += matrix[off + d];\n }\n let norm = 0;\n for (let d = 0; d < dims; d++) {\n v[d] /= ids.length;\n norm += v[d] * v[d];\n }\n norm = Math.sqrt(norm) + 1e-32;\n for (let d = 0; d < dims; d++) v[d] /= norm;\n return v;\n }\n\n // Symmetric model: document and query embedding are the same call.\n return { id: `static:${model}`, dims, batchCap: BATCH_CAP, languages, embedDocuments: async (texts) => texts.map(one), embedQuery: async (text) => one(text) };\n}\n"],"names":["existsSync","readFileSync","join","Tokenizer","SenseError","downloadModel","isDownloadable","MODEL_FILENAMES","MODEL_FILES","modelDir","readLanguages","BATCH_CAP","staticProvider","model","tokenizerJson","dir","languages","file","into","console","error","cached","length","undefined","raw","headerLen","Number","readBigUInt64LE","header","JSON","parse","subarray","toString","entry","Object","entries","find","k","dtype","spec","dims","shape","dataStart","byteOffset","data_offsets","matrix","Float32Array","buffer","slice","tok","unkId","vocab","unk_token","one","text","ids","encode","add_special_tokens","filter","id","isInteger","v","off","d","norm","Math","sqrt","batchCap","embedDocuments","texts","map","embedQuery"],"mappings":"AAAA,SAASA,UAAU,EAAEC,YAAY,QAAQ,UAAU;AACnD,SAASC,IAAI,QAAQ,YAAY;AACjC,SAASC,SAAS,QAAQ,0BAA0B;AACpD,SAASC,UAAU,QAAQ,eAAe;AAC1C,SAASC,aAAa,EAAEC,cAAc,EAAEC,eAAe,EAAEC,WAAW,EAAEC,QAAQ,EAAEC,aAAa,QAAQ,aAAa;AAGlH,MAAMC,YAAY;AAElB,2FAA2F;AAC3F,yDAAyD;AACzD,OAAO,eAAeC,eAAeC,KAAa;;QA8BLC,sBAA7BA,4BAAAA;IA7Bd,IAAIC;IACJ,IAAIC;IACJ,IAAIV,eAAeO,QAAQ;QACzB,qFAAqF;QACrFE,MAAM,MAAMV,cAAcQ,OAAO,CAACI,MAAMC,OAASC,QAAQC,KAAK,CAAC,CAAC,gBAAgB,EAAEP,MAAM,CAAC,EAAEI,KAAK,MAAM,EAAEC,MAAM;QAC9G,uFAAuF;QACvF,MAAMG,SAASX,cAAcG;QAC7BG,YAAYK,UAAUA,OAAOC,MAAM,GAAG,IAAID,SAASE;IACrD,OAAO;QACLR,MAAMN,SAASI;QACf,KAAK,MAAMI,QAAQT,YAAa;YAC9B,IAAI,CAACR,WAAWE,KAAKa,KAAKE,QAAQ;gBAChC,MAAM,IAAIb,WAAW,uBAAuB,CAAC,YAAY,EAAES,MAAM,6BAA6B,EAAEE,IAAI,YAAY,EAAER,gBAAgB,kBAAkB,CAAC;YACvJ;QACF;IACF;IAEA,MAAMiB,MAAMvB,aAAaC,KAAKa,KAAK;IACnC,MAAMU,YAAYC,OAAOF,IAAIG,eAAe,CAAC;IAC7C,MAAMC,SAASC,KAAKC,KAAK,CAACN,IAAIO,QAAQ,CAAC,GAAG,IAAIN,WAAWO,QAAQ,CAAC;IAClE,MAAMC,QAAQC,OAAOC,OAAO,CAACP,QAAQQ,IAAI,CAAC,CAAC,CAACC,EAAE,GAAKA,MAAM;IACzD,IAAI,CAACJ,SAASA,KAAK,CAAC,EAAE,CAACK,KAAK,KAAK,OAAO,MAAM,IAAIlC,WAAW,eAAe,GAAGS,MAAM,oCAAoC,CAAC;IAC1H,MAAM0B,OAAON,KAAK,CAAC,EAAE;IACrB,MAAMO,OAAOD,KAAKE,KAAK,CAAC,EAAE;IAC1B,MAAMC,YAAYlB,IAAImB,UAAU,GAAG,IAAIlB,YAAYc,KAAKK,YAAY,CAAC,EAAE;IACvE,MAAMC,SAASH,YAAY,MAAM,IAAI,IAAII,aAAatB,IAAIuB,MAAM,EAAEL,WAAWH,KAAKE,KAAK,CAAC,EAAE,GAAGD,QAAQ,IAAIM,aAAatB,IAAIuB,MAAM,CAACC,KAAK,CAACN,WAAWA,YAAYH,KAAKE,KAAK,CAAC,EAAE,GAAGD,OAAO;IAErL,MAAM1B,gBAAgBe,KAAKC,KAAK,CAAC7B,aAAaC,KAAKa,KAAK,mBAAmB;IAC3E,MAAMkC,MAAM,IAAI9C,UAAUW,eAAe,CAAC;IAC1C,MAAMoC,iBAAQpC,wBAAAA,cAAcD,KAAK,cAAnBC,6CAAAA,6BAAAA,sBAAqBqC,KAAK,cAA1BrC,iDAAAA,0BAA4B,EAACA,uBAAAA,cAAcD,KAAK,cAAnBC,2CAAAA,qBAAqBsC,SAAS,CAAC,uCAAI,CAAC;IAE/E,SAASC,IAAIC,IAAY;QACvB,mFAAmF;QACnF,uFAAuF;QACvF,yFAAyF;QACzF,MAAMC,MAAM,AAACN,IAAIO,MAAM,CAACF,MAAM;YAAEG,oBAAoB;QAAM,GAAGF,GAAG,CAAcG,MAAM,CAAC,CAACC,KAAOjC,OAAOkC,SAAS,CAACD,OAAOA,OAAOT;QAC5H,MAAMW,IAAI,IAAIf,aAAaN;QAC3B,IAAIe,IAAIjC,MAAM,KAAK,GAAG,OAAOuC;QAC7B,KAAK,MAAMF,MAAMJ,IAAK;YACpB,MAAMO,MAAMH,KAAKnB;YACjB,IAAK,IAAIuB,IAAI,GAAGA,IAAIvB,MAAMuB,IAAKF,CAAC,CAACE,EAAE,IAAIlB,MAAM,CAACiB,MAAMC,EAAE;QACxD;QACA,IAAIC,OAAO;QACX,IAAK,IAAID,IAAI,GAAGA,IAAIvB,MAAMuB,IAAK;YAC7BF,CAAC,CAACE,EAAE,IAAIR,IAAIjC,MAAM;YAClB0C,QAAQH,CAAC,CAACE,EAAE,GAAGF,CAAC,CAACE,EAAE;QACrB;QACAC,OAAOC,KAAKC,IAAI,CAACF,QAAQ;QACzB,IAAK,IAAID,IAAI,GAAGA,IAAIvB,MAAMuB,IAAKF,CAAC,CAACE,EAAE,IAAIC;QACvC,OAAOH;IACT;IAEA,mEAAmE;IACnE,OAAO;QAAEF,IAAI,CAAC,OAAO,EAAE9C,OAAO;QAAE2B;QAAM2B,UAAUxD;QAAWK;QAAWoD,gBAAgB,OAAOC,QAAUA,MAAMC,GAAG,CAACjB;QAAMkB,YAAY,OAAOjB,OAASD,IAAIC;IAAM;AAC/J"}
1
+ {"version":3,"sources":["/Users/kevin/Dev/OpenSource/ai/sensemaking/src/embed/static.ts"],"sourcesContent":["import { existsSync, readFileSync } from 'node:fs';\nimport Module from 'node:module';\nimport { join } from 'node:path';\nimport { SenseError } from '../errors.ts';\nimport { downloadModel, isDownloadable, MODEL_FILENAMES, MODEL_FILES, modelDir, readLanguages } from './store.ts';\nimport type { EmbedProvider } from './types.ts';\n\nconst BATCH_CAP = 64;\n\n// @huggingface/tokenizers is a 540K static import; a top-level import put it on every command's\n// module graph regardless of whether the semantic path ran, shipped as the 0.17.0 regression.\n// Deferred to a synchronous require inside staticProvider so only the embed path pays for it.\nconst _require = typeof require === 'undefined' ? Module.createRequire(import.meta.url) : require;\n\n// Model2Vec safetensors + pure-JS tokenizer. Encode convention from model2vec/model.py: no\n// special tokens, drop unk ids, mean-pool, L2-normalize.\nexport async function staticProvider(model: string): Promise<EmbedProvider> {\n let dir: string;\n let languages: string[] | undefined;\n if (isDownloadable(model)) {\n // Naming a HF id is consent: fetch whatever is missing now, announcing it on stderr.\n dir = await downloadModel(model, (file, into) => console.error(`sense: fetching ${model}/${file} into ${into}`));\n // Cached beside the repo; a local model directory has no card to read, so it has none.\n const cached = readLanguages(model);\n languages = cached && cached.length > 0 ? cached : undefined;\n } else {\n dir = modelDir(model);\n for (const file of MODEL_FILES) {\n if (!existsSync(join(dir, file))) {\n throw new SenseError('EMBED_MODEL_MISSING', `embed model ${model} is not available (looked in ${dir}); expected ${MODEL_FILENAMES} in that directory`);\n }\n }\n }\n\n const raw = readFileSync(join(dir, 'model.safetensors'));\n const headerLen = Number(raw.readBigUInt64LE(0));\n const header = JSON.parse(raw.subarray(8, 8 + headerLen).toString('utf8')) as Record<string, { dtype: string; shape: number[]; data_offsets: number[] }>;\n const entry = Object.entries(header).find(([k]) => k !== '__metadata__');\n if (!entry || entry[1].dtype !== 'F32') throw new SenseError('EMBED_MODEL', `${model}: expected an F32 safetensors matrix`);\n const spec = entry[1];\n const dims = spec.shape[1];\n const dataStart = raw.byteOffset + 8 + headerLen + spec.data_offsets[0];\n const matrix = dataStart % 4 === 0 ? new Float32Array(raw.buffer, dataStart, spec.shape[0] * dims) : new Float32Array(raw.buffer.slice(dataStart, dataStart + spec.shape[0] * dims * 4));\n\n const tokenizerJson = JSON.parse(readFileSync(join(dir, 'tokenizer.json'), 'utf8'));\n const { Tokenizer } = _require('@huggingface/tokenizers') as typeof import('@huggingface/tokenizers');\n const tok = new Tokenizer(tokenizerJson, {});\n const unkId = tokenizerJson.model?.vocab?.[tokenizerJson.model?.unk_token] ?? -1;\n\n function one(text: string): Float32Array {\n // The tokenizer yields undefined (not the unk id) for tokens outside the vocab; an\n // undefined id would index the matrix at NaN and poison the whole mean-pool -- and the\n // int8 conversion then stores the NaN vector as all zeros, silently. Keep integers only.\n const ids = (tok.encode(text, { add_special_tokens: false }).ids as number[]).filter((id) => Number.isInteger(id) && id !== unkId);\n const v = new Float32Array(dims);\n if (ids.length === 0) return v;\n for (const id of ids) {\n const off = id * dims;\n for (let d = 0; d < dims; d++) v[d] += matrix[off + d];\n }\n let norm = 0;\n for (let d = 0; d < dims; d++) {\n v[d] /= ids.length;\n norm += v[d] * v[d];\n }\n norm = Math.sqrt(norm) + 1e-32;\n for (let d = 0; d < dims; d++) v[d] /= norm;\n return v;\n }\n\n // Symmetric model: document and query embedding are the same call.\n return { id: `static:${model}`, dims, batchCap: BATCH_CAP, languages, embedDocuments: async (texts) => texts.map(one), embedQuery: async (text) => one(text) };\n}\n"],"names":["existsSync","readFileSync","Module","join","SenseError","downloadModel","isDownloadable","MODEL_FILENAMES","MODEL_FILES","modelDir","readLanguages","BATCH_CAP","_require","require","createRequire","url","staticProvider","model","tokenizerJson","dir","languages","file","into","console","error","cached","length","undefined","raw","headerLen","Number","readBigUInt64LE","header","JSON","parse","subarray","toString","entry","Object","entries","find","k","dtype","spec","dims","shape","dataStart","byteOffset","data_offsets","matrix","Float32Array","buffer","slice","Tokenizer","tok","unkId","vocab","unk_token","one","text","ids","encode","add_special_tokens","filter","id","isInteger","v","off","d","norm","Math","sqrt","batchCap","embedDocuments","texts","map","embedQuery"],"mappings":"AAAA,SAASA,UAAU,EAAEC,YAAY,QAAQ,UAAU;AACnD,OAAOC,YAAY,cAAc;AACjC,SAASC,IAAI,QAAQ,YAAY;AACjC,SAASC,UAAU,QAAQ,eAAe;AAC1C,SAASC,aAAa,EAAEC,cAAc,EAAEC,eAAe,EAAEC,WAAW,EAAEC,QAAQ,EAAEC,aAAa,QAAQ,aAAa;AAGlH,MAAMC,YAAY;AAElB,gGAAgG;AAChG,8FAA8F;AAC9F,8FAA8F;AAC9F,MAAMC,WAAW,OAAOC,YAAY,cAAcX,OAAOY,aAAa,CAAC,YAAYC,GAAG,IAAIF;AAE1F,2FAA2F;AAC3F,yDAAyD;AACzD,OAAO,eAAeG,eAAeC,KAAa;;QA+BLC,sBAA7BA,4BAAAA;IA9Bd,IAAIC;IACJ,IAAIC;IACJ,IAAId,eAAeW,QAAQ;QACzB,qFAAqF;QACrFE,MAAM,MAAMd,cAAcY,OAAO,CAACI,MAAMC,OAASC,QAAQC,KAAK,CAAC,CAAC,gBAAgB,EAAEP,MAAM,CAAC,EAAEI,KAAK,MAAM,EAAEC,MAAM;QAC9G,uFAAuF;QACvF,MAAMG,SAASf,cAAcO;QAC7BG,YAAYK,UAAUA,OAAOC,MAAM,GAAG,IAAID,SAASE;IACrD,OAAO;QACLR,MAAMV,SAASQ;QACf,KAAK,MAAMI,QAAQb,YAAa;YAC9B,IAAI,CAACR,WAAWG,KAAKgB,KAAKE,QAAQ;gBAChC,MAAM,IAAIjB,WAAW,uBAAuB,CAAC,YAAY,EAAEa,MAAM,6BAA6B,EAAEE,IAAI,YAAY,EAAEZ,gBAAgB,kBAAkB,CAAC;YACvJ;QACF;IACF;IAEA,MAAMqB,MAAM3B,aAAaE,KAAKgB,KAAK;IACnC,MAAMU,YAAYC,OAAOF,IAAIG,eAAe,CAAC;IAC7C,MAAMC,SAASC,KAAKC,KAAK,CAACN,IAAIO,QAAQ,CAAC,GAAG,IAAIN,WAAWO,QAAQ,CAAC;IAClE,MAAMC,QAAQC,OAAOC,OAAO,CAACP,QAAQQ,IAAI,CAAC,CAAC,CAACC,EAAE,GAAKA,MAAM;IACzD,IAAI,CAACJ,SAASA,KAAK,CAAC,EAAE,CAACK,KAAK,KAAK,OAAO,MAAM,IAAItC,WAAW,eAAe,GAAGa,MAAM,oCAAoC,CAAC;IAC1H,MAAM0B,OAAON,KAAK,CAAC,EAAE;IACrB,MAAMO,OAAOD,KAAKE,KAAK,CAAC,EAAE;IAC1B,MAAMC,YAAYlB,IAAImB,UAAU,GAAG,IAAIlB,YAAYc,KAAKK,YAAY,CAAC,EAAE;IACvE,MAAMC,SAASH,YAAY,MAAM,IAAI,IAAII,aAAatB,IAAIuB,MAAM,EAAEL,WAAWH,KAAKE,KAAK,CAAC,EAAE,GAAGD,QAAQ,IAAIM,aAAatB,IAAIuB,MAAM,CAACC,KAAK,CAACN,WAAWA,YAAYH,KAAKE,KAAK,CAAC,EAAE,GAAGD,OAAO;IAErL,MAAM1B,gBAAgBe,KAAKC,KAAK,CAACjC,aAAaE,KAAKgB,KAAK,mBAAmB;IAC3E,MAAM,EAAEkC,SAAS,EAAE,GAAGzC,SAAS;IAC/B,MAAM0C,MAAM,IAAID,UAAUnC,eAAe,CAAC;IAC1C,MAAMqC,iBAAQrC,wBAAAA,cAAcD,KAAK,cAAnBC,6CAAAA,6BAAAA,sBAAqBsC,KAAK,cAA1BtC,iDAAAA,0BAA4B,EAACA,uBAAAA,cAAcD,KAAK,cAAnBC,2CAAAA,qBAAqBuC,SAAS,CAAC,uCAAI,CAAC;IAE/E,SAASC,IAAIC,IAAY;QACvB,mFAAmF;QACnF,uFAAuF;QACvF,yFAAyF;QACzF,MAAMC,MAAM,AAACN,IAAIO,MAAM,CAACF,MAAM;YAAEG,oBAAoB;QAAM,GAAGF,GAAG,CAAcG,MAAM,CAAC,CAACC,KAAOlC,OAAOmC,SAAS,CAACD,OAAOA,OAAOT;QAC5H,MAAMW,IAAI,IAAIhB,aAAaN;QAC3B,IAAIgB,IAAIlC,MAAM,KAAK,GAAG,OAAOwC;QAC7B,KAAK,MAAMF,MAAMJ,IAAK;YACpB,MAAMO,MAAMH,KAAKpB;YACjB,IAAK,IAAIwB,IAAI,GAAGA,IAAIxB,MAAMwB,IAAKF,CAAC,CAACE,EAAE,IAAInB,MAAM,CAACkB,MAAMC,EAAE;QACxD;QACA,IAAIC,OAAO;QACX,IAAK,IAAID,IAAI,GAAGA,IAAIxB,MAAMwB,IAAK;YAC7BF,CAAC,CAACE,EAAE,IAAIR,IAAIlC,MAAM;YAClB2C,QAAQH,CAAC,CAACE,EAAE,GAAGF,CAAC,CAACE,EAAE;QACrB;QACAC,OAAOC,KAAKC,IAAI,CAACF,QAAQ;QACzB,IAAK,IAAID,IAAI,GAAGA,IAAIxB,MAAMwB,IAAKF,CAAC,CAACE,EAAE,IAAIC;QACvC,OAAOH;IACT;IAEA,mEAAmE;IACnE,OAAO;QAAEF,IAAI,CAAC,OAAO,EAAE/C,OAAO;QAAE2B;QAAM4B,UAAU7D;QAAWS;QAAWqD,gBAAgB,OAAOC,QAAUA,MAAMC,GAAG,CAACjB;QAAMkB,YAAY,OAAOjB,OAASD,IAAIC;IAAM;AAC/J"}
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "sensemaking",
3
- "version": "0.17.0",
3
+ "version": "0.17.1",
4
4
  "description": "Query and search your markdown notes with context-aware progressive disclosure: SQL over frontmatter, links, and text, plus semantic search and link-graph ranking. No server, no build step",
5
5
  "keywords": [
6
6
  "markdown",