@tryhamster/gerbil 1.6.3 → 1.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/{architectures-CpmH4Gha.mjs → architectures-BHkqQ9xp.mjs} +1 -1
- package/dist/{architectures-CpmH4Gha.mjs.map → architectures-BHkqQ9xp.mjs.map} +1 -1
- package/dist/browser/index.d.ts +8 -0
- package/dist/browser/index.d.ts.map +1 -1
- package/dist/cli.mjs +13 -13
- package/dist/cli.mjs.map +1 -1
- package/dist/{defaults-KnrRSIeJ.mjs → defaults-C_bJK9zs.mjs} +1 -1
- package/dist/{defaults-KnrRSIeJ.mjs.map → defaults-C_bJK9zs.mjs.map} +1 -1
- package/dist/frameworks/express.d.mts +2 -2
- package/dist/frameworks/express.mjs +3 -3
- package/dist/frameworks/fastify.d.mts +2 -2
- package/dist/frameworks/fastify.mjs +2 -2
- package/dist/frameworks/hono.d.mts +2 -2
- package/dist/frameworks/hono.mjs +2 -2
- package/dist/frameworks/next.d.mts +4 -4
- package/dist/frameworks/next.mjs +2 -2
- package/dist/frameworks/react.d.mts +2 -2
- package/dist/frameworks/trpc.d.mts +2 -2
- package/dist/frameworks/trpc.mjs +2 -2
- package/dist/gerbil-BFk5jV0h.mjs +4 -0
- package/dist/{gerbil-qJhKR8JR.mjs → gerbil-Cgmb4Dit.mjs} +41 -5
- package/dist/gerbil-Cgmb4Dit.mjs.map +1 -0
- package/dist/{gerbil-BHPSiECS.d.mts → gerbil-DU1aRO6v.d.mts} +25 -4
- package/dist/gerbil-DU1aRO6v.d.mts.map +1 -0
- package/dist/gpu/hooks.d.mts +3 -3
- package/dist/gpu/hooks.mjs +2 -2
- package/dist/gpu/index.d.mts +3 -3
- package/dist/gpu/index.mjs +5 -5
- package/dist/{gpu-BZVtzX9N.mjs → gpu-kQVLpV3n.mjs} +4 -4
- package/dist/{gpu-BZVtzX9N.mjs.map → gpu-kQVLpV3n.mjs.map} +1 -1
- package/dist/{index-CRDf-nS4.d.mts → index-B3tjyDJI.d.mts} +2 -2
- package/dist/{index-CRDf-nS4.d.mts.map → index-B3tjyDJI.d.mts.map} +1 -1
- package/dist/{index-CZVIHkaH.d.mts → index-ElJKy9i9.d.mts} +119 -26
- package/dist/index-ElJKy9i9.d.mts.map +1 -0
- package/dist/index.d.mts +6 -6
- package/dist/index.mjs +11 -11
- package/dist/indexeddb-store-B6HYVgQ_.mjs +4 -0
- package/dist/{indexeddb-store-BWIMtxxH.mjs → indexeddb-store-BiamQL7K.mjs} +2 -2
- package/dist/{indexeddb-store-BWIMtxxH.mjs.map → indexeddb-store-BiamQL7K.mjs.map} +1 -1
- package/dist/integrations/ai-sdk.d.mts +2 -2
- package/dist/integrations/ai-sdk.mjs +2 -2
- package/dist/integrations/langchain.d.mts +2 -2
- package/dist/integrations/langchain.mjs +2 -2
- package/dist/integrations/llamaindex.d.mts +2 -2
- package/dist/integrations/llamaindex.mjs +2 -2
- package/dist/integrations/mcp-client.mjs +2 -2
- package/dist/integrations/mcp.d.mts +4 -4
- package/dist/integrations/mcp.mjs +5 -5
- package/dist/{mcp-DIygAeaN.mjs → mcp-DTusP2m5.mjs} +3 -3
- package/dist/{mcp-DIygAeaN.mjs.map → mcp-DTusP2m5.mjs.map} +1 -1
- package/dist/memory/index.d.mts +2 -2
- package/dist/memory/index.mjs +4 -4
- package/dist/{memory-Dj0J1v88.mjs → memory-BCKc_Jfp.mjs} +2 -2
- package/dist/{memory-Dj0J1v88.mjs.map → memory-BCKc_Jfp.mjs.map} +1 -1
- package/dist/memory-BwYh_DZh.mjs +4 -0
- package/dist/{memory-DVN0MnIG.mjs → memory-Dnjhc1du.mjs} +2 -2
- package/dist/{memory-DVN0MnIG.mjs.map → memory-Dnjhc1du.mjs.map} +1 -1
- package/dist/{moonshine-stt-BFIm_Xxk.mjs → moonshine-stt-9wc6t11v.mjs} +341 -127
- package/dist/moonshine-stt-9wc6t11v.mjs.map +1 -0
- package/dist/moonshine-stt-B7rDn6Q9.mjs +4 -0
- package/dist/{one-liner-CGozk7xR.mjs → one-liner-Bk5x7gYH.mjs} +2 -2
- package/dist/{one-liner-CGozk7xR.mjs.map → one-liner-Bk5x7gYH.mjs.map} +1 -1
- package/dist/{outetts-Cr5oGk4z.d.mts → outetts-CAL3_K3j.d.mts} +1 -1
- package/dist/{outetts-Cr5oGk4z.d.mts.map → outetts-CAL3_K3j.d.mts.map} +1 -1
- package/dist/repl-DV6l-jT8.mjs +9 -0
- package/dist/skills/index.d.mts +4 -4
- package/dist/skills/index.mjs +4 -4
- package/dist/{skills-DJ0cCjfD.mjs → skills-BhcwnL2l.mjs} +2 -2
- package/dist/{skills-DJ0cCjfD.mjs.map → skills-BhcwnL2l.mjs.map} +1 -1
- package/dist/{tools-DQ1mPUw5.mjs → tools-DRYfLfKS.mjs} +2 -2
- package/dist/{tools-DQ1mPUw5.mjs.map → tools-DRYfLfKS.mjs.map} +1 -1
- package/dist/tune/index.d.mts +186 -0
- package/dist/tune/index.d.mts.map +1 -0
- package/dist/tune/index.mjs +297 -0
- package/dist/tune/index.mjs.map +1 -0
- package/dist/{types-BBDAW4MS.d.mts → types-BzbBDoaP.d.mts} +10 -2
- package/dist/types-BzbBDoaP.d.mts.map +1 -0
- package/dist/{types-Dqv4m0vo.d.mts → types-CIcNt_XS.d.mts} +1 -1
- package/dist/{types-Dqv4m0vo.d.mts.map → types-CIcNt_XS.d.mts.map} +1 -1
- package/dist/{utils-DKO55ZmZ.mjs → utils-CZBZ8dgR.mjs} +1 -1
- package/dist/{utils-DKO55ZmZ.mjs.map → utils-CZBZ8dgR.mjs.map} +1 -1
- package/dist/{vector-B0panuy6.mjs → vector-BJqdOci_.mjs} +1 -1
- package/dist/{vector-B0panuy6.mjs.map → vector-BJqdOci_.mjs.map} +1 -1
- package/package.json +5 -1
- package/dist/gerbil-BHPSiECS.d.mts.map +0 -1
- package/dist/gerbil-eX40N7oB.mjs +0 -4
- package/dist/gerbil-qJhKR8JR.mjs.map +0 -1
- package/dist/index-CZVIHkaH.d.mts.map +0 -1
- package/dist/indexeddb-store-ClH12Xnl.mjs +0 -4
- package/dist/memory-D1P7Tmda.mjs +0 -4
- package/dist/moonshine-stt-BFIm_Xxk.mjs.map +0 -1
- package/dist/moonshine-stt-D7BqG2pU.mjs +0 -4
- package/dist/repl-C_-ogJ09.mjs +0 -9
- package/dist/types-BBDAW4MS.d.mts.map +0 -1
- /package/dist/{auto-update-BVaLXcDE.mjs → auto-update-DIh0uiTw.mjs} +0 -0
- /package/dist/{chunk-B9cbKln6.mjs → chunk-BeNUM--j.mjs} +0 -0
- /package/dist/{microphone-Bqmoz9_K.mjs → microphone-CsUwVP2O.mjs} +0 -0
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { a as deserializeRecord, i as topK, o as matchesFilter, r as normalize, s as serializeRecord } from "./vector-
|
|
1
|
+
import { a as deserializeRecord, i as topK, o as matchesFilter, r as normalize, s as serializeRecord } from "./vector-BJqdOci_.mjs";
|
|
2
2
|
|
|
3
3
|
//#region src/memory/chunking.ts
|
|
4
4
|
const DEFAULT_CHUNK_SIZE = 1e3;
|
|
@@ -291,4 +291,4 @@ function createMemory(options) {
|
|
|
291
291
|
|
|
292
292
|
//#endregion
|
|
293
293
|
export { createInMemoryStore as a, InMemoryStore as i, createMemory as n, applyRedaction as o, estimateTokens as r, chunkText as s, Memory as t };
|
|
294
|
-
//# sourceMappingURL=memory-
|
|
294
|
+
//# sourceMappingURL=memory-BCKc_Jfp.mjs.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"memory-Dj0J1v88.mjs","names":["chunks: string[]","candidates: { item: MemoryRecord; vector: Float32Array }[]","ranked: Scored<MemoryRecord>[]","out: MemoryRecord[]","records: MemoryRecord[]","packed: MemoryRecord[]"],"sources":["../src/memory/chunking.ts","../src/memory/redaction.ts","../src/memory/stores/memory-store.ts","../src/memory/tokens.ts","../src/memory/memory.ts"],"sourcesContent":["/**\n * Document chunking: split long text into overlapping windows before\n * embedding so retrieval can target relevant passages.\n */\n\nimport type { ChunkOptions } from \"./types.js\";\n\nconst DEFAULT_CHUNK_SIZE = 1000;\nconst DEFAULT_OVERLAP = 200;\n\n/**\n * Split `text` into overlapping character windows.\n *\n * Chunks are `chunkSize` characters with `overlap` characters shared between\n * consecutive chunks, which preserves context across boundaries. Whitespace\n * is trimmed and empty chunks are dropped. Short text returns a single chunk.\n *\n * @throws if `overlap` is not less than `chunkSize` (would loop forever).\n */\nexport function chunkText(text: string, options: ChunkOptions = {}): string[] {\n const chunkSize = options.chunkSize ?? DEFAULT_CHUNK_SIZE;\n const overlap = options.overlap ?? DEFAULT_OVERLAP;\n\n if (chunkSize <= 0) {\n throw new Error(\"chunkSize must be greater than 0\");\n }\n if (overlap < 0 || overlap >= chunkSize) {\n throw new Error(\"overlap must be >= 0 and < chunkSize\");\n }\n\n const trimmed = text.trim();\n if (trimmed.length <= chunkSize) {\n return trimmed.length > 0 ? [trimmed] : [];\n }\n\n const stride = chunkSize - overlap;\n const chunks: string[] = [];\n for (let start = 0; start < trimmed.length; start += stride) {\n const chunk = trimmed.slice(start, start + chunkSize).trim();\n if (chunk.length > 0) {\n chunks.push(chunk);\n }\n if (start + chunkSize >= trimmed.length) {\n break;\n }\n }\n return chunks;\n}\n","/**\n * Privacy redaction applied to memory text on write.\n */\n\nimport type { Redactor } from \"./types.js\";\n\nconst REDACTION_PLACEHOLDER = \"[REDACTED]\";\n\n/**\n * Apply a {@link Redactor} to `text`.\n *\n * - A {@link RegExp} replaces all matches with `[REDACTED]` (the `g` flag is\n * forced so replacement is global regardless of the supplied flags).\n * - A function is invoked and its return value used.\n * - `undefined` returns the text unchanged.\n */\nexport function applyRedaction(text: string, redact?: Redactor): string {\n if (!redact) {\n return text;\n }\n if (typeof redact === \"function\") {\n return redact(text);\n }\n const global = redact.global ? redact : new RegExp(redact.source, `${redact.flags}g`);\n return text.replace(global, REDACTION_PLACEHOLDER);\n}\n","/**\n * In-memory {@link MemoryStore} — the default backend.\n *\n * Works in both Node and the browser. Holds records in a `Map` and performs\n * a brute-force cosine top-k scan on search. Records are expected to carry\n * pre-normalized embeddings (the {@link Memory} facade normalizes on insert).\n */\n\nimport { matchesFilter } from \"../serialize.js\";\nimport type {\n MemoryRecord,\n MemorySearchResult,\n MemoryStore,\n MetadataFilter,\n StoreSearchOptions,\n} from \"../types.js\";\nimport { type Scored, topK } from \"../vector.js\";\n\nconst DEFAULT_K = 5;\n\n/** Brute-force, embedding-agnostic in-memory vector store. */\nexport class InMemoryStore implements MemoryStore {\n private readonly records = new Map<string, MemoryRecord>();\n\n add(record: MemoryRecord): Promise<void> {\n this.records.set(record.id, record);\n return Promise.resolve();\n }\n\n addMany(records: MemoryRecord[]): Promise<void> {\n for (const record of records) {\n this.records.set(record.id, record);\n }\n return Promise.resolve();\n }\n\n get(id: string): Promise<MemoryRecord | undefined> {\n return Promise.resolve(this.records.get(id));\n }\n\n search(\n queryVector: Float32Array,\n options: StoreSearchOptions = {},\n ): Promise<MemorySearchResult[]> {\n const k = options.k ?? DEFAULT_K;\n const candidates: { item: MemoryRecord; vector: Float32Array }[] = [];\n for (const record of this.records.values()) {\n if (!record.embedding) {\n continue;\n }\n if (!matchesFilter(record.metadata, options.filter)) {\n continue;\n }\n candidates.push({ item: record, vector: record.embedding });\n }\n const ranked: Scored<MemoryRecord>[] = topK(queryVector, candidates, k, options.minScore);\n return Promise.resolve(ranked.map((entry) => ({ record: entry.item, score: entry.score })));\n }\n\n delete(id: string): Promise<boolean> {\n return Promise.resolve(this.records.delete(id));\n }\n\n list(filter?: MetadataFilter): Promise<MemoryRecord[]> {\n const out: MemoryRecord[] = [];\n for (const record of this.records.values()) {\n if (matchesFilter(record.metadata, filter)) {\n out.push(record);\n }\n }\n return Promise.resolve(out);\n }\n\n clear(): Promise<void> {\n this.records.clear();\n return Promise.resolve();\n }\n\n size(): Promise<number> {\n return Promise.resolve(this.records.size);\n }\n}\n\n/** Create an {@link InMemoryStore}. */\nexport function createInMemoryStore(): MemoryStore {\n return new InMemoryStore();\n}\n","/**\n * Approximate token counting for context packing.\n *\n * We deliberately avoid a real tokenizer dependency here. The heuristic is\n * the widely used \"~4 characters per token\" rule for English-ish text, with\n * a small floor so very short strings still cost at least one token. This is\n * an estimate used only for budgeting — slightly conservative is fine, since\n * the goal is to stay under a model's context window, not to be exact.\n */\n\nconst CHARS_PER_TOKEN = 4;\n\n/**\n * Estimate the number of tokens in `text`.\n *\n * Uses the ~4-chars-per-token heuristic. Returns 0 for empty strings.\n */\nexport function estimateTokens(text: string): number {\n if (text.length === 0) {\n return 0;\n }\n return Math.max(1, Math.ceil(text.length / CHARS_PER_TOKEN));\n}\n","/**\n * The {@link Memory} facade: the small, obvious public API over a pluggable\n * {@link MemoryStore} and an injected {@link Embedder}.\n *\n * Lifecycle of a write: redact -> (optional) chunk -> embed -> normalize ->\n * store. Reads embed the query once and delegate cosine top-k to the store.\n * {@link Memory.recall} adds token-budgeted context packing on top of search.\n */\n\nimport { chunkText } from \"./chunking.js\";\nimport { applyRedaction } from \"./redaction.js\";\nimport { deserializeRecord, serializeRecord } from \"./serialize.js\";\nimport { createInMemoryStore } from \"./stores/memory-store.js\";\nimport { estimateTokens } from \"./tokens.js\";\nimport type {\n AddOptions,\n ChunkOptions,\n Embedder,\n MemoryExport,\n MemoryOptions,\n MemoryRecord,\n MemorySearchResult,\n MemoryStore,\n MetadataFilter,\n RecallOptions,\n RecallResult,\n Redactor,\n SearchOptions,\n} from \"./types.js\";\nimport { normalize } from \"./vector.js\";\n\nconst DEFAULT_RECALL_K = 20;\nconst DEFAULT_TOKEN_BUDGET = 1024;\nconst DEFAULT_SEPARATOR = \"\\n\\n\";\n\nfunction generateId(): string {\n // crypto.randomUUID exists in modern Node (>=16) and browsers.\n const cryptoObj = globalThis.crypto;\n if (cryptoObj?.randomUUID) {\n return cryptoObj.randomUUID();\n }\n return `mem-${Date.now().toString(36)}-${Math.random().toString(36).slice(2, 10)}`;\n}\n\n/**\n * On-device persistent memory with semantic recall.\n *\n * Construct via {@link createMemory}. Engine-agnostic: bring any\n * {@link Embedder} (Gerbil's native embeddings by default) and any\n * {@link MemoryStore} backend.\n */\nexport class Memory {\n private readonly embedder: Embedder;\n private readonly store: MemoryStore;\n private readonly redact?: Redactor;\n private readonly defaultChunk?: ChunkOptions;\n\n constructor(options: MemoryOptions) {\n if (typeof options.embed !== \"function\") {\n throw new Error(\"createMemory requires an `embed` function\");\n }\n this.embedder = options.embed;\n this.store = options.store ?? createInMemoryStore();\n this.redact = options.redact;\n this.defaultChunk = options.chunk;\n }\n\n /** The underlying store, for advanced use (custom queries, swapping). */\n get backend(): MemoryStore {\n return this.store;\n }\n\n /**\n * Add text to memory. Long text can be split into overlapping chunks\n * (one record per chunk). Returns the ids of the created record(s).\n */\n async add(text: string, options: AddOptions = {}): Promise<string[]> {\n const redacted = applyRedaction(text, this.redact);\n\n const chunkConfig = this.resolveChunking(options.chunk);\n const pieces = chunkConfig ? chunkText(redacted, chunkConfig) : [redacted];\n const nonEmpty = pieces.filter((piece) => piece.length > 0);\n if (nonEmpty.length === 0) {\n return [];\n }\n\n if (options.id && nonEmpty.length > 1) {\n throw new Error(\"Cannot use an explicit `id` when text is split into multiple chunks\");\n }\n\n const vectors = await this.embedder(nonEmpty);\n const createdAt = Date.now();\n const metadata = options.metadata ?? {};\n const records: MemoryRecord[] = nonEmpty.map((piece, index) => ({\n id: index === 0 && options.id ? options.id : generateId(),\n text: piece,\n embedding: normalize(vectors[index]),\n metadata,\n createdAt,\n }));\n\n await this.store.addMany(records);\n return records.map((record) => record.id);\n }\n\n private resolveChunking(chunk: AddOptions[\"chunk\"]): ChunkOptions | undefined {\n if (chunk === true) {\n return this.defaultChunk ?? {};\n }\n if (chunk && typeof chunk === \"object\") {\n return chunk;\n }\n return;\n }\n\n /** Fetch a record by id. */\n get(id: string): Promise<MemoryRecord | undefined> {\n return this.store.get(id);\n }\n\n /** Semantic search: returns the top-k records by cosine similarity. */\n async search(query: string, options: SearchOptions = {}): Promise<MemorySearchResult[]> {\n const [vector] = await this.embedder([query]);\n return this.store.search(normalize(vector), {\n k: options.k,\n filter: options.filter,\n minScore: options.minScore,\n });\n }\n\n /**\n * Retrieve relevant memories and greedily pack them into a token-budgeted\n * context block, highest-scoring first, stopping before the budget is\n * exceeded. This is the per-turn context-rebuild step that turns the store\n * into agent memory. Token counts are approximate (see {@link estimateTokens}).\n */\n async recall(query: string, options: RecallOptions = {}): Promise<RecallResult> {\n const budget = options.tokenBudget ?? DEFAULT_TOKEN_BUDGET;\n const separator = options.separator ?? DEFAULT_SEPARATOR;\n const candidates = await this.search(query, {\n k: options.k ?? DEFAULT_RECALL_K,\n filter: options.filter,\n minScore: options.minScore,\n });\n\n const separatorTokens = estimateTokens(separator);\n const packed: MemoryRecord[] = [];\n let tokensUsed = 0;\n for (const { record } of candidates) {\n const recordTokens = estimateTokens(record.text);\n const withSeparator = packed.length === 0 ? recordTokens : recordTokens + separatorTokens;\n if (tokensUsed + withSeparator > budget) {\n continue; // Try smaller later candidates rather than stop outright.\n }\n packed.push(record);\n tokensUsed += withSeparator;\n }\n\n return {\n context: packed.map((record) => record.text).join(separator),\n records: packed,\n tokensUsed,\n };\n }\n\n /** Delete a record by id. Returns `true` if it existed. */\n delete(id: string): Promise<boolean> {\n return this.store.delete(id);\n }\n\n /** List all records, optionally filtered by metadata. */\n list(filter?: MetadataFilter): Promise<MemoryRecord[]> {\n return this.store.list(filter);\n }\n\n /** Remove all records. */\n clear(): Promise<void> {\n return this.store.clear();\n }\n\n /** Total number of stored records. */\n size(): Promise<number> {\n return this.store.size();\n }\n\n /** Export the full corpus as a JSON-serializable snapshot. */\n async export(): Promise<MemoryExport> {\n const records = await this.store.list();\n return {\n version: 1,\n records: records.map(serializeRecord),\n };\n }\n\n /**\n * Import records from a snapshot produced by {@link Memory.export}.\n * Existing records with the same id are overwritten.\n */\n async import(snapshot: MemoryExport): Promise<void> {\n const records = snapshot.records.map(deserializeRecord);\n await this.store.addMany(records);\n }\n}\n\n/**\n * Create a {@link Memory} instance.\n *\n * @example\n * ```ts\n * const mem = createMemory({ embed, store });\n * await mem.add(\"Paris is the capital of France\", { metadata: { topic: \"geo\" } });\n * const hits = await mem.search(\"French capital\", { k: 3 });\n * const { context } = await mem.recall(\"French capital\", { tokenBudget: 512 });\n * ```\n */\nexport function createMemory(options: MemoryOptions): Memory {\n return new Memory(options);\n}\n"],"mappings":";;;AAOA,MAAM,qBAAqB;AAC3B,MAAM,kBAAkB;;;;;;;;;;AAWxB,SAAgB,UAAU,MAAc,UAAwB,EAAE,EAAY;CAC5E,MAAM,YAAY,QAAQ,aAAa;CACvC,MAAM,UAAU,QAAQ,WAAW;AAEnC,KAAI,aAAa,EACf,OAAM,IAAI,MAAM,mCAAmC;AAErD,KAAI,UAAU,KAAK,WAAW,UAC5B,OAAM,IAAI,MAAM,uCAAuC;CAGzD,MAAM,UAAU,KAAK,MAAM;AAC3B,KAAI,QAAQ,UAAU,UACpB,QAAO,QAAQ,SAAS,IAAI,CAAC,QAAQ,GAAG,EAAE;CAG5C,MAAM,SAAS,YAAY;CAC3B,MAAMA,SAAmB,EAAE;AAC3B,MAAK,IAAI,QAAQ,GAAG,QAAQ,QAAQ,QAAQ,SAAS,QAAQ;EAC3D,MAAM,QAAQ,QAAQ,MAAM,OAAO,QAAQ,UAAU,CAAC,MAAM;AAC5D,MAAI,MAAM,SAAS,EACjB,QAAO,KAAK,MAAM;AAEpB,MAAI,QAAQ,aAAa,QAAQ,OAC/B;;AAGJ,QAAO;;;;;ACxCT,MAAM,wBAAwB;;;;;;;;;AAU9B,SAAgB,eAAe,MAAc,QAA2B;AACtE,KAAI,CAAC,OACH,QAAO;AAET,KAAI,OAAO,WAAW,WACpB,QAAO,OAAO,KAAK;CAErB,MAAM,SAAS,OAAO,SAAS,SAAS,IAAI,OAAO,OAAO,QAAQ,GAAG,OAAO,MAAM,GAAG;AACrF,QAAO,KAAK,QAAQ,QAAQ,sBAAsB;;;;;;;;;;;;ACNpD,MAAM,YAAY;;AAGlB,IAAa,gBAAb,MAAkD;CAChD,AAAiB,0BAAU,IAAI,KAA2B;CAE1D,IAAI,QAAqC;AACvC,OAAK,QAAQ,IAAI,OAAO,IAAI,OAAO;AACnC,SAAO,QAAQ,SAAS;;CAG1B,QAAQ,SAAwC;AAC9C,OAAK,MAAM,UAAU,QACnB,MAAK,QAAQ,IAAI,OAAO,IAAI,OAAO;AAErC,SAAO,QAAQ,SAAS;;CAG1B,IAAI,IAA+C;AACjD,SAAO,QAAQ,QAAQ,KAAK,QAAQ,IAAI,GAAG,CAAC;;CAG9C,OACE,aACA,UAA8B,EAAE,EACD;EAC/B,MAAM,IAAI,QAAQ,KAAK;EACvB,MAAMC,aAA6D,EAAE;AACrE,OAAK,MAAM,UAAU,KAAK,QAAQ,QAAQ,EAAE;AAC1C,OAAI,CAAC,OAAO,UACV;AAEF,OAAI,CAAC,cAAc,OAAO,UAAU,QAAQ,OAAO,CACjD;AAEF,cAAW,KAAK;IAAE,MAAM;IAAQ,QAAQ,OAAO;IAAW,CAAC;;EAE7D,MAAMC,SAAiC,KAAK,aAAa,YAAY,GAAG,QAAQ,SAAS;AACzF,SAAO,QAAQ,QAAQ,OAAO,KAAK,WAAW;GAAE,QAAQ,MAAM;GAAM,OAAO,MAAM;GAAO,EAAE,CAAC;;CAG7F,OAAO,IAA8B;AACnC,SAAO,QAAQ,QAAQ,KAAK,QAAQ,OAAO,GAAG,CAAC;;CAGjD,KAAK,QAAkD;EACrD,MAAMC,MAAsB,EAAE;AAC9B,OAAK,MAAM,UAAU,KAAK,QAAQ,QAAQ,CACxC,KAAI,cAAc,OAAO,UAAU,OAAO,CACxC,KAAI,KAAK,OAAO;AAGpB,SAAO,QAAQ,QAAQ,IAAI;;CAG7B,QAAuB;AACrB,OAAK,QAAQ,OAAO;AACpB,SAAO,QAAQ,SAAS;;CAG1B,OAAwB;AACtB,SAAO,QAAQ,QAAQ,KAAK,QAAQ,KAAK;;;;AAK7C,SAAgB,sBAAmC;AACjD,QAAO,IAAI,eAAe;;;;;;;;;;;;;;AC3E5B,MAAM,kBAAkB;;;;;;AAOxB,SAAgB,eAAe,MAAsB;AACnD,KAAI,KAAK,WAAW,EAClB,QAAO;AAET,QAAO,KAAK,IAAI,GAAG,KAAK,KAAK,KAAK,SAAS,gBAAgB,CAAC;;;;;;;;;;;;;ACU9D,MAAM,mBAAmB;AACzB,MAAM,uBAAuB;AAC7B,MAAM,oBAAoB;AAE1B,SAAS,aAAqB;CAE5B,MAAM,YAAY,WAAW;AAC7B,KAAI,WAAW,WACb,QAAO,UAAU,YAAY;AAE/B,QAAO,OAAO,KAAK,KAAK,CAAC,SAAS,GAAG,CAAC,GAAG,KAAK,QAAQ,CAAC,SAAS,GAAG,CAAC,MAAM,GAAG,GAAG;;;;;;;;;AAUlF,IAAa,SAAb,MAAoB;CAClB,AAAiB;CACjB,AAAiB;CACjB,AAAiB;CACjB,AAAiB;CAEjB,YAAY,SAAwB;AAClC,MAAI,OAAO,QAAQ,UAAU,WAC3B,OAAM,IAAI,MAAM,4CAA4C;AAE9D,OAAK,WAAW,QAAQ;AACxB,OAAK,QAAQ,QAAQ,SAAS,qBAAqB;AACnD,OAAK,SAAS,QAAQ;AACtB,OAAK,eAAe,QAAQ;;;CAI9B,IAAI,UAAuB;AACzB,SAAO,KAAK;;;;;;CAOd,MAAM,IAAI,MAAc,UAAsB,EAAE,EAAqB;EACnE,MAAM,WAAW,eAAe,MAAM,KAAK,OAAO;EAElD,MAAM,cAAc,KAAK,gBAAgB,QAAQ,MAAM;EAEvD,MAAM,YADS,cAAc,UAAU,UAAU,YAAY,GAAG,CAAC,SAAS,EAClD,QAAQ,UAAU,MAAM,SAAS,EAAE;AAC3D,MAAI,SAAS,WAAW,EACtB,QAAO,EAAE;AAGX,MAAI,QAAQ,MAAM,SAAS,SAAS,EAClC,OAAM,IAAI,MAAM,sEAAsE;EAGxF,MAAM,UAAU,MAAM,KAAK,SAAS,SAAS;EAC7C,MAAM,YAAY,KAAK,KAAK;EAC5B,MAAM,WAAW,QAAQ,YAAY,EAAE;EACvC,MAAMC,UAA0B,SAAS,KAAK,OAAO,WAAW;GAC9D,IAAI,UAAU,KAAK,QAAQ,KAAK,QAAQ,KAAK,YAAY;GACzD,MAAM;GACN,WAAW,UAAU,QAAQ,OAAO;GACpC;GACA;GACD,EAAE;AAEH,QAAM,KAAK,MAAM,QAAQ,QAAQ;AACjC,SAAO,QAAQ,KAAK,WAAW,OAAO,GAAG;;CAG3C,AAAQ,gBAAgB,OAAsD;AAC5E,MAAI,UAAU,KACZ,QAAO,KAAK,gBAAgB,EAAE;AAEhC,MAAI,SAAS,OAAO,UAAU,SAC5B,QAAO;;;CAMX,IAAI,IAA+C;AACjD,SAAO,KAAK,MAAM,IAAI,GAAG;;;CAI3B,MAAM,OAAO,OAAe,UAAyB,EAAE,EAAiC;EACtF,MAAM,CAAC,UAAU,MAAM,KAAK,SAAS,CAAC,MAAM,CAAC;AAC7C,SAAO,KAAK,MAAM,OAAO,UAAU,OAAO,EAAE;GAC1C,GAAG,QAAQ;GACX,QAAQ,QAAQ;GAChB,UAAU,QAAQ;GACnB,CAAC;;;;;;;;CASJ,MAAM,OAAO,OAAe,UAAyB,EAAE,EAAyB;EAC9E,MAAM,SAAS,QAAQ,eAAe;EACtC,MAAM,YAAY,QAAQ,aAAa;EACvC,MAAM,aAAa,MAAM,KAAK,OAAO,OAAO;GAC1C,GAAG,QAAQ,KAAK;GAChB,QAAQ,QAAQ;GAChB,UAAU,QAAQ;GACnB,CAAC;EAEF,MAAM,kBAAkB,eAAe,UAAU;EACjD,MAAMC,SAAyB,EAAE;EACjC,IAAI,aAAa;AACjB,OAAK,MAAM,EAAE,YAAY,YAAY;GACnC,MAAM,eAAe,eAAe,OAAO,KAAK;GAChD,MAAM,gBAAgB,OAAO,WAAW,IAAI,eAAe,eAAe;AAC1E,OAAI,aAAa,gBAAgB,OAC/B;AAEF,UAAO,KAAK,OAAO;AACnB,iBAAc;;AAGhB,SAAO;GACL,SAAS,OAAO,KAAK,WAAW,OAAO,KAAK,CAAC,KAAK,UAAU;GAC5D,SAAS;GACT;GACD;;;CAIH,OAAO,IAA8B;AACnC,SAAO,KAAK,MAAM,OAAO,GAAG;;;CAI9B,KAAK,QAAkD;AACrD,SAAO,KAAK,MAAM,KAAK,OAAO;;;CAIhC,QAAuB;AACrB,SAAO,KAAK,MAAM,OAAO;;;CAI3B,OAAwB;AACtB,SAAO,KAAK,MAAM,MAAM;;;CAI1B,MAAM,SAAgC;AAEpC,SAAO;GACL,SAAS;GACT,UAHc,MAAM,KAAK,MAAM,MAAM,EAGpB,IAAI,gBAAgB;GACtC;;;;;;CAOH,MAAM,OAAO,UAAuC;EAClD,MAAM,UAAU,SAAS,QAAQ,IAAI,kBAAkB;AACvD,QAAM,KAAK,MAAM,QAAQ,QAAQ;;;;;;;;;;;;;;AAerC,SAAgB,aAAa,SAAgC;AAC3D,QAAO,IAAI,OAAO,QAAQ"}
|
|
1
|
+
{"version":3,"file":"memory-BCKc_Jfp.mjs","names":["chunks: string[]","candidates: { item: MemoryRecord; vector: Float32Array }[]","ranked: Scored<MemoryRecord>[]","out: MemoryRecord[]","records: MemoryRecord[]","packed: MemoryRecord[]"],"sources":["../src/memory/chunking.ts","../src/memory/redaction.ts","../src/memory/stores/memory-store.ts","../src/memory/tokens.ts","../src/memory/memory.ts"],"sourcesContent":["/**\n * Document chunking: split long text into overlapping windows before\n * embedding so retrieval can target relevant passages.\n */\n\nimport type { ChunkOptions } from \"./types.js\";\n\nconst DEFAULT_CHUNK_SIZE = 1000;\nconst DEFAULT_OVERLAP = 200;\n\n/**\n * Split `text` into overlapping character windows.\n *\n * Chunks are `chunkSize` characters with `overlap` characters shared between\n * consecutive chunks, which preserves context across boundaries. Whitespace\n * is trimmed and empty chunks are dropped. Short text returns a single chunk.\n *\n * @throws if `overlap` is not less than `chunkSize` (would loop forever).\n */\nexport function chunkText(text: string, options: ChunkOptions = {}): string[] {\n const chunkSize = options.chunkSize ?? DEFAULT_CHUNK_SIZE;\n const overlap = options.overlap ?? DEFAULT_OVERLAP;\n\n if (chunkSize <= 0) {\n throw new Error(\"chunkSize must be greater than 0\");\n }\n if (overlap < 0 || overlap >= chunkSize) {\n throw new Error(\"overlap must be >= 0 and < chunkSize\");\n }\n\n const trimmed = text.trim();\n if (trimmed.length <= chunkSize) {\n return trimmed.length > 0 ? [trimmed] : [];\n }\n\n const stride = chunkSize - overlap;\n const chunks: string[] = [];\n for (let start = 0; start < trimmed.length; start += stride) {\n const chunk = trimmed.slice(start, start + chunkSize).trim();\n if (chunk.length > 0) {\n chunks.push(chunk);\n }\n if (start + chunkSize >= trimmed.length) {\n break;\n }\n }\n return chunks;\n}\n","/**\n * Privacy redaction applied to memory text on write.\n */\n\nimport type { Redactor } from \"./types.js\";\n\nconst REDACTION_PLACEHOLDER = \"[REDACTED]\";\n\n/**\n * Apply a {@link Redactor} to `text`.\n *\n * - A {@link RegExp} replaces all matches with `[REDACTED]` (the `g` flag is\n * forced so replacement is global regardless of the supplied flags).\n * - A function is invoked and its return value used.\n * - `undefined` returns the text unchanged.\n */\nexport function applyRedaction(text: string, redact?: Redactor): string {\n if (!redact) {\n return text;\n }\n if (typeof redact === \"function\") {\n return redact(text);\n }\n const global = redact.global ? redact : new RegExp(redact.source, `${redact.flags}g`);\n return text.replace(global, REDACTION_PLACEHOLDER);\n}\n","/**\n * In-memory {@link MemoryStore} — the default backend.\n *\n * Works in both Node and the browser. Holds records in a `Map` and performs\n * a brute-force cosine top-k scan on search. Records are expected to carry\n * pre-normalized embeddings (the {@link Memory} facade normalizes on insert).\n */\n\nimport { matchesFilter } from \"../serialize.js\";\nimport type {\n MemoryRecord,\n MemorySearchResult,\n MemoryStore,\n MetadataFilter,\n StoreSearchOptions,\n} from \"../types.js\";\nimport { type Scored, topK } from \"../vector.js\";\n\nconst DEFAULT_K = 5;\n\n/** Brute-force, embedding-agnostic in-memory vector store. */\nexport class InMemoryStore implements MemoryStore {\n private readonly records = new Map<string, MemoryRecord>();\n\n add(record: MemoryRecord): Promise<void> {\n this.records.set(record.id, record);\n return Promise.resolve();\n }\n\n addMany(records: MemoryRecord[]): Promise<void> {\n for (const record of records) {\n this.records.set(record.id, record);\n }\n return Promise.resolve();\n }\n\n get(id: string): Promise<MemoryRecord | undefined> {\n return Promise.resolve(this.records.get(id));\n }\n\n search(\n queryVector: Float32Array,\n options: StoreSearchOptions = {},\n ): Promise<MemorySearchResult[]> {\n const k = options.k ?? DEFAULT_K;\n const candidates: { item: MemoryRecord; vector: Float32Array }[] = [];\n for (const record of this.records.values()) {\n if (!record.embedding) {\n continue;\n }\n if (!matchesFilter(record.metadata, options.filter)) {\n continue;\n }\n candidates.push({ item: record, vector: record.embedding });\n }\n const ranked: Scored<MemoryRecord>[] = topK(queryVector, candidates, k, options.minScore);\n return Promise.resolve(ranked.map((entry) => ({ record: entry.item, score: entry.score })));\n }\n\n delete(id: string): Promise<boolean> {\n return Promise.resolve(this.records.delete(id));\n }\n\n list(filter?: MetadataFilter): Promise<MemoryRecord[]> {\n const out: MemoryRecord[] = [];\n for (const record of this.records.values()) {\n if (matchesFilter(record.metadata, filter)) {\n out.push(record);\n }\n }\n return Promise.resolve(out);\n }\n\n clear(): Promise<void> {\n this.records.clear();\n return Promise.resolve();\n }\n\n size(): Promise<number> {\n return Promise.resolve(this.records.size);\n }\n}\n\n/** Create an {@link InMemoryStore}. */\nexport function createInMemoryStore(): MemoryStore {\n return new InMemoryStore();\n}\n","/**\n * Approximate token counting for context packing.\n *\n * We deliberately avoid a real tokenizer dependency here. The heuristic is\n * the widely used \"~4 characters per token\" rule for English-ish text, with\n * a small floor so very short strings still cost at least one token. This is\n * an estimate used only for budgeting — slightly conservative is fine, since\n * the goal is to stay under a model's context window, not to be exact.\n */\n\nconst CHARS_PER_TOKEN = 4;\n\n/**\n * Estimate the number of tokens in `text`.\n *\n * Uses the ~4-chars-per-token heuristic. Returns 0 for empty strings.\n */\nexport function estimateTokens(text: string): number {\n if (text.length === 0) {\n return 0;\n }\n return Math.max(1, Math.ceil(text.length / CHARS_PER_TOKEN));\n}\n","/**\n * The {@link Memory} facade: the small, obvious public API over a pluggable\n * {@link MemoryStore} and an injected {@link Embedder}.\n *\n * Lifecycle of a write: redact -> (optional) chunk -> embed -> normalize ->\n * store. Reads embed the query once and delegate cosine top-k to the store.\n * {@link Memory.recall} adds token-budgeted context packing on top of search.\n */\n\nimport { chunkText } from \"./chunking.js\";\nimport { applyRedaction } from \"./redaction.js\";\nimport { deserializeRecord, serializeRecord } from \"./serialize.js\";\nimport { createInMemoryStore } from \"./stores/memory-store.js\";\nimport { estimateTokens } from \"./tokens.js\";\nimport type {\n AddOptions,\n ChunkOptions,\n Embedder,\n MemoryExport,\n MemoryOptions,\n MemoryRecord,\n MemorySearchResult,\n MemoryStore,\n MetadataFilter,\n RecallOptions,\n RecallResult,\n Redactor,\n SearchOptions,\n} from \"./types.js\";\nimport { normalize } from \"./vector.js\";\n\nconst DEFAULT_RECALL_K = 20;\nconst DEFAULT_TOKEN_BUDGET = 1024;\nconst DEFAULT_SEPARATOR = \"\\n\\n\";\n\nfunction generateId(): string {\n // crypto.randomUUID exists in modern Node (>=16) and browsers.\n const cryptoObj = globalThis.crypto;\n if (cryptoObj?.randomUUID) {\n return cryptoObj.randomUUID();\n }\n return `mem-${Date.now().toString(36)}-${Math.random().toString(36).slice(2, 10)}`;\n}\n\n/**\n * On-device persistent memory with semantic recall.\n *\n * Construct via {@link createMemory}. Engine-agnostic: bring any\n * {@link Embedder} (Gerbil's native embeddings by default) and any\n * {@link MemoryStore} backend.\n */\nexport class Memory {\n private readonly embedder: Embedder;\n private readonly store: MemoryStore;\n private readonly redact?: Redactor;\n private readonly defaultChunk?: ChunkOptions;\n\n constructor(options: MemoryOptions) {\n if (typeof options.embed !== \"function\") {\n throw new Error(\"createMemory requires an `embed` function\");\n }\n this.embedder = options.embed;\n this.store = options.store ?? createInMemoryStore();\n this.redact = options.redact;\n this.defaultChunk = options.chunk;\n }\n\n /** The underlying store, for advanced use (custom queries, swapping). */\n get backend(): MemoryStore {\n return this.store;\n }\n\n /**\n * Add text to memory. Long text can be split into overlapping chunks\n * (one record per chunk). Returns the ids of the created record(s).\n */\n async add(text: string, options: AddOptions = {}): Promise<string[]> {\n const redacted = applyRedaction(text, this.redact);\n\n const chunkConfig = this.resolveChunking(options.chunk);\n const pieces = chunkConfig ? chunkText(redacted, chunkConfig) : [redacted];\n const nonEmpty = pieces.filter((piece) => piece.length > 0);\n if (nonEmpty.length === 0) {\n return [];\n }\n\n if (options.id && nonEmpty.length > 1) {\n throw new Error(\"Cannot use an explicit `id` when text is split into multiple chunks\");\n }\n\n const vectors = await this.embedder(nonEmpty);\n const createdAt = Date.now();\n const metadata = options.metadata ?? {};\n const records: MemoryRecord[] = nonEmpty.map((piece, index) => ({\n id: index === 0 && options.id ? options.id : generateId(),\n text: piece,\n embedding: normalize(vectors[index]),\n metadata,\n createdAt,\n }));\n\n await this.store.addMany(records);\n return records.map((record) => record.id);\n }\n\n private resolveChunking(chunk: AddOptions[\"chunk\"]): ChunkOptions | undefined {\n if (chunk === true) {\n return this.defaultChunk ?? {};\n }\n if (chunk && typeof chunk === \"object\") {\n return chunk;\n }\n return;\n }\n\n /** Fetch a record by id. */\n get(id: string): Promise<MemoryRecord | undefined> {\n return this.store.get(id);\n }\n\n /** Semantic search: returns the top-k records by cosine similarity. */\n async search(query: string, options: SearchOptions = {}): Promise<MemorySearchResult[]> {\n const [vector] = await this.embedder([query]);\n return this.store.search(normalize(vector), {\n k: options.k,\n filter: options.filter,\n minScore: options.minScore,\n });\n }\n\n /**\n * Retrieve relevant memories and greedily pack them into a token-budgeted\n * context block, highest-scoring first, stopping before the budget is\n * exceeded. This is the per-turn context-rebuild step that turns the store\n * into agent memory. Token counts are approximate (see {@link estimateTokens}).\n */\n async recall(query: string, options: RecallOptions = {}): Promise<RecallResult> {\n const budget = options.tokenBudget ?? DEFAULT_TOKEN_BUDGET;\n const separator = options.separator ?? DEFAULT_SEPARATOR;\n const candidates = await this.search(query, {\n k: options.k ?? DEFAULT_RECALL_K,\n filter: options.filter,\n minScore: options.minScore,\n });\n\n const separatorTokens = estimateTokens(separator);\n const packed: MemoryRecord[] = [];\n let tokensUsed = 0;\n for (const { record } of candidates) {\n const recordTokens = estimateTokens(record.text);\n const withSeparator = packed.length === 0 ? recordTokens : recordTokens + separatorTokens;\n if (tokensUsed + withSeparator > budget) {\n continue; // Try smaller later candidates rather than stop outright.\n }\n packed.push(record);\n tokensUsed += withSeparator;\n }\n\n return {\n context: packed.map((record) => record.text).join(separator),\n records: packed,\n tokensUsed,\n };\n }\n\n /** Delete a record by id. Returns `true` if it existed. */\n delete(id: string): Promise<boolean> {\n return this.store.delete(id);\n }\n\n /** List all records, optionally filtered by metadata. */\n list(filter?: MetadataFilter): Promise<MemoryRecord[]> {\n return this.store.list(filter);\n }\n\n /** Remove all records. */\n clear(): Promise<void> {\n return this.store.clear();\n }\n\n /** Total number of stored records. */\n size(): Promise<number> {\n return this.store.size();\n }\n\n /** Export the full corpus as a JSON-serializable snapshot. */\n async export(): Promise<MemoryExport> {\n const records = await this.store.list();\n return {\n version: 1,\n records: records.map(serializeRecord),\n };\n }\n\n /**\n * Import records from a snapshot produced by {@link Memory.export}.\n * Existing records with the same id are overwritten.\n */\n async import(snapshot: MemoryExport): Promise<void> {\n const records = snapshot.records.map(deserializeRecord);\n await this.store.addMany(records);\n }\n}\n\n/**\n * Create a {@link Memory} instance.\n *\n * @example\n * ```ts\n * const mem = createMemory({ embed, store });\n * await mem.add(\"Paris is the capital of France\", { metadata: { topic: \"geo\" } });\n * const hits = await mem.search(\"French capital\", { k: 3 });\n * const { context } = await mem.recall(\"French capital\", { tokenBudget: 512 });\n * ```\n */\nexport function createMemory(options: MemoryOptions): Memory {\n return new Memory(options);\n}\n"],"mappings":";;;AAOA,MAAM,qBAAqB;AAC3B,MAAM,kBAAkB;;;;;;;;;;AAWxB,SAAgB,UAAU,MAAc,UAAwB,EAAE,EAAY;CAC5E,MAAM,YAAY,QAAQ,aAAa;CACvC,MAAM,UAAU,QAAQ,WAAW;AAEnC,KAAI,aAAa,EACf,OAAM,IAAI,MAAM,mCAAmC;AAErD,KAAI,UAAU,KAAK,WAAW,UAC5B,OAAM,IAAI,MAAM,uCAAuC;CAGzD,MAAM,UAAU,KAAK,MAAM;AAC3B,KAAI,QAAQ,UAAU,UACpB,QAAO,QAAQ,SAAS,IAAI,CAAC,QAAQ,GAAG,EAAE;CAG5C,MAAM,SAAS,YAAY;CAC3B,MAAMA,SAAmB,EAAE;AAC3B,MAAK,IAAI,QAAQ,GAAG,QAAQ,QAAQ,QAAQ,SAAS,QAAQ;EAC3D,MAAM,QAAQ,QAAQ,MAAM,OAAO,QAAQ,UAAU,CAAC,MAAM;AAC5D,MAAI,MAAM,SAAS,EACjB,QAAO,KAAK,MAAM;AAEpB,MAAI,QAAQ,aAAa,QAAQ,OAC/B;;AAGJ,QAAO;;;;;ACxCT,MAAM,wBAAwB;;;;;;;;;AAU9B,SAAgB,eAAe,MAAc,QAA2B;AACtE,KAAI,CAAC,OACH,QAAO;AAET,KAAI,OAAO,WAAW,WACpB,QAAO,OAAO,KAAK;CAErB,MAAM,SAAS,OAAO,SAAS,SAAS,IAAI,OAAO,OAAO,QAAQ,GAAG,OAAO,MAAM,GAAG;AACrF,QAAO,KAAK,QAAQ,QAAQ,sBAAsB;;;;;;;;;;;;ACNpD,MAAM,YAAY;;AAGlB,IAAa,gBAAb,MAAkD;CAChD,AAAiB,0BAAU,IAAI,KAA2B;CAE1D,IAAI,QAAqC;AACvC,OAAK,QAAQ,IAAI,OAAO,IAAI,OAAO;AACnC,SAAO,QAAQ,SAAS;;CAG1B,QAAQ,SAAwC;AAC9C,OAAK,MAAM,UAAU,QACnB,MAAK,QAAQ,IAAI,OAAO,IAAI,OAAO;AAErC,SAAO,QAAQ,SAAS;;CAG1B,IAAI,IAA+C;AACjD,SAAO,QAAQ,QAAQ,KAAK,QAAQ,IAAI,GAAG,CAAC;;CAG9C,OACE,aACA,UAA8B,EAAE,EACD;EAC/B,MAAM,IAAI,QAAQ,KAAK;EACvB,MAAMC,aAA6D,EAAE;AACrE,OAAK,MAAM,UAAU,KAAK,QAAQ,QAAQ,EAAE;AAC1C,OAAI,CAAC,OAAO,UACV;AAEF,OAAI,CAAC,cAAc,OAAO,UAAU,QAAQ,OAAO,CACjD;AAEF,cAAW,KAAK;IAAE,MAAM;IAAQ,QAAQ,OAAO;IAAW,CAAC;;EAE7D,MAAMC,SAAiC,KAAK,aAAa,YAAY,GAAG,QAAQ,SAAS;AACzF,SAAO,QAAQ,QAAQ,OAAO,KAAK,WAAW;GAAE,QAAQ,MAAM;GAAM,OAAO,MAAM;GAAO,EAAE,CAAC;;CAG7F,OAAO,IAA8B;AACnC,SAAO,QAAQ,QAAQ,KAAK,QAAQ,OAAO,GAAG,CAAC;;CAGjD,KAAK,QAAkD;EACrD,MAAMC,MAAsB,EAAE;AAC9B,OAAK,MAAM,UAAU,KAAK,QAAQ,QAAQ,CACxC,KAAI,cAAc,OAAO,UAAU,OAAO,CACxC,KAAI,KAAK,OAAO;AAGpB,SAAO,QAAQ,QAAQ,IAAI;;CAG7B,QAAuB;AACrB,OAAK,QAAQ,OAAO;AACpB,SAAO,QAAQ,SAAS;;CAG1B,OAAwB;AACtB,SAAO,QAAQ,QAAQ,KAAK,QAAQ,KAAK;;;;AAK7C,SAAgB,sBAAmC;AACjD,QAAO,IAAI,eAAe;;;;;;;;;;;;;;AC3E5B,MAAM,kBAAkB;;;;;;AAOxB,SAAgB,eAAe,MAAsB;AACnD,KAAI,KAAK,WAAW,EAClB,QAAO;AAET,QAAO,KAAK,IAAI,GAAG,KAAK,KAAK,KAAK,SAAS,gBAAgB,CAAC;;;;;;;;;;;;;ACU9D,MAAM,mBAAmB;AACzB,MAAM,uBAAuB;AAC7B,MAAM,oBAAoB;AAE1B,SAAS,aAAqB;CAE5B,MAAM,YAAY,WAAW;AAC7B,KAAI,WAAW,WACb,QAAO,UAAU,YAAY;AAE/B,QAAO,OAAO,KAAK,KAAK,CAAC,SAAS,GAAG,CAAC,GAAG,KAAK,QAAQ,CAAC,SAAS,GAAG,CAAC,MAAM,GAAG,GAAG;;;;;;;;;AAUlF,IAAa,SAAb,MAAoB;CAClB,AAAiB;CACjB,AAAiB;CACjB,AAAiB;CACjB,AAAiB;CAEjB,YAAY,SAAwB;AAClC,MAAI,OAAO,QAAQ,UAAU,WAC3B,OAAM,IAAI,MAAM,4CAA4C;AAE9D,OAAK,WAAW,QAAQ;AACxB,OAAK,QAAQ,QAAQ,SAAS,qBAAqB;AACnD,OAAK,SAAS,QAAQ;AACtB,OAAK,eAAe,QAAQ;;;CAI9B,IAAI,UAAuB;AACzB,SAAO,KAAK;;;;;;CAOd,MAAM,IAAI,MAAc,UAAsB,EAAE,EAAqB;EACnE,MAAM,WAAW,eAAe,MAAM,KAAK,OAAO;EAElD,MAAM,cAAc,KAAK,gBAAgB,QAAQ,MAAM;EAEvD,MAAM,YADS,cAAc,UAAU,UAAU,YAAY,GAAG,CAAC,SAAS,EAClD,QAAQ,UAAU,MAAM,SAAS,EAAE;AAC3D,MAAI,SAAS,WAAW,EACtB,QAAO,EAAE;AAGX,MAAI,QAAQ,MAAM,SAAS,SAAS,EAClC,OAAM,IAAI,MAAM,sEAAsE;EAGxF,MAAM,UAAU,MAAM,KAAK,SAAS,SAAS;EAC7C,MAAM,YAAY,KAAK,KAAK;EAC5B,MAAM,WAAW,QAAQ,YAAY,EAAE;EACvC,MAAMC,UAA0B,SAAS,KAAK,OAAO,WAAW;GAC9D,IAAI,UAAU,KAAK,QAAQ,KAAK,QAAQ,KAAK,YAAY;GACzD,MAAM;GACN,WAAW,UAAU,QAAQ,OAAO;GACpC;GACA;GACD,EAAE;AAEH,QAAM,KAAK,MAAM,QAAQ,QAAQ;AACjC,SAAO,QAAQ,KAAK,WAAW,OAAO,GAAG;;CAG3C,AAAQ,gBAAgB,OAAsD;AAC5E,MAAI,UAAU,KACZ,QAAO,KAAK,gBAAgB,EAAE;AAEhC,MAAI,SAAS,OAAO,UAAU,SAC5B,QAAO;;;CAMX,IAAI,IAA+C;AACjD,SAAO,KAAK,MAAM,IAAI,GAAG;;;CAI3B,MAAM,OAAO,OAAe,UAAyB,EAAE,EAAiC;EACtF,MAAM,CAAC,UAAU,MAAM,KAAK,SAAS,CAAC,MAAM,CAAC;AAC7C,SAAO,KAAK,MAAM,OAAO,UAAU,OAAO,EAAE;GAC1C,GAAG,QAAQ;GACX,QAAQ,QAAQ;GAChB,UAAU,QAAQ;GACnB,CAAC;;;;;;;;CASJ,MAAM,OAAO,OAAe,UAAyB,EAAE,EAAyB;EAC9E,MAAM,SAAS,QAAQ,eAAe;EACtC,MAAM,YAAY,QAAQ,aAAa;EACvC,MAAM,aAAa,MAAM,KAAK,OAAO,OAAO;GAC1C,GAAG,QAAQ,KAAK;GAChB,QAAQ,QAAQ;GAChB,UAAU,QAAQ;GACnB,CAAC;EAEF,MAAM,kBAAkB,eAAe,UAAU;EACjD,MAAMC,SAAyB,EAAE;EACjC,IAAI,aAAa;AACjB,OAAK,MAAM,EAAE,YAAY,YAAY;GACnC,MAAM,eAAe,eAAe,OAAO,KAAK;GAChD,MAAM,gBAAgB,OAAO,WAAW,IAAI,eAAe,eAAe;AAC1E,OAAI,aAAa,gBAAgB,OAC/B;AAEF,UAAO,KAAK,OAAO;AACnB,iBAAc;;AAGhB,SAAO;GACL,SAAS,OAAO,KAAK,WAAW,OAAO,KAAK,CAAC,KAAK,UAAU;GAC5D,SAAS;GACT;GACD;;;CAIH,OAAO,IAA8B;AACnC,SAAO,KAAK,MAAM,OAAO,GAAG;;;CAI9B,KAAK,QAAkD;AACrD,SAAO,KAAK,MAAM,KAAK,OAAO;;;CAIhC,QAAuB;AACrB,SAAO,KAAK,MAAM,OAAO;;;CAI3B,OAAwB;AACtB,SAAO,KAAK,MAAM,MAAM;;;CAI1B,MAAM,SAAgC;AAEpC,SAAO;GACL,SAAS;GACT,UAHc,MAAM,KAAK,MAAM,MAAM,EAGpB,IAAI,gBAAgB;GACtC;;;;;;CAOH,MAAM,OAAO,UAAuC;EAClD,MAAM,UAAU,SAAS,QAAQ,IAAI,kBAAkB;AACvD,QAAM,KAAK,MAAM,QAAQ,QAAQ;;;;;;;;;;;;;;AAerC,SAAgB,aAAa,SAAgC;AAC3D,QAAO,IAAI,OAAO,QAAQ"}
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { a as deserializeRecord, i as topK, o as matchesFilter, s as serializeRecord } from "./vector-
|
|
1
|
+
import { a as deserializeRecord, i as topK, o as matchesFilter, s as serializeRecord } from "./vector-BJqdOci_.mjs";
|
|
2
2
|
import { dirname } from "node:path";
|
|
3
3
|
import { mkdir, readFile, writeFile } from "node:fs/promises";
|
|
4
4
|
|
|
@@ -129,4 +129,4 @@ function createFileStore(path$1) {
|
|
|
129
129
|
|
|
130
130
|
//#endregion
|
|
131
131
|
export { createFileStore as n, createGerbilEmbedder as r, FileStore as t };
|
|
132
|
-
//# sourceMappingURL=memory-
|
|
132
|
+
//# sourceMappingURL=memory-Dnjhc1du.mjs.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"memory-
|
|
1
|
+
{"version":3,"file":"memory-Dnjhc1du.mjs","names":["path","snapshot: MemoryExport","candidates: { item: MemoryRecord; vector: Float32Array }[]","out: MemoryRecord[]"],"sources":["../src/memory/gerbil-embedder.ts","../src/memory/stores/file-store.ts"],"sourcesContent":["/**\n * Adapter that wires Gerbil's native embeddings into the memory module.\n *\n * The memory module only needs `(texts) => Promise<Float32Array[]>`. This\n * adapter accepts anything that exposes an `embedBatch` returning records with\n * a numeric `vector` — which covers a {@link Gerbil} instance, the one-liner\n * `embedBatch`, and the browser `useEmbedding().embedBatch`. Keeping the\n * contract structural avoids a hard dependency on the core engine types and\n * keeps the module engine-agnostic.\n */\n\nimport type { Embedder } from \"./types.js\";\n\n/** Minimal shape of a batch embedder returning `{ vector }` records. */\nexport type BatchEmbedderLike = {\n embedBatch(texts: string[]): Promise<{ vector: number[] }[]>;\n};\n\n/**\n * Build an {@link Embedder} from a Gerbil instance (or any object exposing a\n * compatible `embedBatch`). Vectors are converted to {@link Float32Array}.\n *\n * @example\n * ```ts\n * import { Gerbil } from \"@tryhamster/gerbil\";\n * import { createMemory, createGerbilEmbedder } from \"@tryhamster/gerbil/memory\";\n *\n * const g = new Gerbil();\n * await g.loadModel(\"embeddinggemma-300m\");\n * const mem = createMemory({ embed: createGerbilEmbedder(g) });\n * ```\n */\nexport function createGerbilEmbedder(engine: BatchEmbedderLike): Embedder {\n return async (texts: string[]) => {\n const results = await engine.embedBatch(texts);\n return results.map((result) => Float32Array.from(result.vector));\n };\n}\n","/**\n * File-backed JSON {@link MemoryStore} for simple Node persistence.\n *\n * Loads the whole corpus into memory and writes the full JSON file on every\n * mutation (debounced via an awaited write queue). This is intentionally\n * simple and aimed at small-to-medium corpora; see follow-ups for a\n * SQLite/OPFS backend at scale. Node-only — do not import in browser code.\n */\n\nimport { mkdir, readFile, writeFile } from \"node:fs/promises\";\nimport { dirname } from \"node:path\";\nimport { deserializeRecord, matchesFilter, serializeRecord } from \"../serialize.js\";\nimport type {\n MemoryExport,\n MemoryRecord,\n MemorySearchResult,\n MemoryStore,\n MetadataFilter,\n StoreSearchOptions,\n} from \"../types.js\";\nimport { type Scored, topK } from \"../vector.js\";\n\nconst DEFAULT_K = 5;\n\n/** JSON-on-disk vector store for Node. */\nexport class FileStore implements MemoryStore {\n private readonly path: string;\n private records = new Map<string, MemoryRecord>();\n private loaded = false;\n private writeQueue: Promise<void> = Promise.resolve();\n\n constructor(path: string) {\n this.path = path;\n }\n\n private async ensureLoaded(): Promise<void> {\n if (this.loaded) {\n return;\n }\n try {\n const raw = await readFile(this.path, \"utf8\");\n const parsed = JSON.parse(raw) as MemoryExport;\n for (const record of parsed.records) {\n const hydrated = deserializeRecord(record);\n this.records.set(hydrated.id, hydrated);\n }\n } catch (error) {\n // Missing file is fine — start empty. Re-throw anything else.\n if ((error as NodeJS.ErrnoException).code !== \"ENOENT\") {\n throw error;\n }\n }\n this.loaded = true;\n }\n\n private persist(): Promise<void> {\n const snapshot: MemoryExport = {\n version: 1,\n records: [...this.records.values()].map(serializeRecord),\n };\n const payload = JSON.stringify(snapshot);\n // Serialize writes so concurrent mutations don't interleave on disk.\n this.writeQueue = this.writeQueue.then(async () => {\n await mkdir(dirname(this.path), { recursive: true });\n await writeFile(this.path, payload, \"utf8\");\n });\n return this.writeQueue;\n }\n\n async add(record: MemoryRecord): Promise<void> {\n await this.ensureLoaded();\n this.records.set(record.id, record);\n await this.persist();\n }\n\n async addMany(records: MemoryRecord[]): Promise<void> {\n await this.ensureLoaded();\n for (const record of records) {\n this.records.set(record.id, record);\n }\n await this.persist();\n }\n\n async get(id: string): Promise<MemoryRecord | undefined> {\n await this.ensureLoaded();\n return this.records.get(id);\n }\n\n async search(\n queryVector: Float32Array,\n options: StoreSearchOptions = {},\n ): Promise<MemorySearchResult[]> {\n await this.ensureLoaded();\n const k = options.k ?? DEFAULT_K;\n const candidates: { item: MemoryRecord; vector: Float32Array }[] = [];\n for (const record of this.records.values()) {\n if (!record.embedding) {\n continue;\n }\n if (!matchesFilter(record.metadata, options.filter)) {\n continue;\n }\n candidates.push({ item: record, vector: record.embedding });\n }\n const ranked: Scored<MemoryRecord>[] = topK(queryVector, candidates, k, options.minScore);\n return ranked.map((entry) => ({ record: entry.item, score: entry.score }));\n }\n\n async delete(id: string): Promise<boolean> {\n await this.ensureLoaded();\n const existed = this.records.delete(id);\n if (existed) {\n await this.persist();\n }\n return existed;\n }\n\n async list(filter?: MetadataFilter): Promise<MemoryRecord[]> {\n await this.ensureLoaded();\n const out: MemoryRecord[] = [];\n for (const record of this.records.values()) {\n if (matchesFilter(record.metadata, filter)) {\n out.push(record);\n }\n }\n return out;\n }\n\n async clear(): Promise<void> {\n await this.ensureLoaded();\n this.records.clear();\n await this.persist();\n }\n\n async size(): Promise<number> {\n await this.ensureLoaded();\n return this.records.size;\n }\n}\n\n/** Create a {@link FileStore} backed by `path` (JSON). */\nexport function createFileStore(path: string): MemoryStore {\n return new FileStore(path);\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;AAgCA,SAAgB,qBAAqB,QAAqC;AACxE,QAAO,OAAO,UAAoB;AAEhC,UADgB,MAAM,OAAO,WAAW,MAAM,EAC/B,KAAK,WAAW,aAAa,KAAK,OAAO,OAAO,CAAC;;;;;;;;;;;;;;ACbpE,MAAM,YAAY;;AAGlB,IAAa,YAAb,MAA8C;CAC5C,AAAiB;CACjB,AAAQ,0BAAU,IAAI,KAA2B;CACjD,AAAQ,SAAS;CACjB,AAAQ,aAA4B,QAAQ,SAAS;CAErD,YAAY,QAAc;AACxB,OAAK,OAAOA;;CAGd,MAAc,eAA8B;AAC1C,MAAI,KAAK,OACP;AAEF,MAAI;GACF,MAAM,MAAM,MAAM,SAAS,KAAK,MAAM,OAAO;GAC7C,MAAM,SAAS,KAAK,MAAM,IAAI;AAC9B,QAAK,MAAM,UAAU,OAAO,SAAS;IACnC,MAAM,WAAW,kBAAkB,OAAO;AAC1C,SAAK,QAAQ,IAAI,SAAS,IAAI,SAAS;;WAElC,OAAO;AAEd,OAAK,MAAgC,SAAS,SAC5C,OAAM;;AAGV,OAAK,SAAS;;CAGhB,AAAQ,UAAyB;EAC/B,MAAMC,WAAyB;GAC7B,SAAS;GACT,SAAS,CAAC,GAAG,KAAK,QAAQ,QAAQ,CAAC,CAAC,IAAI,gBAAgB;GACzD;EACD,MAAM,UAAU,KAAK,UAAU,SAAS;AAExC,OAAK,aAAa,KAAK,WAAW,KAAK,YAAY;AACjD,SAAM,MAAM,QAAQ,KAAK,KAAK,EAAE,EAAE,WAAW,MAAM,CAAC;AACpD,SAAM,UAAU,KAAK,MAAM,SAAS,OAAO;IAC3C;AACF,SAAO,KAAK;;CAGd,MAAM,IAAI,QAAqC;AAC7C,QAAM,KAAK,cAAc;AACzB,OAAK,QAAQ,IAAI,OAAO,IAAI,OAAO;AACnC,QAAM,KAAK,SAAS;;CAGtB,MAAM,QAAQ,SAAwC;AACpD,QAAM,KAAK,cAAc;AACzB,OAAK,MAAM,UAAU,QACnB,MAAK,QAAQ,IAAI,OAAO,IAAI,OAAO;AAErC,QAAM,KAAK,SAAS;;CAGtB,MAAM,IAAI,IAA+C;AACvD,QAAM,KAAK,cAAc;AACzB,SAAO,KAAK,QAAQ,IAAI,GAAG;;CAG7B,MAAM,OACJ,aACA,UAA8B,EAAE,EACD;AAC/B,QAAM,KAAK,cAAc;EACzB,MAAM,IAAI,QAAQ,KAAK;EACvB,MAAMC,aAA6D,EAAE;AACrE,OAAK,MAAM,UAAU,KAAK,QAAQ,QAAQ,EAAE;AAC1C,OAAI,CAAC,OAAO,UACV;AAEF,OAAI,CAAC,cAAc,OAAO,UAAU,QAAQ,OAAO,CACjD;AAEF,cAAW,KAAK;IAAE,MAAM;IAAQ,QAAQ,OAAO;IAAW,CAAC;;AAG7D,SADuC,KAAK,aAAa,YAAY,GAAG,QAAQ,SAAS,CAC3E,KAAK,WAAW;GAAE,QAAQ,MAAM;GAAM,OAAO,MAAM;GAAO,EAAE;;CAG5E,MAAM,OAAO,IAA8B;AACzC,QAAM,KAAK,cAAc;EACzB,MAAM,UAAU,KAAK,QAAQ,OAAO,GAAG;AACvC,MAAI,QACF,OAAM,KAAK,SAAS;AAEtB,SAAO;;CAGT,MAAM,KAAK,QAAkD;AAC3D,QAAM,KAAK,cAAc;EACzB,MAAMC,MAAsB,EAAE;AAC9B,OAAK,MAAM,UAAU,KAAK,QAAQ,QAAQ,CACxC,KAAI,cAAc,OAAO,UAAU,OAAO,CACxC,KAAI,KAAK,OAAO;AAGpB,SAAO;;CAGT,MAAM,QAAuB;AAC3B,QAAM,KAAK,cAAc;AACzB,OAAK,QAAQ,OAAO;AACpB,QAAM,KAAK,SAAS;;CAGtB,MAAM,OAAwB;AAC5B,QAAM,KAAK,cAAc;AACzB,SAAO,KAAK,QAAQ;;;;AAKxB,SAAgB,gBAAgB,QAA2B;AACzD,QAAO,IAAI,UAAUH,OAAK"}
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { C as quantizeInt4, D as createDefaultHFKeyMapper, S as DEFAULT_GROUP_SIZE, T as DTYPE_BYTES, a as generateMoonshineEncoderGraph, b as nanoCodecWeightMap, i as generateMoonshineDecoderGraph, o as moonshineEncoderFrames, p as foldNanoCodecWeightNorm, s as parseMoonshineConfig, t as generateGraph, w as CANONICAL_KEYS } from "./architectures-
|
|
1
|
+
import { C as quantizeInt4, D as createDefaultHFKeyMapper, S as DEFAULT_GROUP_SIZE, T as DTYPE_BYTES, a as generateMoonshineEncoderGraph, b as nanoCodecWeightMap, i as generateMoonshineDecoderGraph, o as moonshineEncoderFrames, p as foldNanoCodecWeightNorm, s as parseMoonshineConfig, t as generateGraph, w as CANONICAL_KEYS } from "./architectures-BHkqQ9xp.mjs";
|
|
2
2
|
|
|
3
3
|
//#region src/gpu/device.ts
|
|
4
4
|
const COPY_SRC = 4;
|
|
@@ -10961,6 +10961,330 @@ function idbCacheAvailable() {
|
|
|
10961
10961
|
return idbAvailable();
|
|
10962
10962
|
}
|
|
10963
10963
|
|
|
10964
|
+
//#endregion
|
|
10965
|
+
//#region src/gpu/safetensors.ts
|
|
10966
|
+
/**
|
|
10967
|
+
* Parse safetensors header from an ArrayBuffer.
|
|
10968
|
+
*
|
|
10969
|
+
* Can be called with just the header bytes (for streaming — parse header first,
|
|
10970
|
+
* then fetch tensor data by offset) or with the entire file.
|
|
10971
|
+
*/
|
|
10972
|
+
function parseSafetensorsHeader(buffer) {
|
|
10973
|
+
if (buffer.byteLength < 8) throw new Error(`Buffer too small to contain safetensors header length (need 8 bytes, got ${buffer.byteLength}).`);
|
|
10974
|
+
const view = new DataView(buffer);
|
|
10975
|
+
const headerLength = Number(view.getBigUint64(0, true));
|
|
10976
|
+
if (headerLength > buffer.byteLength - 8) throw new Error(`Safetensors header length (${headerLength}) exceeds buffer size (${buffer.byteLength - 8}). Pass at least the first ${headerLength + 8} bytes.`);
|
|
10977
|
+
const headerBytes = new Uint8Array(buffer, 8, headerLength);
|
|
10978
|
+
const headerStr = new TextDecoder().decode(headerBytes);
|
|
10979
|
+
const header = JSON.parse(headerStr);
|
|
10980
|
+
const dataStart = 8 + headerLength;
|
|
10981
|
+
const entries = [];
|
|
10982
|
+
let metadata = null;
|
|
10983
|
+
for (const [name, info] of Object.entries(header)) {
|
|
10984
|
+
if (name === "__metadata__") {
|
|
10985
|
+
metadata = info;
|
|
10986
|
+
continue;
|
|
10987
|
+
}
|
|
10988
|
+
const { dtype, shape, data_offsets } = info;
|
|
10989
|
+
entries.push({
|
|
10990
|
+
name,
|
|
10991
|
+
dtype,
|
|
10992
|
+
shape,
|
|
10993
|
+
dataOffset: data_offsets[0],
|
|
10994
|
+
dataLength: data_offsets[1] - data_offsets[0]
|
|
10995
|
+
});
|
|
10996
|
+
}
|
|
10997
|
+
entries.sort((a, b) => a.dataOffset - b.dataOffset);
|
|
10998
|
+
return {
|
|
10999
|
+
headerLength,
|
|
11000
|
+
dataStart,
|
|
11001
|
+
entries,
|
|
11002
|
+
metadata
|
|
11003
|
+
};
|
|
11004
|
+
}
|
|
11005
|
+
/** Byte alignment required for each dtype. */
|
|
11006
|
+
function dtypeAlignment(dtype) {
|
|
11007
|
+
switch (dtype) {
|
|
11008
|
+
case "F64":
|
|
11009
|
+
case "I64":
|
|
11010
|
+
case "U64": return 8;
|
|
11011
|
+
case "F32":
|
|
11012
|
+
case "I32":
|
|
11013
|
+
case "U32": return 4;
|
|
11014
|
+
case "F16":
|
|
11015
|
+
case "BF16":
|
|
11016
|
+
case "I16":
|
|
11017
|
+
case "U16": return 2;
|
|
11018
|
+
case "I8":
|
|
11019
|
+
case "U8":
|
|
11020
|
+
case "BOOL": return 1;
|
|
11021
|
+
default: return 1;
|
|
11022
|
+
}
|
|
11023
|
+
}
|
|
11024
|
+
/**
|
|
11025
|
+
* Create a typed array for the given dtype.
|
|
11026
|
+
*
|
|
11027
|
+
* If `offset` is aligned to the element size, returns a zero-copy view
|
|
11028
|
+
* into `buffer`. Otherwise, copies the relevant slice into a new
|
|
11029
|
+
* properly-aligned ArrayBuffer.
|
|
11030
|
+
*/
|
|
11031
|
+
function makeTypedView(buffer, offset, byteLength, dtype) {
|
|
11032
|
+
const aligned = offset % dtypeAlignment(dtype) === 0;
|
|
11033
|
+
const src = aligned ? buffer : buffer.slice(offset, offset + byteLength);
|
|
11034
|
+
const base = aligned ? offset : 0;
|
|
11035
|
+
switch (dtype) {
|
|
11036
|
+
case "F32": return new Float32Array(src, base, byteLength / 4);
|
|
11037
|
+
case "F16": return new Uint16Array(src, base, byteLength / 2);
|
|
11038
|
+
case "BF16": return new Uint16Array(src, base, byteLength / 2);
|
|
11039
|
+
case "I32": return new Int32Array(src, base, byteLength / 4);
|
|
11040
|
+
case "U32": return new Uint32Array(src, base, byteLength / 4);
|
|
11041
|
+
case "I8": return new Int8Array(src, base, byteLength);
|
|
11042
|
+
case "U8": return new Uint8Array(src, base, byteLength);
|
|
11043
|
+
case "I16": return new Int16Array(src, base, byteLength / 2);
|
|
11044
|
+
case "U16": return new Uint16Array(src, base, byteLength / 2);
|
|
11045
|
+
case "F64": return new Float64Array(src, base, byteLength / 8);
|
|
11046
|
+
case "I64": return new BigInt64Array(src, base, byteLength / 8);
|
|
11047
|
+
case "U64": return new BigUint64Array(src, base, byteLength / 8);
|
|
11048
|
+
case "BOOL": return new Uint8Array(src, base, byteLength);
|
|
11049
|
+
default: throw new Error(`Unsupported safetensors dtype: ${dtype}`);
|
|
11050
|
+
}
|
|
11051
|
+
}
|
|
11052
|
+
/**
|
|
11053
|
+
* Get a typed array view for a tensor entry from a complete safetensors buffer.
|
|
11054
|
+
* Zero-copy when the byte offset is properly aligned; copies otherwise.
|
|
11055
|
+
*/
|
|
11056
|
+
function getTensorData(buffer, file, entry) {
|
|
11057
|
+
return makeTypedView(buffer, file.dataStart + entry.dataOffset, entry.dataLength, entry.dtype);
|
|
11058
|
+
}
|
|
11059
|
+
/**
|
|
11060
|
+
* Convert BF16 (bfloat16) data to F32.
|
|
11061
|
+
* BF16 is the upper 16 bits of an IEEE 754 float32, so conversion is a simple left-shift.
|
|
11062
|
+
*/
|
|
11063
|
+
function bf16ToF32(bf16Bytes) {
|
|
11064
|
+
const bf16 = new Uint16Array(bf16Bytes.buffer, bf16Bytes.byteOffset, bf16Bytes.byteLength / 2);
|
|
11065
|
+
const f32 = new Float32Array(bf16.length);
|
|
11066
|
+
const u32 = new Uint32Array(f32.buffer);
|
|
11067
|
+
for (let i = 0; i < bf16.length; i++) u32[i] = bf16[i] << 16;
|
|
11068
|
+
return f32;
|
|
11069
|
+
}
|
|
11070
|
+
/**
|
|
11071
|
+
* Convert F16 (IEEE 754 half-precision) data to F32.
|
|
11072
|
+
*/
|
|
11073
|
+
function f16ToF32(f16View) {
|
|
11074
|
+
const u16 = f16View instanceof Uint16Array ? f16View : new Uint16Array(f16View.buffer, f16View.byteOffset, f16View.byteLength / 2);
|
|
11075
|
+
const f32 = new Float32Array(u16.length);
|
|
11076
|
+
for (let i = 0; i < u16.length; i++) {
|
|
11077
|
+
const h = u16[i];
|
|
11078
|
+
const sign = h >> 15 & 1;
|
|
11079
|
+
const exp = h >> 10 & 31;
|
|
11080
|
+
const frac = h & 1023;
|
|
11081
|
+
if (exp === 0) f32[i] = (sign ? -1 : 1) * 2 ** -14 * (frac / 1024);
|
|
11082
|
+
else if (exp === 31) f32[i] = frac === 0 ? sign ? -Infinity : Infinity : NaN;
|
|
11083
|
+
else f32[i] = (sign ? -1 : 1) * 2 ** (exp - 15) * (1 + frac / 1024);
|
|
11084
|
+
}
|
|
11085
|
+
return f32;
|
|
11086
|
+
}
|
|
11087
|
+
|
|
11088
|
+
//#endregion
|
|
11089
|
+
//#region src/gpu/lora.ts
|
|
11090
|
+
function resolveAdapterBase(spec, revision) {
|
|
11091
|
+
if (spec.startsWith("file:")) return {
|
|
11092
|
+
base: spec.slice(5),
|
|
11093
|
+
local: true
|
|
11094
|
+
};
|
|
11095
|
+
if (spec.startsWith("http://") || spec.startsWith("https://")) return {
|
|
11096
|
+
base: spec,
|
|
11097
|
+
local: false
|
|
11098
|
+
};
|
|
11099
|
+
return {
|
|
11100
|
+
base: `https://huggingface.co/${spec.startsWith("hf:") ? spec.slice(3) : spec}/resolve/${revision}`,
|
|
11101
|
+
local: false
|
|
11102
|
+
};
|
|
11103
|
+
}
|
|
11104
|
+
async function readLocalFile(dir, name) {
|
|
11105
|
+
try {
|
|
11106
|
+
const fs = await import("node:fs");
|
|
11107
|
+
const p = (await import("node:path")).join(dir, name);
|
|
11108
|
+
if (!fs.existsSync(p)) return null;
|
|
11109
|
+
const buf = fs.readFileSync(p);
|
|
11110
|
+
return buf.buffer.slice(buf.byteOffset, buf.byteOffset + buf.byteLength);
|
|
11111
|
+
} catch {
|
|
11112
|
+
return null;
|
|
11113
|
+
}
|
|
11114
|
+
}
|
|
11115
|
+
async function fetchFile(base, local, name, hfToken) {
|
|
11116
|
+
if (local) return readLocalFile(base, name);
|
|
11117
|
+
const headers = {};
|
|
11118
|
+
if (hfToken) headers.Authorization = `Bearer ${hfToken}`;
|
|
11119
|
+
const res = await fetch(`${base}/${name}`, { headers });
|
|
11120
|
+
if (!res.ok) return null;
|
|
11121
|
+
return res.arrayBuffer();
|
|
11122
|
+
}
|
|
11123
|
+
/** Convert a safetensors tensor slice to f32 (adapters are tiny; eager is fine). */
|
|
11124
|
+
function sliceToF32(buf, dataStart, entry) {
|
|
11125
|
+
const start = dataStart + entry.dataOffset;
|
|
11126
|
+
switch (entry.dtype) {
|
|
11127
|
+
case "F32": return new Float32Array(buf.slice(start, start + entry.dataLength));
|
|
11128
|
+
case "BF16": return bf16ToF32(new Uint8Array(buf, start, entry.dataLength));
|
|
11129
|
+
case "F16": return f16ToF32(new Uint8Array(buf, start, entry.dataLength));
|
|
11130
|
+
default: throw new Error(`LoRA: unsupported adapter tensor dtype ${entry.dtype}`);
|
|
11131
|
+
}
|
|
11132
|
+
}
|
|
11133
|
+
/**
|
|
11134
|
+
* Download + parse an adapter (`adapter_config.json` + `adapter_model.safetensors`).
|
|
11135
|
+
* Returns null when the source has no adapter (so callers can no-op cleanly).
|
|
11136
|
+
*/
|
|
11137
|
+
async function fetchAdapter(spec, opts = {}) {
|
|
11138
|
+
const { base, local } = resolveAdapterBase(spec, opts.revision ?? "main");
|
|
11139
|
+
opts.onProgress?.(`Loading adapter ${spec}…`);
|
|
11140
|
+
const configBuf = await fetchFile(base, local, "adapter_config.json", opts.hfToken);
|
|
11141
|
+
if (!configBuf) throw new Error(`LoRA: no adapter_config.json at ${spec}. A flavor needs a PEFT adapter (adapter_config.json + adapter_model.safetensors).`);
|
|
11142
|
+
const config = JSON.parse(new TextDecoder().decode(configBuf));
|
|
11143
|
+
const weightsBuf = await fetchFile(base, local, "adapter_model.safetensors", opts.hfToken);
|
|
11144
|
+
if (!weightsBuf) throw new Error(`LoRA: no adapter_model.safetensors at ${spec} (only .safetensors adapters are supported, not legacy .bin).`);
|
|
11145
|
+
const file = parseSafetensorsHeader(weightsBuf);
|
|
11146
|
+
const tensors = /* @__PURE__ */ new Map();
|
|
11147
|
+
for (const entry of file.entries) {
|
|
11148
|
+
if (!entry.name.includes("lora_A") && !entry.name.includes("lora_B")) continue;
|
|
11149
|
+
tensors.set(entry.name, {
|
|
11150
|
+
data: sliceToF32(weightsBuf, file.dataStart, entry),
|
|
11151
|
+
shape: entry.shape
|
|
11152
|
+
});
|
|
11153
|
+
}
|
|
11154
|
+
if (tensors.size === 0) throw new Error(`LoRA: adapter at ${spec} has no lora_A/lora_B tensors.`);
|
|
11155
|
+
return {
|
|
11156
|
+
config,
|
|
11157
|
+
tensors,
|
|
11158
|
+
label: spec
|
|
11159
|
+
};
|
|
11160
|
+
}
|
|
11161
|
+
/**
|
|
11162
|
+
* Strip a PEFT tensor key down to its module path and canonicalize it to the
|
|
11163
|
+
* base weight key the engine uses.
|
|
11164
|
+
*
|
|
11165
|
+
* PEFT names a target like
|
|
11166
|
+
* base_model.model.model.layers.0.self_attn.q_proj.lora_A.weight
|
|
11167
|
+
* We drop the PEFT wrapper prefix + the lora_A/B suffix to get the module path
|
|
11168
|
+
* (…layers.0.self_attn.q_proj), append ".weight", and run the SAME key mapper
|
|
11169
|
+
* the base load used — so Qwen ("model." strip), Gemma, and LFM2 (out_proj→
|
|
11170
|
+
* o_proj) all canonicalize identically to their base tensors.
|
|
11171
|
+
*/
|
|
11172
|
+
function canonicalWeightKey(peftKey, keyMapper) {
|
|
11173
|
+
const marker = peftKey.includes(".lora_A") ? ".lora_A" : ".lora_B";
|
|
11174
|
+
let stem = peftKey.slice(0, peftKey.indexOf(marker));
|
|
11175
|
+
stem = stem.replace(/^base_model\.model\./, "").replace(/^base_model\./, "");
|
|
11176
|
+
return keyMapper(`${stem.startsWith("model.") ? stem : `model.${stem}`}.weight`);
|
|
11177
|
+
}
|
|
11178
|
+
/** Longest matching suffix in a {moduleName: value} pattern map. */
|
|
11179
|
+
function patternLookup(patterns, peftStem) {
|
|
11180
|
+
if (!patterns) return void 0;
|
|
11181
|
+
let best;
|
|
11182
|
+
let bestLen = -1;
|
|
11183
|
+
for (const [name, val] of Object.entries(patterns)) if (peftStem.includes(name) && name.length > bestLen) {
|
|
11184
|
+
best = val;
|
|
11185
|
+
bestLen = name.length;
|
|
11186
|
+
}
|
|
11187
|
+
return best;
|
|
11188
|
+
}
|
|
11189
|
+
/**
|
|
11190
|
+
* Pair lora_A/lora_B tensors and resolve each to a canonical `LoRADelta`.
|
|
11191
|
+
* `keyMapper` must be the same mapper the base model load used.
|
|
11192
|
+
*/
|
|
11193
|
+
function buildLoRADeltas(adapter, keyMapper) {
|
|
11194
|
+
const { config, tensors } = adapter;
|
|
11195
|
+
const stems = /* @__PURE__ */ new Set();
|
|
11196
|
+
for (const name of tensors.keys()) {
|
|
11197
|
+
const marker = name.includes(".lora_A") ? ".lora_A" : ".lora_B";
|
|
11198
|
+
stems.add(name.slice(0, name.indexOf(marker)));
|
|
11199
|
+
}
|
|
11200
|
+
const deltas = [];
|
|
11201
|
+
for (const stem of stems) {
|
|
11202
|
+
const A = tensors.get(`${stem}.lora_A.weight`) ?? tensors.get(`${stem}.lora_A`);
|
|
11203
|
+
const B = tensors.get(`${stem}.lora_B.weight`) ?? tensors.get(`${stem}.lora_B`);
|
|
11204
|
+
if (!A || !B) continue;
|
|
11205
|
+
const key = canonicalWeightKey(`${stem}.lora_A.weight`, keyMapper);
|
|
11206
|
+
if (!key) continue;
|
|
11207
|
+
const r = A.shape[0];
|
|
11208
|
+
const inFeatures = A.shape[1];
|
|
11209
|
+
const outFeatures = B.shape[0];
|
|
11210
|
+
if (B.shape[1] !== r) {
|
|
11211
|
+
console.warn(`[gerbil] LoRA: rank mismatch on ${stem} (A r=${r}, B r=${B.shape[1]}); skipping.`);
|
|
11212
|
+
continue;
|
|
11213
|
+
}
|
|
11214
|
+
const alpha = patternLookup(config.alpha_pattern, stem) ?? config.lora_alpha;
|
|
11215
|
+
const effR = patternLookup(config.rank_pattern, stem) ?? r;
|
|
11216
|
+
const scaling = config.use_rslora ? alpha / Math.sqrt(effR) : alpha / effR;
|
|
11217
|
+
deltas.push({
|
|
11218
|
+
key,
|
|
11219
|
+
A: A.data,
|
|
11220
|
+
B: B.data,
|
|
11221
|
+
r,
|
|
11222
|
+
inFeatures,
|
|
11223
|
+
outFeatures,
|
|
11224
|
+
scaling,
|
|
11225
|
+
fanInFanOut: config.fan_in_fan_out === true
|
|
11226
|
+
});
|
|
11227
|
+
}
|
|
11228
|
+
return deltas;
|
|
11229
|
+
}
|
|
11230
|
+
/**
|
|
11231
|
+
* Fold every resolved delta into the store's base weights in place
|
|
11232
|
+
* (W += scaling · B·A). Runs after weights are loaded and before architecture
|
|
11233
|
+
* transforms/quantization. Returns the count of tensors merged (for logging).
|
|
11234
|
+
*
|
|
11235
|
+
* Skips — loudly — any target whose base tensor is missing, shape-mismatched, or
|
|
11236
|
+
* not a float pack (a pre-quantized MLX/GPTQ base can't be merged this way).
|
|
11237
|
+
*/
|
|
11238
|
+
async function applyLoRAToStore(store, deltas, onProgress) {
|
|
11239
|
+
let merged = 0;
|
|
11240
|
+
let skipped = 0;
|
|
11241
|
+
for (const d of deltas) {
|
|
11242
|
+
const base = await store.get(d.key);
|
|
11243
|
+
if (!base) {
|
|
11244
|
+
skipped++;
|
|
11245
|
+
continue;
|
|
11246
|
+
}
|
|
11247
|
+
if (!(base.data instanceof Float32Array)) {
|
|
11248
|
+
console.warn(`[gerbil] LoRA: base tensor ${d.key} is not float (pre-quantized base?); skipping merge.`);
|
|
11249
|
+
skipped++;
|
|
11250
|
+
continue;
|
|
11251
|
+
}
|
|
11252
|
+
const [outF, inF] = base.shape;
|
|
11253
|
+
const rows = d.fanInFanOut ? inF : outF;
|
|
11254
|
+
const cols = d.fanInFanOut ? outF : inF;
|
|
11255
|
+
if (rows !== d.outFeatures || cols !== d.inFeatures) {
|
|
11256
|
+
console.warn(`[gerbil] LoRA: shape mismatch on ${d.key} (base ${base.shape}, delta ${d.outFeatures}×${d.inFeatures}); skipping.`);
|
|
11257
|
+
skipped++;
|
|
11258
|
+
continue;
|
|
11259
|
+
}
|
|
11260
|
+
const W = new Float32Array(base.data);
|
|
11261
|
+
const { A, B, r, scaling, fanInFanOut } = d;
|
|
11262
|
+
for (let o = 0; o < d.outFeatures; o++) {
|
|
11263
|
+
const bRow = o * r;
|
|
11264
|
+
for (let k = 0; k < r; k++) {
|
|
11265
|
+
const bScaled = scaling * B[bRow + k];
|
|
11266
|
+
if (bScaled === 0) continue;
|
|
11267
|
+
const aRow = k * d.inFeatures;
|
|
11268
|
+
if (fanInFanOut) for (let i = 0; i < d.inFeatures; i++) W[i * outF + o] += bScaled * A[aRow + i];
|
|
11269
|
+
else {
|
|
11270
|
+
const wRow = o * inF;
|
|
11271
|
+
for (let i = 0; i < d.inFeatures; i++) W[wRow + i] += bScaled * A[aRow + i];
|
|
11272
|
+
}
|
|
11273
|
+
}
|
|
11274
|
+
}
|
|
11275
|
+
await store.set(d.key, {
|
|
11276
|
+
data: W,
|
|
11277
|
+
shape: base.shape
|
|
11278
|
+
});
|
|
11279
|
+
merged++;
|
|
11280
|
+
}
|
|
11281
|
+
onProgress?.(`Adapter merged into ${merged} tensors${skipped ? ` (${skipped} skipped)` : ""}.`);
|
|
11282
|
+
return {
|
|
11283
|
+
merged,
|
|
11284
|
+
skipped
|
|
11285
|
+
};
|
|
11286
|
+
}
|
|
11287
|
+
|
|
10964
11288
|
//#endregion
|
|
10965
11289
|
//#region src/gpu/mlx-adapter.ts
|
|
10966
11290
|
/**
|
|
@@ -11342,130 +11666,6 @@ async function opfsEvictStale() {
|
|
|
11342
11666
|
} catch {}
|
|
11343
11667
|
}
|
|
11344
11668
|
|
|
11345
|
-
//#endregion
|
|
11346
|
-
//#region src/gpu/safetensors.ts
|
|
11347
|
-
/**
|
|
11348
|
-
* Parse safetensors header from an ArrayBuffer.
|
|
11349
|
-
*
|
|
11350
|
-
* Can be called with just the header bytes (for streaming — parse header first,
|
|
11351
|
-
* then fetch tensor data by offset) or with the entire file.
|
|
11352
|
-
*/
|
|
11353
|
-
function parseSafetensorsHeader(buffer) {
|
|
11354
|
-
if (buffer.byteLength < 8) throw new Error(`Buffer too small to contain safetensors header length (need 8 bytes, got ${buffer.byteLength}).`);
|
|
11355
|
-
const view = new DataView(buffer);
|
|
11356
|
-
const headerLength = Number(view.getBigUint64(0, true));
|
|
11357
|
-
if (headerLength > buffer.byteLength - 8) throw new Error(`Safetensors header length (${headerLength}) exceeds buffer size (${buffer.byteLength - 8}). Pass at least the first ${headerLength + 8} bytes.`);
|
|
11358
|
-
const headerBytes = new Uint8Array(buffer, 8, headerLength);
|
|
11359
|
-
const headerStr = new TextDecoder().decode(headerBytes);
|
|
11360
|
-
const header = JSON.parse(headerStr);
|
|
11361
|
-
const dataStart = 8 + headerLength;
|
|
11362
|
-
const entries = [];
|
|
11363
|
-
let metadata = null;
|
|
11364
|
-
for (const [name, info] of Object.entries(header)) {
|
|
11365
|
-
if (name === "__metadata__") {
|
|
11366
|
-
metadata = info;
|
|
11367
|
-
continue;
|
|
11368
|
-
}
|
|
11369
|
-
const { dtype, shape, data_offsets } = info;
|
|
11370
|
-
entries.push({
|
|
11371
|
-
name,
|
|
11372
|
-
dtype,
|
|
11373
|
-
shape,
|
|
11374
|
-
dataOffset: data_offsets[0],
|
|
11375
|
-
dataLength: data_offsets[1] - data_offsets[0]
|
|
11376
|
-
});
|
|
11377
|
-
}
|
|
11378
|
-
entries.sort((a, b) => a.dataOffset - b.dataOffset);
|
|
11379
|
-
return {
|
|
11380
|
-
headerLength,
|
|
11381
|
-
dataStart,
|
|
11382
|
-
entries,
|
|
11383
|
-
metadata
|
|
11384
|
-
};
|
|
11385
|
-
}
|
|
11386
|
-
/** Byte alignment required for each dtype. */
|
|
11387
|
-
function dtypeAlignment(dtype) {
|
|
11388
|
-
switch (dtype) {
|
|
11389
|
-
case "F64":
|
|
11390
|
-
case "I64":
|
|
11391
|
-
case "U64": return 8;
|
|
11392
|
-
case "F32":
|
|
11393
|
-
case "I32":
|
|
11394
|
-
case "U32": return 4;
|
|
11395
|
-
case "F16":
|
|
11396
|
-
case "BF16":
|
|
11397
|
-
case "I16":
|
|
11398
|
-
case "U16": return 2;
|
|
11399
|
-
case "I8":
|
|
11400
|
-
case "U8":
|
|
11401
|
-
case "BOOL": return 1;
|
|
11402
|
-
default: return 1;
|
|
11403
|
-
}
|
|
11404
|
-
}
|
|
11405
|
-
/**
|
|
11406
|
-
* Create a typed array for the given dtype.
|
|
11407
|
-
*
|
|
11408
|
-
* If `offset` is aligned to the element size, returns a zero-copy view
|
|
11409
|
-
* into `buffer`. Otherwise, copies the relevant slice into a new
|
|
11410
|
-
* properly-aligned ArrayBuffer.
|
|
11411
|
-
*/
|
|
11412
|
-
function makeTypedView(buffer, offset, byteLength, dtype) {
|
|
11413
|
-
const aligned = offset % dtypeAlignment(dtype) === 0;
|
|
11414
|
-
const src = aligned ? buffer : buffer.slice(offset, offset + byteLength);
|
|
11415
|
-
const base = aligned ? offset : 0;
|
|
11416
|
-
switch (dtype) {
|
|
11417
|
-
case "F32": return new Float32Array(src, base, byteLength / 4);
|
|
11418
|
-
case "F16": return new Uint16Array(src, base, byteLength / 2);
|
|
11419
|
-
case "BF16": return new Uint16Array(src, base, byteLength / 2);
|
|
11420
|
-
case "I32": return new Int32Array(src, base, byteLength / 4);
|
|
11421
|
-
case "U32": return new Uint32Array(src, base, byteLength / 4);
|
|
11422
|
-
case "I8": return new Int8Array(src, base, byteLength);
|
|
11423
|
-
case "U8": return new Uint8Array(src, base, byteLength);
|
|
11424
|
-
case "I16": return new Int16Array(src, base, byteLength / 2);
|
|
11425
|
-
case "U16": return new Uint16Array(src, base, byteLength / 2);
|
|
11426
|
-
case "F64": return new Float64Array(src, base, byteLength / 8);
|
|
11427
|
-
case "I64": return new BigInt64Array(src, base, byteLength / 8);
|
|
11428
|
-
case "U64": return new BigUint64Array(src, base, byteLength / 8);
|
|
11429
|
-
case "BOOL": return new Uint8Array(src, base, byteLength);
|
|
11430
|
-
default: throw new Error(`Unsupported safetensors dtype: ${dtype}`);
|
|
11431
|
-
}
|
|
11432
|
-
}
|
|
11433
|
-
/**
|
|
11434
|
-
* Get a typed array view for a tensor entry from a complete safetensors buffer.
|
|
11435
|
-
* Zero-copy when the byte offset is properly aligned; copies otherwise.
|
|
11436
|
-
*/
|
|
11437
|
-
function getTensorData(buffer, file, entry) {
|
|
11438
|
-
return makeTypedView(buffer, file.dataStart + entry.dataOffset, entry.dataLength, entry.dtype);
|
|
11439
|
-
}
|
|
11440
|
-
/**
|
|
11441
|
-
* Convert BF16 (bfloat16) data to F32.
|
|
11442
|
-
* BF16 is the upper 16 bits of an IEEE 754 float32, so conversion is a simple left-shift.
|
|
11443
|
-
*/
|
|
11444
|
-
function bf16ToF32(bf16Bytes) {
|
|
11445
|
-
const bf16 = new Uint16Array(bf16Bytes.buffer, bf16Bytes.byteOffset, bf16Bytes.byteLength / 2);
|
|
11446
|
-
const f32 = new Float32Array(bf16.length);
|
|
11447
|
-
const u32 = new Uint32Array(f32.buffer);
|
|
11448
|
-
for (let i = 0; i < bf16.length; i++) u32[i] = bf16[i] << 16;
|
|
11449
|
-
return f32;
|
|
11450
|
-
}
|
|
11451
|
-
/**
|
|
11452
|
-
* Convert F16 (IEEE 754 half-precision) data to F32.
|
|
11453
|
-
*/
|
|
11454
|
-
function f16ToF32(f16View) {
|
|
11455
|
-
const u16 = f16View instanceof Uint16Array ? f16View : new Uint16Array(f16View.buffer, f16View.byteOffset, f16View.byteLength / 2);
|
|
11456
|
-
const f32 = new Float32Array(u16.length);
|
|
11457
|
-
for (let i = 0; i < u16.length; i++) {
|
|
11458
|
-
const h = u16[i];
|
|
11459
|
-
const sign = h >> 15 & 1;
|
|
11460
|
-
const exp = h >> 10 & 31;
|
|
11461
|
-
const frac = h & 1023;
|
|
11462
|
-
if (exp === 0) f32[i] = (sign ? -1 : 1) * 2 ** -14 * (frac / 1024);
|
|
11463
|
-
else if (exp === 31) f32[i] = frac === 0 ? sign ? -Infinity : Infinity : NaN;
|
|
11464
|
-
else f32[i] = (sign ? -1 : 1) * 2 ** (exp - 15) * (1 + frac / 1024);
|
|
11465
|
-
}
|
|
11466
|
-
return f32;
|
|
11467
|
-
}
|
|
11468
|
-
|
|
11469
11669
|
//#endregion
|
|
11470
11670
|
//#region src/gpu/tokenizer.ts
|
|
11471
11671
|
/**
|
|
@@ -13054,6 +13254,20 @@ async function loadModel(options) {
|
|
|
13054
13254
|
filesLoaded++;
|
|
13055
13255
|
}
|
|
13056
13256
|
onProgress?.(95, 100, "Weights loaded.");
|
|
13257
|
+
if (options.adapter) {
|
|
13258
|
+
onProgress?.(95, 100, "Loading adapter…");
|
|
13259
|
+
const adapter = await fetchAdapter(options.adapter, {
|
|
13260
|
+
hfToken,
|
|
13261
|
+
revision,
|
|
13262
|
+
onProgress: (m) => onProgress?.(95, 100, m)
|
|
13263
|
+
});
|
|
13264
|
+
if (adapter) {
|
|
13265
|
+
const deltas = buildLoRADeltas(adapter, keyMapper);
|
|
13266
|
+
if (deltas.length === 0) console.warn(`[gerbil] LoRA: adapter ${adapter.label} resolved to 0 mergeable targets for ${architectureName}.`);
|
|
13267
|
+
const { merged, skipped } = await applyLoRAToStore(store, deltas, (m) => onProgress?.(96, 100, m));
|
|
13268
|
+
console.log(`[gerbil] adapter ${adapter.label}: merged ${merged} tensor(s)${skipped ? `, skipped ${skipped}` : ""}.`);
|
|
13269
|
+
}
|
|
13270
|
+
}
|
|
13057
13271
|
if (architectureName === "Qwen3_5ForConditionalGeneration") {
|
|
13058
13272
|
const textCfg = rawConfig.text_config ?? rawConfig;
|
|
13059
13273
|
const numLayers = textCfg.num_hidden_layers;
|
|
@@ -14282,5 +14496,5 @@ function selectGraphWeights(graph, weights) {
|
|
|
14282
14496
|
}
|
|
14283
14497
|
|
|
14284
14498
|
//#endregion
|
|
14285
|
-
export {
|
|
14286
|
-
//# sourceMappingURL=moonshine-stt-
|
|
14499
|
+
export { getOrCreatePipeline as C, destroyBuffers as S, verifyGPU as T, MATMUL_BIAS_F16C_SPEC as _, loadMoonshine as a, createStorageBuffer as b, quantizeBackboneInt4 as c, Tokenizer as d, applyLoRAToStore as f, KERNEL_REGISTRY as g, Executor as h, loadModel as i, quantizeKaniBackbone as l, fetchAdapter as m, MoonshineEncoderExecutor as n, loadOuteTTS as o, buildLoRADeltas as p, loadKaniTTS as r, loadParlerTTS as s, MoonshineSTT as t, remapPrunedToken as u, clearPipelineCache as v, initGPU as w, createUniformBuffer as x, createBindGroup as y };
|
|
14500
|
+
//# sourceMappingURL=moonshine-stt-9wc6t11v.mjs.map
|