@memberjunction/ai-vectors 5.21.0 → 5.23.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +254 -115
- package/dist/generic/IEmbedding.d.ts +21 -2
- package/dist/generic/IEmbedding.d.ts.map +1 -1
- package/dist/generic/IVectorDatabase.d.ts +37 -4
- package/dist/generic/IVectorDatabase.d.ts.map +1 -1
- package/dist/generic/IVectorIndex.d.ts +37 -8
- package/dist/generic/IVectorIndex.d.ts.map +1 -1
- package/dist/generic/SharedIndexMetadata.d.ts +52 -0
- package/dist/generic/SharedIndexMetadata.d.ts.map +1 -0
- package/dist/generic/SharedIndexMetadata.js +11 -0
- package/dist/generic/SharedIndexMetadata.js.map +1 -0
- package/dist/generic/TextChunker.d.ts +78 -0
- package/dist/generic/TextChunker.d.ts.map +1 -0
- package/dist/generic/TextChunker.js +182 -0
- package/dist/generic/TextChunker.js.map +1 -0
- package/dist/generic/TextExtractor.d.ts +48 -0
- package/dist/generic/TextExtractor.d.ts.map +1 -0
- package/dist/generic/TextExtractor.js +126 -0
- package/dist/generic/TextExtractor.js.map +1 -0
- package/dist/index.d.ts +3 -0
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +3 -0
- package/dist/index.js.map +1 -1
- package/dist/models/VectorBase.d.ts +4 -1
- package/dist/models/VectorBase.d.ts.map +1 -1
- package/dist/models/VectorBase.js +7 -3
- package/dist/models/VectorBase.js.map +1 -1
- package/package.json +8 -8
package/README.md
CHANGED
|
@@ -1,24 +1,44 @@
|
|
|
1
1
|
# @memberjunction/ai-vectors
|
|
2
2
|
|
|
3
|
-
|
|
3
|
+
Core foundation package for vector operations in MemberJunction. Provides text processing utilities (chunking, extraction), base classes for vectorization pipelines, and interfaces for embedding providers and vector databases.
|
|
4
|
+
|
|
5
|
+
## Installation
|
|
6
|
+
|
|
7
|
+
```bash
|
|
8
|
+
npm install @memberjunction/ai-vectors
|
|
9
|
+
```
|
|
10
|
+
|
|
11
|
+
## What's Included
|
|
12
|
+
|
|
13
|
+
| Export | Type | Purpose |
|
|
14
|
+
|---|---|---|
|
|
15
|
+
| `TextChunker` | Class | Token-aware text splitting with sentence, paragraph, and fixed strategies |
|
|
16
|
+
| `TextExtractor` | Class | HTML stripping, entity decoding, MIME-type routing, token truncation |
|
|
17
|
+
| `VectorBase` | Class | Base class providing RunView, Metadata, AIEngine integration for subclasses |
|
|
18
|
+
| `IEmbedding` | Interface | Contract for single and batch text embedding generation |
|
|
19
|
+
| `IVectorDatabase` | Interface | Contract for vector database management (create/delete/list indexes) |
|
|
20
|
+
| `IVectorIndex` | Interface | Contract for CRUD operations on vector records within an index |
|
|
21
|
+
| `ChunkTextParams` | Type | Configuration for `TextChunker.ChunkText()` |
|
|
22
|
+
| `TextChunk` | Type | Output chunk with text, offsets, token count, and index |
|
|
23
|
+
| `PageRecordsParams` | Type | Paginated entity record retrieval configuration |
|
|
4
24
|
|
|
5
25
|
## Architecture
|
|
6
26
|
|
|
7
27
|
```mermaid
|
|
8
28
|
graph TD
|
|
9
29
|
subgraph Core["@memberjunction/ai-vectors"]
|
|
30
|
+
TC["TextChunker"]
|
|
31
|
+
TE["TextExtractor"]
|
|
10
32
|
VB["VectorBase"]
|
|
11
33
|
IE["IEmbedding"]
|
|
12
34
|
IVD["IVectorDatabase"]
|
|
13
35
|
IVI["IVectorIndex"]
|
|
14
|
-
PT["PageRecordsParams"]
|
|
15
36
|
end
|
|
16
37
|
|
|
17
38
|
subgraph MJCore["MemberJunction Core"]
|
|
18
39
|
MD["Metadata"]
|
|
19
40
|
RV["RunView"]
|
|
20
41
|
BE["BaseEntity"]
|
|
21
|
-
UI["UserInfo"]
|
|
22
42
|
end
|
|
23
43
|
|
|
24
44
|
subgraph AIEngine["AI Engine"]
|
|
@@ -39,6 +59,8 @@ graph TD
|
|
|
39
59
|
AIM --> MOD
|
|
40
60
|
AIM --> VDB
|
|
41
61
|
SYNC --> VB
|
|
62
|
+
SYNC --> TC
|
|
63
|
+
SYNC --> TE
|
|
42
64
|
DUPE --> VB
|
|
43
65
|
|
|
44
66
|
style Core fill:#2d6a9f,stroke:#1a4971,color:#fff
|
|
@@ -47,167 +69,253 @@ graph TD
|
|
|
47
69
|
style Consumers fill:#7c5295,stroke:#563a6b,color:#fff
|
|
48
70
|
```
|
|
49
71
|
|
|
50
|
-
##
|
|
72
|
+
## TextChunker
|
|
51
73
|
|
|
52
|
-
|
|
53
|
-
|
|
74
|
+
Token-aware text splitting that respects natural language boundaries. All methods are static.
|
|
75
|
+
|
|
76
|
+
### Strategies
|
|
77
|
+
|
|
78
|
+
| Strategy | Splits On | Best For |
|
|
79
|
+
|---|---|---|
|
|
80
|
+
| `sentence` | Sentence-ending punctuation (`.` `!` `?`) | Prose, articles, descriptions |
|
|
81
|
+
| `paragraph` | Double newlines (`\n\n`) | Structured documents, Markdown, reports |
|
|
82
|
+
| `fixed` | Whitespace boundaries at the character limit | Logs, code, unstructured data |
|
|
83
|
+
|
|
84
|
+
### Basic Usage
|
|
85
|
+
|
|
86
|
+
```typescript
|
|
87
|
+
import { TextChunker, ChunkTextParams, TextChunk } from '@memberjunction/ai-vectors';
|
|
88
|
+
|
|
89
|
+
const article = `Machine learning models require training data.
|
|
90
|
+
The quality of training data directly impacts model performance.
|
|
91
|
+
Data preprocessing is a critical step in any ML pipeline.
|
|
92
|
+
|
|
93
|
+
Feature engineering transforms raw data into meaningful representations.
|
|
94
|
+
Good features can dramatically improve model accuracy.`;
|
|
95
|
+
|
|
96
|
+
// Sentence strategy (default)
|
|
97
|
+
const chunks: TextChunk[] = TextChunker.ChunkText({
|
|
98
|
+
Text: article,
|
|
99
|
+
MaxChunkTokens: 128,
|
|
100
|
+
Strategy: 'sentence'
|
|
101
|
+
});
|
|
102
|
+
|
|
103
|
+
for (const chunk of chunks) {
|
|
104
|
+
console.log(`Chunk ${chunk.Index}: ${chunk.TokenCount} tokens, offset ${chunk.StartOffset}-${chunk.EndOffset}`);
|
|
105
|
+
console.log(chunk.Text);
|
|
106
|
+
}
|
|
54
107
|
```
|
|
55
108
|
|
|
56
|
-
|
|
109
|
+
### Paragraph Strategy
|
|
57
110
|
|
|
58
|
-
|
|
111
|
+
```typescript
|
|
112
|
+
const markdownDoc = `## Introduction
|
|
59
113
|
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
- **Type definitions** -- `PageRecordsParams` for paginated data retrieval across entities
|
|
114
|
+
This document covers the architecture of our data pipeline.
|
|
115
|
+
It handles ingestion, transformation, and storage.
|
|
63
116
|
|
|
64
|
-
|
|
117
|
+
## Processing
|
|
65
118
|
|
|
66
|
-
|
|
119
|
+
Records are validated against schema constraints.
|
|
120
|
+
Invalid records are routed to a dead-letter queue.
|
|
67
121
|
|
|
68
|
-
|
|
122
|
+
## Storage
|
|
69
123
|
|
|
70
|
-
|
|
124
|
+
Processed data is stored in both relational and vector databases.
|
|
125
|
+
Vector embeddings enable semantic search across all records.`;
|
|
71
126
|
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
#PageRecordsByEntityID~T~(params) T[]
|
|
80
|
-
#GetAIModel(id?) AIModelEntityExtended
|
|
81
|
-
#GetVectorDatabase(id?) VectorDatabaseEntity
|
|
82
|
-
#RunViewForSingleValue~T~(entityName, filter) T
|
|
83
|
-
#SaveEntity(entity) boolean
|
|
84
|
-
#BuildExtraFilter(compositeKeys) string
|
|
85
|
-
}
|
|
127
|
+
const chunks = TextChunker.ChunkText({
|
|
128
|
+
Text: markdownDoc,
|
|
129
|
+
MaxChunkTokens: 256,
|
|
130
|
+
Strategy: 'paragraph'
|
|
131
|
+
});
|
|
132
|
+
// Each paragraph becomes a chunk (or paragraphs merge if they fit together)
|
|
133
|
+
```
|
|
86
134
|
|
|
87
|
-
|
|
88
|
-
+VectorizeEntity(params, user) VectorizeEntityResponse
|
|
89
|
-
+Config(forceRefresh, user) void
|
|
90
|
-
}
|
|
135
|
+
### Fixed Strategy
|
|
91
136
|
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
137
|
+
```typescript
|
|
138
|
+
const logData = `2024-01-15T10:00:00Z INFO Server started on port 4000
|
|
139
|
+
2024-01-15T10:00:01Z INFO Connected to database
|
|
140
|
+
2024-01-15T10:00:02Z WARN High memory usage detected: 85%
|
|
141
|
+
2024-01-15T10:00:03Z ERROR Connection timeout after 30000ms`;
|
|
142
|
+
|
|
143
|
+
const chunks = TextChunker.ChunkText({
|
|
144
|
+
Text: logData,
|
|
145
|
+
MaxChunkTokens: 64,
|
|
146
|
+
Strategy: 'fixed'
|
|
147
|
+
});
|
|
148
|
+
```
|
|
95
149
|
|
|
96
|
-
|
|
97
|
-
VectorBase <|-- DuplicateRecordDetector
|
|
150
|
+
### Configuring Overlap
|
|
98
151
|
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
152
|
+
Overlap repeats trailing content from the previous chunk at the start of the next chunk, preserving context across chunk boundaries. Defaults to 10% of `MaxChunkTokens`.
|
|
153
|
+
|
|
154
|
+
```typescript
|
|
155
|
+
// Explicit overlap: 50 tokens of shared context between chunks
|
|
156
|
+
const chunks = TextChunker.ChunkText({
|
|
157
|
+
Text: longDocument,
|
|
158
|
+
MaxChunkTokens: 512,
|
|
159
|
+
OverlapTokens: 50,
|
|
160
|
+
Strategy: 'sentence'
|
|
161
|
+
});
|
|
162
|
+
|
|
163
|
+
// No overlap
|
|
164
|
+
const chunks = TextChunker.ChunkText({
|
|
165
|
+
Text: longDocument,
|
|
166
|
+
MaxChunkTokens: 512,
|
|
167
|
+
OverlapTokens: 0,
|
|
168
|
+
Strategy: 'sentence'
|
|
169
|
+
});
|
|
102
170
|
```
|
|
103
171
|
|
|
104
|
-
|
|
172
|
+
### Token Estimation
|
|
105
173
|
|
|
106
|
-
|
|
107
|
-
|---|---|
|
|
108
|
-
| `GetRecordsByEntityID` | Load all entity records, optionally filtered by composite keys |
|
|
109
|
-
| `PageRecordsByEntityID` | Paginated retrieval with configurable page size, result type, and filter |
|
|
110
|
-
| `GetAIModel` | Locate an embedding model from `AIEngine.Instance.Models` by ID or get the first available |
|
|
111
|
-
| `GetVectorDatabase` | Locate a vector database from `AIEngine.Instance.VectorDatabases` by ID or get the first available |
|
|
112
|
-
| `RunViewForSingleValue` | Query for a single entity record matching a filter |
|
|
113
|
-
| `SaveEntity` | Save a `BaseEntity` with the current user context automatically applied |
|
|
114
|
-
| `BuildExtraFilter` | Convert an array of `CompositeKey` objects into a SQL filter string |
|
|
174
|
+
`EstimateTokenCount` provides a fast approximation using the ~4 characters per token heuristic for English text. This is suitable for chunking where exact counts are not critical.
|
|
115
175
|
|
|
116
|
-
|
|
176
|
+
```typescript
|
|
177
|
+
const tokens = TextChunker.EstimateTokenCount('This is a sample sentence.');
|
|
178
|
+
// Returns: 7 (26 characters / 4)
|
|
179
|
+
|
|
180
|
+
// For production accuracy with specific models, use tiktoken directly
|
|
181
|
+
// and pass the result to MaxChunkTokens for precise control
|
|
182
|
+
```
|
|
117
183
|
|
|
118
|
-
|
|
184
|
+
### TextChunk Output Shape
|
|
119
185
|
|
|
120
|
-
|
|
186
|
+
Each chunk includes full position metadata for traceability back to the source:
|
|
121
187
|
|
|
122
188
|
```typescript
|
|
123
|
-
interface
|
|
124
|
-
|
|
125
|
-
|
|
189
|
+
interface TextChunk {
|
|
190
|
+
Text: string; // The chunk text content
|
|
191
|
+
StartOffset: number; // Start character offset in original text
|
|
192
|
+
EndOffset: number; // End character offset (exclusive)
|
|
193
|
+
TokenCount: number; // Approximate token count
|
|
194
|
+
Index: number; // 0-based chunk index
|
|
126
195
|
}
|
|
127
196
|
```
|
|
128
197
|
|
|
129
|
-
|
|
198
|
+
## TextExtractor
|
|
199
|
+
|
|
200
|
+
Static utilities for extracting clean plain text from various content formats. Dependency-light (regex-based, no DOM parser required).
|
|
130
201
|
|
|
131
|
-
|
|
202
|
+
### HTML Extraction
|
|
132
203
|
|
|
133
204
|
```typescript
|
|
134
|
-
|
|
135
|
-
|
|
136
|
-
|
|
137
|
-
|
|
138
|
-
|
|
139
|
-
|
|
205
|
+
import { TextExtractor } from '@memberjunction/ai-vectors';
|
|
206
|
+
|
|
207
|
+
const html = `
|
|
208
|
+
<html>
|
|
209
|
+
<head><style>body { color: red; }</style></head>
|
|
210
|
+
<body>
|
|
211
|
+
<h1>Welcome</h1>
|
|
212
|
+
<p>This is a <strong>formatted</strong> paragraph with & entities.</p>
|
|
213
|
+
<script>alert('removed');</script>
|
|
214
|
+
<ul>
|
|
215
|
+
<li>Item one</li>
|
|
216
|
+
<li>Item two</li>
|
|
217
|
+
</ul>
|
|
218
|
+
</body>
|
|
219
|
+
</html>`;
|
|
220
|
+
|
|
221
|
+
const text = TextExtractor.ExtractFromHTML(html);
|
|
222
|
+
// "Welcome\nThis is a formatted paragraph with & entities.\nItem one\nItem two"
|
|
140
223
|
```
|
|
141
224
|
|
|
142
|
-
|
|
225
|
+
What it does:
|
|
226
|
+
- Removes `<script>` and `<style>` elements entirely
|
|
227
|
+
- Converts block-level elements (`<p>`, `<div>`, `<h1>`-`<h6>`, `<li>`, `<br>`, etc.) to newlines
|
|
228
|
+
- Strips all remaining HTML tags
|
|
229
|
+
- Decodes named entities (`&`, `<`, `>`, `"`, ` `, `—`, `…`, etc.)
|
|
230
|
+
- Decodes numeric entities (decimal `©` and hex `©`)
|
|
231
|
+
- Normalizes whitespace (collapses runs of spaces, limits consecutive newlines to 2)
|
|
143
232
|
|
|
144
|
-
|
|
233
|
+
### Plain Text Normalization
|
|
145
234
|
|
|
146
235
|
```typescript
|
|
147
|
-
|
|
148
|
-
|
|
149
|
-
|
|
150
|
-
getRecord(recordID: unknown, options?: unknown): unknown;
|
|
151
|
-
getRecords(recordIDs: unknown[], options?: unknown): unknown;
|
|
152
|
-
updateRecord(record: unknown, options?: unknown): unknown;
|
|
153
|
-
updateRecords(records: unknown[], options?: unknown): unknown;
|
|
154
|
-
deleteRecord(recordID: unknown, options?: unknown): unknown;
|
|
155
|
-
deleteRecords(recordIDs: unknown[], options?: unknown): unknown;
|
|
156
|
-
}
|
|
236
|
+
const raw = " Some text\x00with\x07control\x1Fcharacters\n\n\n\n\nand extra spaces ";
|
|
237
|
+
const clean = TextExtractor.ExtractFromPlainText(raw);
|
|
238
|
+
// "Some textwithcontrolcharacters\n\nand extra spaces"
|
|
157
239
|
```
|
|
158
240
|
|
|
159
|
-
|
|
241
|
+
Removes control characters (`\x00`-`\x1F` except `\n` and `\t`), normalizes whitespace, trims.
|
|
160
242
|
|
|
161
|
-
|
|
243
|
+
### MIME-Type Routing
|
|
162
244
|
|
|
163
245
|
```typescript
|
|
164
|
-
|
|
165
|
-
|
|
166
|
-
|
|
167
|
-
|
|
168
|
-
|
|
169
|
-
|
|
170
|
-
|
|
246
|
+
// Automatically selects the right extraction method
|
|
247
|
+
const fromHTML = TextExtractor.ExtractByMimeType(htmlContent, 'text/html');
|
|
248
|
+
const fromPlain = TextExtractor.ExtractByMimeType(plainContent, 'text/plain');
|
|
249
|
+
const fromCSV = TextExtractor.ExtractByMimeType(csvContent, 'text/csv'); // Falls back to plain text
|
|
250
|
+
|
|
251
|
+
// For binary formats (PDF, DOCX), extract text with a dedicated library first,
|
|
252
|
+
// then pass through ExtractFromPlainText for normalization:
|
|
253
|
+
// const pdfText = await pdfParse(buffer);
|
|
254
|
+
// const clean = TextExtractor.ExtractFromPlainText(pdfText);
|
|
171
255
|
```
|
|
172
256
|
|
|
173
|
-
|
|
257
|
+
### Token Truncation
|
|
174
258
|
|
|
175
|
-
|
|
259
|
+
```typescript
|
|
260
|
+
// Truncate text to fit within a model's context window
|
|
261
|
+
const truncated = TextExtractor.TruncateToTokenLimit(veryLongText, 8192);
|
|
262
|
+
// Truncates at the last whitespace boundary before the estimated character limit
|
|
263
|
+
```
|
|
264
|
+
|
|
265
|
+
## VectorBase
|
|
266
|
+
|
|
267
|
+
Abstract base class that downstream vector packages extend. Provides integrated access to MemberJunction's Metadata, RunView, and AIEngine systems.
|
|
268
|
+
|
|
269
|
+
### Class Diagram
|
|
270
|
+
|
|
271
|
+
```mermaid
|
|
272
|
+
classDiagram
|
|
273
|
+
class VectorBase {
|
|
274
|
+
+Metadata : Metadata
|
|
275
|
+
+RunView : RunView
|
|
276
|
+
+CurrentUser : UserInfo
|
|
277
|
+
#GetRecordsByEntityID(entityID, recordIDs?) BaseEntity[]
|
|
278
|
+
#PageRecordsByEntityID~T~(params) T[]
|
|
279
|
+
#GetAIModel(id?) MJAIModelEntityExtended
|
|
280
|
+
#GetVectorDatabase(id?) MJVectorDatabaseEntity
|
|
281
|
+
#RunViewForSingleValue~T~(entityName, filter) T | null
|
|
282
|
+
#SaveEntity(entity) boolean
|
|
283
|
+
#BuildExtraFilter(compositeKeys) string
|
|
284
|
+
}
|
|
285
|
+
```
|
|
176
286
|
|
|
177
|
-
|
|
287
|
+
### Extending VectorBase
|
|
178
288
|
|
|
179
289
|
```typescript
|
|
180
290
|
import { VectorBase, PageRecordsParams } from '@memberjunction/ai-vectors';
|
|
181
291
|
import { BaseEntity } from '@memberjunction/core';
|
|
182
292
|
|
|
183
293
|
export class MyVectorProcessor extends VectorBase {
|
|
184
|
-
async
|
|
185
|
-
//
|
|
294
|
+
async ProcessEntity(entityId: string): Promise<void> {
|
|
295
|
+
// Load all records for an entity
|
|
186
296
|
const records = await this.GetRecordsByEntityID(entityId);
|
|
187
297
|
|
|
188
298
|
// Access configured AI models and vector databases
|
|
189
|
-
const model = this.GetAIModel();
|
|
190
|
-
const vectorDb = this.GetVectorDatabase();
|
|
299
|
+
const model = this.GetAIModel(); // First available embedding model
|
|
300
|
+
const vectorDb = this.GetVectorDatabase(); // First available vector DB
|
|
191
301
|
|
|
192
302
|
for (const record of records) {
|
|
193
|
-
//
|
|
303
|
+
// Generate embeddings, upsert into vector DB
|
|
194
304
|
}
|
|
195
305
|
}
|
|
196
306
|
|
|
197
|
-
async
|
|
307
|
+
async ProcessInPages(entityId: string): Promise<void> {
|
|
198
308
|
let page = 1;
|
|
199
309
|
let hasMore = true;
|
|
200
310
|
|
|
201
311
|
while (hasMore) {
|
|
202
|
-
const
|
|
312
|
+
const records = await this.PageRecordsByEntityID<Record<string, unknown>>({
|
|
203
313
|
EntityID: entityId,
|
|
204
314
|
PageNumber: page,
|
|
205
315
|
PageSize: 100,
|
|
206
316
|
ResultType: 'simple',
|
|
207
317
|
Filter: "Status = 'Active'"
|
|
208
|
-
};
|
|
209
|
-
|
|
210
|
-
const records = await this.PageRecordsByEntityID<Record<string, unknown>>(params);
|
|
318
|
+
});
|
|
211
319
|
hasMore = records.length === 100;
|
|
212
320
|
page++;
|
|
213
321
|
}
|
|
@@ -222,7 +330,7 @@ import { VectorBase } from '@memberjunction/ai-vectors';
|
|
|
222
330
|
import { CompositeKey } from '@memberjunction/core';
|
|
223
331
|
|
|
224
332
|
class FilteredProcessor extends VectorBase {
|
|
225
|
-
async
|
|
333
|
+
async GetSpecificRecords(entityId: string): Promise<void> {
|
|
226
334
|
const keys: CompositeKey[] = [
|
|
227
335
|
{ KeyValuePairs: [{ FieldName: 'ID', Value: 'abc-123' }] },
|
|
228
336
|
{ KeyValuePairs: [{ FieldName: 'ID', Value: 'def-456' }] }
|
|
@@ -234,9 +342,45 @@ class FilteredProcessor extends VectorBase {
|
|
|
234
342
|
}
|
|
235
343
|
```
|
|
236
344
|
|
|
237
|
-
##
|
|
345
|
+
## API Reference
|
|
346
|
+
|
|
347
|
+
### TextChunker (Static Methods)
|
|
348
|
+
|
|
349
|
+
| Method | Parameters | Returns | Description |
|
|
350
|
+
|---|---|---|---|
|
|
351
|
+
| `ChunkText` | `params: ChunkTextParams` | `TextChunk[]` | Split text into token-bounded chunks using the specified strategy |
|
|
352
|
+
| `EstimateTokenCount` | `text: string` | `number` | Fast token count approximation (~4 chars/token) |
|
|
353
|
+
|
|
354
|
+
### TextExtractor (Static Methods)
|
|
238
355
|
|
|
239
|
-
|
|
356
|
+
| Method | Parameters | Returns | Description |
|
|
357
|
+
|---|---|---|---|
|
|
358
|
+
| `ExtractFromHTML` | `html: string` | `string` | Strip tags, decode entities, normalize whitespace |
|
|
359
|
+
| `ExtractFromPlainText` | `text: string` | `string` | Remove control characters, normalize whitespace |
|
|
360
|
+
| `ExtractByMimeType` | `content: string, mimeType: string` | `string` | Route to the appropriate extraction method by MIME type |
|
|
361
|
+
| `TruncateToTokenLimit` | `text: string, maxTokens: number` | `string` | Truncate at whitespace boundary within the token budget |
|
|
362
|
+
|
|
363
|
+
### VectorBase (Protected Methods for Subclasses)
|
|
364
|
+
|
|
365
|
+
| Method | Returns | Description |
|
|
366
|
+
|---|---|---|
|
|
367
|
+
| `GetRecordsByEntityID(entityID, recordIDs?)` | `Promise<BaseEntity[]>` | Load entity records, optionally filtered by composite keys |
|
|
368
|
+
| `PageRecordsByEntityID<T>(params)` | `Promise<T[]>` | Paginated retrieval with configurable page size and filter |
|
|
369
|
+
| `GetAIModel(id?)` | `MJAIModelEntityExtended` | Locate an embedding model by ID or get the first available |
|
|
370
|
+
| `GetVectorDatabase(id?)` | `MJVectorDatabaseEntity` | Locate a vector database by ID or get the first available |
|
|
371
|
+
| `RunViewForSingleValue<T>(entityName, filter)` | `Promise<T \| null>` | Query for a single entity record matching a filter |
|
|
372
|
+
| `SaveEntity(entity)` | `Promise<boolean>` | Save a BaseEntity with CurrentUser context applied |
|
|
373
|
+
| `BuildExtraFilter(compositeKeys)` | `string` | Convert CompositeKey array to a SQL filter string |
|
|
374
|
+
|
|
375
|
+
### Interfaces
|
|
376
|
+
|
|
377
|
+
| Interface | Methods | Purpose |
|
|
378
|
+
|---|---|---|
|
|
379
|
+
| `IEmbedding` | `createEmbedding`, `createBatchEmbedding` | Text embedding generation |
|
|
380
|
+
| `IVectorDatabase` | `listIndexes`, `createIndex`, `deleteIndex`, `editIndex` | Vector database management |
|
|
381
|
+
| `IVectorIndex` | `createRecord(s)`, `getRecord(s)`, `updateRecord(s)`, `deleteRecord(s)` | Vector record CRUD |
|
|
382
|
+
|
|
383
|
+
## Package Ecosystem
|
|
240
384
|
|
|
241
385
|
| Package | Depends On Core | Purpose |
|
|
242
386
|
|---|---|---|
|
|
@@ -244,19 +388,11 @@ This package sits at the base of the MemberJunction vector package hierarchy:
|
|
|
244
388
|
| `@memberjunction/ai-vector-sync` | Yes | Entity-to-vector synchronization |
|
|
245
389
|
| `@memberjunction/ai-vector-dupe` | Yes | Duplicate detection via vector similarity |
|
|
246
390
|
| `@memberjunction/ai-vectors-memory` | No | In-memory vector search and clustering |
|
|
247
|
-
| `@memberjunction/ai-vectors-pinecone` | No | Pinecone implementation of
|
|
391
|
+
| `@memberjunction/ai-vectors-pinecone` | No | Pinecone implementation of VectorDBBase |
|
|
248
392
|
|
|
249
|
-
##
|
|
393
|
+
## Further Reading
|
|
250
394
|
|
|
251
|
-
|
|
252
|
-
|---|---|
|
|
253
|
-
| `@memberjunction/core` | Metadata, RunView, BaseEntity, UserInfo |
|
|
254
|
-
| `@memberjunction/global` | MJGlobal class factory |
|
|
255
|
-
| `@memberjunction/core-entities` | VectorDatabaseEntity and other entity types |
|
|
256
|
-
| `@memberjunction/aiengine` | AIEngine singleton for model and database discovery |
|
|
257
|
-
| `@memberjunction/ai` | Base AI abstractions |
|
|
258
|
-
| `@memberjunction/ai-core-plus` | AIModelEntityExtended |
|
|
259
|
-
| `@memberjunction/ai-vectordb` | VectorDBBase and related types |
|
|
395
|
+
- [Text Processing Guide](docs/TEXT_PROCESSING_GUIDE.md) -- in-depth guide on chunking strategies, overlap tuning, HTML edge cases, and integration with vectorization/autotagging pipelines
|
|
260
396
|
|
|
261
397
|
## Development
|
|
262
398
|
|
|
@@ -264,8 +400,11 @@ This package sits at the base of the MemberJunction vector package hierarchy:
|
|
|
264
400
|
# Build
|
|
265
401
|
npm run build
|
|
266
402
|
|
|
267
|
-
#
|
|
268
|
-
npm run
|
|
403
|
+
# Run tests
|
|
404
|
+
npm run test
|
|
405
|
+
|
|
406
|
+
# Watch mode
|
|
407
|
+
npm run test:watch
|
|
269
408
|
```
|
|
270
409
|
|
|
271
410
|
## License
|
|
@@ -1,5 +1,24 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Interface for embedding providers.
|
|
3
|
+
* @deprecated Use BaseEmbeddings from @memberjunction/ai instead, which provides
|
|
4
|
+
* a more complete API with proper typing (EmbedTextResult, EmbedTextsResult).
|
|
5
|
+
*/
|
|
1
6
|
export interface IEmbedding {
|
|
2
|
-
createEmbedding(text: string, options?:
|
|
3
|
-
createBatchEmbedding(text: string[], options?:
|
|
7
|
+
createEmbedding(text: string, options?: EmbeddingOptions): Promise<EmbeddingResult>;
|
|
8
|
+
createBatchEmbedding(text: string[], options?: EmbeddingOptions): Promise<BatchEmbeddingResult>;
|
|
9
|
+
}
|
|
10
|
+
export interface EmbeddingOptions {
|
|
11
|
+
/** The model to use for embedding generation */
|
|
12
|
+
model?: string;
|
|
13
|
+
/** Number of dimensions for the output vectors */
|
|
14
|
+
dimensions?: number;
|
|
15
|
+
}
|
|
16
|
+
export interface EmbeddingResult {
|
|
17
|
+
vector: number[];
|
|
18
|
+
tokenCount?: number;
|
|
19
|
+
}
|
|
20
|
+
export interface BatchEmbeddingResult {
|
|
21
|
+
vectors: number[][];
|
|
22
|
+
tokenCounts?: number[];
|
|
4
23
|
}
|
|
5
24
|
//# sourceMappingURL=IEmbedding.d.ts.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"IEmbedding.d.ts","sourceRoot":"","sources":["../../src/generic/IEmbedding.ts"],"names":[],"mappings":"AAAA,MAAM,WAAW,UAAU;IACvB,eAAe,CAAC,IAAI,EAAE,MAAM,EAAE,OAAO,CAAC,EAAE,
|
|
1
|
+
{"version":3,"file":"IEmbedding.d.ts","sourceRoot":"","sources":["../../src/generic/IEmbedding.ts"],"names":[],"mappings":"AAAA;;;;GAIG;AACH,MAAM,WAAW,UAAU;IACvB,eAAe,CAAC,IAAI,EAAE,MAAM,EAAE,OAAO,CAAC,EAAE,gBAAgB,GAAG,OAAO,CAAC,eAAe,CAAC,CAAC;IACpF,oBAAoB,CAAC,IAAI,EAAE,MAAM,EAAE,EAAE,OAAO,CAAC,EAAE,gBAAgB,GAAG,OAAO,CAAC,oBAAoB,CAAC,CAAC;CACnG;AAED,MAAM,WAAW,gBAAgB;IAC7B,gDAAgD;IAChD,KAAK,CAAC,EAAE,MAAM,CAAC;IACf,kDAAkD;IAClD,UAAU,CAAC,EAAE,MAAM,CAAC;CACvB;AAED,MAAM,WAAW,eAAe;IAC5B,MAAM,EAAE,MAAM,EAAE,CAAC;IACjB,UAAU,CAAC,EAAE,MAAM,CAAC;CACvB;AAED,MAAM,WAAW,oBAAoB;IACjC,OAAO,EAAE,MAAM,EAAE,EAAE,CAAC;IACpB,WAAW,CAAC,EAAE,MAAM,EAAE,CAAC;CAC1B"}
|
|
@@ -1,7 +1,40 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Interface for vector database providers.
|
|
3
|
+
* @deprecated Use VectorDBBase from @memberjunction/ai-vectordb instead, which provides
|
|
4
|
+
* a complete abstract class with proper typing (IndexList, BaseResponse, CreateIndexParams).
|
|
5
|
+
*/
|
|
1
6
|
export interface IVectorDatabase {
|
|
2
|
-
listIndexes(options?:
|
|
3
|
-
createIndex(options:
|
|
4
|
-
deleteIndex(indexID:
|
|
5
|
-
editIndex(indexID:
|
|
7
|
+
listIndexes(options?: VectorDatabaseOptions): Promise<VectorIndexInfo[]>;
|
|
8
|
+
createIndex(options: CreateVectorIndexOptions): Promise<VectorDatabaseResult>;
|
|
9
|
+
deleteIndex(indexID: string, options?: VectorDatabaseOptions): Promise<VectorDatabaseResult>;
|
|
10
|
+
editIndex(indexID: string, options?: EditVectorIndexOptions): Promise<VectorDatabaseResult>;
|
|
11
|
+
}
|
|
12
|
+
export interface VectorDatabaseOptions {
|
|
13
|
+
/** Timeout in milliseconds for the operation */
|
|
14
|
+
timeoutMs?: number;
|
|
15
|
+
}
|
|
16
|
+
export interface CreateVectorIndexOptions extends VectorDatabaseOptions {
|
|
17
|
+
/** Name of the index to create */
|
|
18
|
+
name: string;
|
|
19
|
+
/** Number of dimensions for vectors in this index */
|
|
20
|
+
dimension: number;
|
|
21
|
+
/** Distance metric: cosine, euclidean, or dotproduct */
|
|
22
|
+
metric?: 'cosine' | 'euclidean' | 'dotproduct';
|
|
23
|
+
}
|
|
24
|
+
export interface EditVectorIndexOptions extends VectorDatabaseOptions {
|
|
25
|
+
/** New name for the index */
|
|
26
|
+
name?: string;
|
|
27
|
+
}
|
|
28
|
+
export interface VectorIndexInfo {
|
|
29
|
+
/** Name of the index */
|
|
30
|
+
name: string;
|
|
31
|
+
/** Number of dimensions for vectors in this index */
|
|
32
|
+
dimension: number;
|
|
33
|
+
/** Distance metric used */
|
|
34
|
+
metric: string;
|
|
35
|
+
}
|
|
36
|
+
export interface VectorDatabaseResult {
|
|
37
|
+
success: boolean;
|
|
38
|
+
message: string;
|
|
6
39
|
}
|
|
7
40
|
//# sourceMappingURL=IVectorDatabase.d.ts.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"IVectorDatabase.d.ts","sourceRoot":"","sources":["../../src/generic/IVectorDatabase.ts"],"names":[],"mappings":"AAAA,MAAM,WAAW,eAAe;IAC5B,WAAW,CAAC,OAAO,CAAC,EAAE,
|
|
1
|
+
{"version":3,"file":"IVectorDatabase.d.ts","sourceRoot":"","sources":["../../src/generic/IVectorDatabase.ts"],"names":[],"mappings":"AAAA;;;;GAIG;AACH,MAAM,WAAW,eAAe;IAC5B,WAAW,CAAC,OAAO,CAAC,EAAE,qBAAqB,GAAG,OAAO,CAAC,eAAe,EAAE,CAAC,CAAC;IACzE,WAAW,CAAC,OAAO,EAAE,wBAAwB,GAAG,OAAO,CAAC,oBAAoB,CAAC,CAAC;IAC9E,WAAW,CAAC,OAAO,EAAE,MAAM,EAAE,OAAO,CAAC,EAAE,qBAAqB,GAAG,OAAO,CAAC,oBAAoB,CAAC,CAAC;IAC7F,SAAS,CAAC,OAAO,EAAE,MAAM,EAAE,OAAO,CAAC,EAAE,sBAAsB,GAAG,OAAO,CAAC,oBAAoB,CAAC,CAAC;CAC/F;AAED,MAAM,WAAW,qBAAqB;IAClC,gDAAgD;IAChD,SAAS,CAAC,EAAE,MAAM,CAAC;CACtB;AAED,MAAM,WAAW,wBAAyB,SAAQ,qBAAqB;IACnE,kCAAkC;IAClC,IAAI,EAAE,MAAM,CAAC;IACb,qDAAqD;IACrD,SAAS,EAAE,MAAM,CAAC;IAClB,wDAAwD;IACxD,MAAM,CAAC,EAAE,QAAQ,GAAG,WAAW,GAAG,YAAY,CAAC;CAClD;AAED,MAAM,WAAW,sBAAuB,SAAQ,qBAAqB;IACjE,6BAA6B;IAC7B,IAAI,CAAC,EAAE,MAAM,CAAC;CACjB;AAED,MAAM,WAAW,eAAe;IAC5B,wBAAwB;IACxB,IAAI,EAAE,MAAM,CAAC;IACb,qDAAqD;IACrD,SAAS,EAAE,MAAM,CAAC;IAClB,2BAA2B;IAC3B,MAAM,EAAE,MAAM,CAAC;CAClB;AAED,MAAM,WAAW,oBAAoB;IACjC,OAAO,EAAE,OAAO,CAAC;IACjB,OAAO,EAAE,MAAM,CAAC;CACnB"}
|
|
@@ -1,11 +1,40 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Interface for vector index operations (CRUD on individual vectors).
|
|
3
|
+
* @deprecated Use VectorDBBase from @memberjunction/ai-vectordb instead, which provides
|
|
4
|
+
* a complete abstract class with proper typing (VectorRecord, BaseResponse, UpdateOptions).
|
|
5
|
+
*/
|
|
1
6
|
export interface IVectorIndex {
|
|
2
|
-
createRecord(record:
|
|
3
|
-
createRecords(records:
|
|
4
|
-
getRecord(recordID:
|
|
5
|
-
getRecords(recordIDs:
|
|
6
|
-
updateRecord(record:
|
|
7
|
-
updateRecords(records:
|
|
8
|
-
deleteRecord(recordID:
|
|
9
|
-
deleteRecords(recordIDs:
|
|
7
|
+
createRecord(record: VectorIndexRecord, options?: VectorIndexOptions): Promise<VectorIndexResult>;
|
|
8
|
+
createRecords(records: VectorIndexRecord[], options?: VectorIndexOptions): Promise<VectorIndexResult>;
|
|
9
|
+
getRecord(recordID: string, options?: VectorIndexOptions): Promise<VectorIndexRecordResult>;
|
|
10
|
+
getRecords(recordIDs: string[], options?: VectorIndexOptions): Promise<VectorIndexRecordsResult>;
|
|
11
|
+
updateRecord(record: VectorIndexRecord, options?: VectorIndexOptions): Promise<VectorIndexResult>;
|
|
12
|
+
updateRecords(records: VectorIndexRecord[], options?: VectorIndexOptions): Promise<VectorIndexResult>;
|
|
13
|
+
deleteRecord(recordID: string, options?: VectorIndexOptions): Promise<VectorIndexResult>;
|
|
14
|
+
deleteRecords(recordIDs: string[], options?: VectorIndexOptions): Promise<VectorIndexResult>;
|
|
15
|
+
}
|
|
16
|
+
export interface VectorIndexOptions {
|
|
17
|
+
/** Namespace or partition for the records */
|
|
18
|
+
namespace?: string;
|
|
19
|
+
/** Timeout in milliseconds */
|
|
20
|
+
timeoutMs?: number;
|
|
21
|
+
}
|
|
22
|
+
export interface VectorIndexRecord {
|
|
23
|
+
/** Unique identifier for the vector record */
|
|
24
|
+
id: string;
|
|
25
|
+
/** The vector embedding values */
|
|
26
|
+
values: number[];
|
|
27
|
+
/** Optional metadata associated with the record */
|
|
28
|
+
metadata?: Record<string, string | boolean | number | string[]>;
|
|
29
|
+
}
|
|
30
|
+
export interface VectorIndexResult {
|
|
31
|
+
success: boolean;
|
|
32
|
+
message: string;
|
|
33
|
+
}
|
|
34
|
+
export interface VectorIndexRecordResult extends VectorIndexResult {
|
|
35
|
+
record?: VectorIndexRecord;
|
|
36
|
+
}
|
|
37
|
+
export interface VectorIndexRecordsResult extends VectorIndexResult {
|
|
38
|
+
records?: VectorIndexRecord[];
|
|
10
39
|
}
|
|
11
40
|
//# sourceMappingURL=IVectorIndex.d.ts.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"IVectorIndex.d.ts","sourceRoot":"","sources":["../../src/generic/IVectorIndex.ts"],"names":[],"mappings":"AAAA,MAAM,WAAW,YAAY;IACzB,YAAY,CAAC,MAAM,EAAE,
|
|
1
|
+
{"version":3,"file":"IVectorIndex.d.ts","sourceRoot":"","sources":["../../src/generic/IVectorIndex.ts"],"names":[],"mappings":"AAAA;;;;GAIG;AACH,MAAM,WAAW,YAAY;IACzB,YAAY,CAAC,MAAM,EAAE,iBAAiB,EAAE,OAAO,CAAC,EAAE,kBAAkB,GAAG,OAAO,CAAC,iBAAiB,CAAC,CAAC;IAClG,aAAa,CAAC,OAAO,EAAE,iBAAiB,EAAE,EAAE,OAAO,CAAC,EAAE,kBAAkB,GAAG,OAAO,CAAC,iBAAiB,CAAC,CAAC;IACtG,SAAS,CAAC,QAAQ,EAAE,MAAM,EAAE,OAAO,CAAC,EAAE,kBAAkB,GAAG,OAAO,CAAC,uBAAuB,CAAC,CAAC;IAC5F,UAAU,CAAC,SAAS,EAAE,MAAM,EAAE,EAAE,OAAO,CAAC,EAAE,kBAAkB,GAAG,OAAO,CAAC,wBAAwB,CAAC,CAAC;IACjG,YAAY,CAAC,MAAM,EAAE,iBAAiB,EAAE,OAAO,CAAC,EAAE,kBAAkB,GAAG,OAAO,CAAC,iBAAiB,CAAC,CAAC;IAClG,aAAa,CAAC,OAAO,EAAE,iBAAiB,EAAE,EAAE,OAAO,CAAC,EAAE,kBAAkB,GAAG,OAAO,CAAC,iBAAiB,CAAC,CAAC;IACtG,YAAY,CAAC,QAAQ,EAAE,MAAM,EAAE,OAAO,CAAC,EAAE,kBAAkB,GAAG,OAAO,CAAC,iBAAiB,CAAC,CAAC;IACzF,aAAa,CAAC,SAAS,EAAE,MAAM,EAAE,EAAE,OAAO,CAAC,EAAE,kBAAkB,GAAG,OAAO,CAAC,iBAAiB,CAAC,CAAC;CAChG;AAED,MAAM,WAAW,kBAAkB;IAC/B,6CAA6C;IAC7C,SAAS,CAAC,EAAE,MAAM,CAAC;IACnB,8BAA8B;IAC9B,SAAS,CAAC,EAAE,MAAM,CAAC;CACtB;AAED,MAAM,WAAW,iBAAiB;IAC9B,8CAA8C;IAC9C,EAAE,EAAE,MAAM,CAAC;IACX,kCAAkC;IAClC,MAAM,EAAE,MAAM,EAAE,CAAC;IACjB,mDAAmD;IACnD,QAAQ,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,MAAM,GAAG,OAAO,GAAG,MAAM,GAAG,MAAM,EAAE,CAAC,CAAC;CACnE;AAED,MAAM,WAAW,iBAAiB;IAC9B,OAAO,EAAE,OAAO,CAAC;IACjB,OAAO,EAAE,MAAM,CAAC;CACnB;AAED,MAAM,WAAW,uBAAwB,SAAQ,iBAAiB;IAC9D,MAAM,CAAC,EAAE,iBAAiB,CAAC;CAC9B;AAED,MAAM,WAAW,wBAAyB,SAAQ,iBAAiB;IAC/D,OAAO,CAAC,EAAE,iBAAiB,EAAE,CAAC;CACjC"}
|