@memberjunction/ai-vectors 5.21.0 → 5.22.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +254 -115
- package/dist/generic/IEmbedding.d.ts +21 -2
- package/dist/generic/IEmbedding.d.ts.map +1 -1
- package/dist/generic/IVectorDatabase.d.ts +37 -4
- package/dist/generic/IVectorDatabase.d.ts.map +1 -1
- package/dist/generic/IVectorIndex.d.ts +37 -8
- package/dist/generic/IVectorIndex.d.ts.map +1 -1
- package/dist/generic/SharedIndexMetadata.d.ts +52 -0
- package/dist/generic/SharedIndexMetadata.d.ts.map +1 -0
- package/dist/generic/SharedIndexMetadata.js +11 -0
- package/dist/generic/SharedIndexMetadata.js.map +1 -0
- package/dist/generic/TextChunker.d.ts +78 -0
- package/dist/generic/TextChunker.d.ts.map +1 -0
- package/dist/generic/TextChunker.js +182 -0
- package/dist/generic/TextChunker.js.map +1 -0
- package/dist/generic/TextExtractor.d.ts +48 -0
- package/dist/generic/TextExtractor.d.ts.map +1 -0
- package/dist/generic/TextExtractor.js +126 -0
- package/dist/generic/TextExtractor.js.map +1 -0
- package/dist/index.d.ts +3 -0
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +3 -0
- package/dist/index.js.map +1 -1
- package/dist/models/VectorBase.d.ts +4 -1
- package/dist/models/VectorBase.d.ts.map +1 -1
- package/dist/models/VectorBase.js +7 -3
- package/dist/models/VectorBase.js.map +1 -1
- package/package.json +8 -8
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview Shared index metadata types for the unified Knowledge Hub vector index.
|
|
3
|
+
*
|
|
4
|
+
* All vectorized content (entity records, content items, file attachments, autotagged
|
|
5
|
+
* content) stored in a single shared index carries these metadata fields to enable
|
|
6
|
+
* filtered retrieval across heterogeneous sources.
|
|
7
|
+
*
|
|
8
|
+
* @module @memberjunction/ai-vectors
|
|
9
|
+
*/
|
|
10
|
+
/**
|
|
11
|
+
* Classification of how the vectorized content originated.
|
|
12
|
+
*/
|
|
13
|
+
export type VectorSourceType = 'entity' | 'content-item' | 'file' | 'web-page';
|
|
14
|
+
/**
|
|
15
|
+
* Metadata stored alongside every vector in the shared index.
|
|
16
|
+
* Used for post-hoc filtering and provenance tracking.
|
|
17
|
+
*/
|
|
18
|
+
export interface SharedVectorMetadata {
|
|
19
|
+
/** The MJ entity the vector came from (e.g., "Contacts", "Content Items") */
|
|
20
|
+
EntityName: string;
|
|
21
|
+
/** Origin classification */
|
|
22
|
+
SourceType: VectorSourceType;
|
|
23
|
+
/** MIME-like content category (e.g., "text/plain", "text/html", "application/pdf") */
|
|
24
|
+
ContentType: string;
|
|
25
|
+
/** Human-readable tags produced by the autotagging pipeline */
|
|
26
|
+
Tags: string[];
|
|
27
|
+
/** Composite key serialized string pointing back to the source record */
|
|
28
|
+
RecordID: string;
|
|
29
|
+
/** The Entity Document template used for vectorization */
|
|
30
|
+
EntityDocumentID: string;
|
|
31
|
+
/** Integer for multi-chunk documents */
|
|
32
|
+
ChunkIndex: number;
|
|
33
|
+
/** ISO date when this vector was indexed */
|
|
34
|
+
IndexedAt: string;
|
|
35
|
+
}
|
|
36
|
+
/**
|
|
37
|
+
* Filter options for querying the shared vector index.
|
|
38
|
+
* All fields are optional -- omitted fields mean "no filter on this dimension".
|
|
39
|
+
*/
|
|
40
|
+
export interface SharedIndexFilterOptions {
|
|
41
|
+
/** Restrict to specific entity names */
|
|
42
|
+
EntityNames?: string[];
|
|
43
|
+
/** Restrict to specific source types */
|
|
44
|
+
SourceTypes?: VectorSourceType[];
|
|
45
|
+
/** Restrict to specific MIME content types */
|
|
46
|
+
ContentTypes?: string[];
|
|
47
|
+
/** Require all specified tags to be present */
|
|
48
|
+
Tags?: string[];
|
|
49
|
+
/** Restrict to specific Entity Document IDs */
|
|
50
|
+
EntityDocumentIDs?: string[];
|
|
51
|
+
}
|
|
52
|
+
//# sourceMappingURL=SharedIndexMetadata.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"SharedIndexMetadata.d.ts","sourceRoot":"","sources":["../../src/generic/SharedIndexMetadata.ts"],"names":[],"mappings":"AAAA;;;;;;;;GAQG;AAEH;;GAEG;AACH,MAAM,MAAM,gBAAgB,GAAG,QAAQ,GAAG,cAAc,GAAG,MAAM,GAAG,UAAU,CAAC;AAE/E;;;GAGG;AACH,MAAM,WAAW,oBAAoB;IACjC,6EAA6E;IAC7E,UAAU,EAAE,MAAM,CAAC;IACnB,4BAA4B;IAC5B,UAAU,EAAE,gBAAgB,CAAC;IAC7B,sFAAsF;IACtF,WAAW,EAAE,MAAM,CAAC;IACpB,+DAA+D;IAC/D,IAAI,EAAE,MAAM,EAAE,CAAC;IACf,yEAAyE;IACzE,QAAQ,EAAE,MAAM,CAAC;IACjB,0DAA0D;IAC1D,gBAAgB,EAAE,MAAM,CAAC;IACzB,wCAAwC;IACxC,UAAU,EAAE,MAAM,CAAC;IACnB,4CAA4C;IAC5C,SAAS,EAAE,MAAM,CAAC;CACrB;AAED;;;GAGG;AACH,MAAM,WAAW,wBAAwB;IACrC,wCAAwC;IACxC,WAAW,CAAC,EAAE,MAAM,EAAE,CAAC;IACvB,wCAAwC;IACxC,WAAW,CAAC,EAAE,gBAAgB,EAAE,CAAC;IACjC,8CAA8C;IAC9C,YAAY,CAAC,EAAE,MAAM,EAAE,CAAC;IACxB,+CAA+C;IAC/C,IAAI,CAAC,EAAE,MAAM,EAAE,CAAC;IAChB,+CAA+C;IAC/C,iBAAiB,CAAC,EAAE,MAAM,EAAE,CAAC;CAChC"}
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview Shared index metadata types for the unified Knowledge Hub vector index.
|
|
3
|
+
*
|
|
4
|
+
* All vectorized content (entity records, content items, file attachments, autotagged
|
|
5
|
+
* content) stored in a single shared index carries these metadata fields to enable
|
|
6
|
+
* filtered retrieval across heterogeneous sources.
|
|
7
|
+
*
|
|
8
|
+
* @module @memberjunction/ai-vectors
|
|
9
|
+
*/
|
|
10
|
+
export {};
|
|
11
|
+
//# sourceMappingURL=SharedIndexMetadata.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"SharedIndexMetadata.js","sourceRoot":"","sources":["../../src/generic/SharedIndexMetadata.ts"],"names":[],"mappings":"AAAA;;;;;;;;GAQG"}
|
|
@@ -0,0 +1,78 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview Token-aware text chunking with sentence boundary detection.
|
|
3
|
+
*
|
|
4
|
+
* Provides configurable text splitting for use in vectorization, autotagging,
|
|
5
|
+
* and any pipeline that needs to break text into embedder-friendly chunks.
|
|
6
|
+
*
|
|
7
|
+
* @module @memberjunction/ai-vectors
|
|
8
|
+
*/
|
|
9
|
+
/**
|
|
10
|
+
* Parameters for chunking text.
|
|
11
|
+
*/
|
|
12
|
+
export interface ChunkTextParams {
|
|
13
|
+
/** The text to chunk */
|
|
14
|
+
Text: string;
|
|
15
|
+
/** Maximum tokens per chunk (default: 512) */
|
|
16
|
+
MaxChunkTokens?: number;
|
|
17
|
+
/** Overlap tokens between consecutive chunks (default: ~10% of MaxChunkTokens) */
|
|
18
|
+
OverlapTokens?: number;
|
|
19
|
+
/** Chunking strategy (default: 'sentence') */
|
|
20
|
+
Strategy?: 'sentence' | 'paragraph' | 'fixed';
|
|
21
|
+
}
|
|
22
|
+
/**
|
|
23
|
+
* A single chunk of text with position metadata.
|
|
24
|
+
*/
|
|
25
|
+
export interface TextChunk {
|
|
26
|
+
/** The chunk text content */
|
|
27
|
+
Text: string;
|
|
28
|
+
/** Start character offset in the original text */
|
|
29
|
+
StartOffset: number;
|
|
30
|
+
/** End character offset in the original text (exclusive) */
|
|
31
|
+
EndOffset: number;
|
|
32
|
+
/** Approximate token count for this chunk */
|
|
33
|
+
TokenCount: number;
|
|
34
|
+
/** 0-based chunk index */
|
|
35
|
+
Index: number;
|
|
36
|
+
}
|
|
37
|
+
/**
|
|
38
|
+
* Token-aware text chunker that respects natural boundaries.
|
|
39
|
+
*
|
|
40
|
+
* Supports three strategies:
|
|
41
|
+
* - **sentence**: Splits on sentence boundaries (`.`, `!`, `?`), never mid-sentence.
|
|
42
|
+
* Best for prose and natural language text.
|
|
43
|
+
* - **paragraph**: Splits on paragraph boundaries (`\n\n`). Best for structured documents.
|
|
44
|
+
* - **fixed**: Splits on whitespace boundaries at the token limit. Fastest but least semantic.
|
|
45
|
+
*/
|
|
46
|
+
export declare class TextChunker {
|
|
47
|
+
/**
|
|
48
|
+
* Split text into chunks that fit within the token limit.
|
|
49
|
+
*/
|
|
50
|
+
static ChunkText(params: ChunkTextParams): TextChunk[];
|
|
51
|
+
/**
|
|
52
|
+
* Estimate token count using whitespace splitting.
|
|
53
|
+
* This is a fast approximation; for production accuracy, use tiktoken.
|
|
54
|
+
*/
|
|
55
|
+
static EstimateTokenCount(text: string): number;
|
|
56
|
+
private static chunkBySentence;
|
|
57
|
+
private static chunkByParagraph;
|
|
58
|
+
private static chunkByFixed;
|
|
59
|
+
/**
|
|
60
|
+
* Split text into sentences using common sentence-ending punctuation.
|
|
61
|
+
* Handles abbreviations, decimals, and common edge cases.
|
|
62
|
+
*/
|
|
63
|
+
private static splitSentences;
|
|
64
|
+
/**
|
|
65
|
+
* Merge small text units (sentences or paragraphs) into chunks that fit within
|
|
66
|
+
* the token limit, with overlap between consecutive chunks.
|
|
67
|
+
*/
|
|
68
|
+
private static mergeUnitsIntoChunks;
|
|
69
|
+
/**
|
|
70
|
+
* Get the trailing units that fit within the overlap token budget.
|
|
71
|
+
*/
|
|
72
|
+
private static getOverlapUnits;
|
|
73
|
+
/**
|
|
74
|
+
* Build a TextChunk from a list of text units, finding their offset in the original text.
|
|
75
|
+
*/
|
|
76
|
+
private static buildChunkFromUnits;
|
|
77
|
+
}
|
|
78
|
+
//# sourceMappingURL=TextChunker.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"TextChunker.d.ts","sourceRoot":"","sources":["../../src/generic/TextChunker.ts"],"names":[],"mappings":"AAAA;;;;;;;GAOG;AAEH;;GAEG;AACH,MAAM,WAAW,eAAe;IAC5B,wBAAwB;IACxB,IAAI,EAAE,MAAM,CAAC;IACb,8CAA8C;IAC9C,cAAc,CAAC,EAAE,MAAM,CAAC;IACxB,kFAAkF;IAClF,aAAa,CAAC,EAAE,MAAM,CAAC;IACvB,8CAA8C;IAC9C,QAAQ,CAAC,EAAE,UAAU,GAAG,WAAW,GAAG,OAAO,CAAC;CACjD;AAED;;GAEG;AACH,MAAM,WAAW,SAAS;IACtB,6BAA6B;IAC7B,IAAI,EAAE,MAAM,CAAC;IACb,kDAAkD;IAClD,WAAW,EAAE,MAAM,CAAC;IACpB,4DAA4D;IAC5D,SAAS,EAAE,MAAM,CAAC;IAClB,6CAA6C;IAC7C,UAAU,EAAE,MAAM,CAAC;IACnB,0BAA0B;IAC1B,KAAK,EAAE,MAAM,CAAC;CACjB;AAED;;;;;;;;GAQG;AACH,qBAAa,WAAW;IACpB;;OAEG;WACW,SAAS,CAAC,MAAM,EAAE,eAAe,GAAG,SAAS,EAAE;IAsB7D;;;OAGG;WACW,kBAAkB,CAAC,IAAI,EAAE,MAAM,GAAG,MAAM;IAUtD,OAAO,CAAC,MAAM,CAAC,eAAe;IAK9B,OAAO,CAAC,MAAM,CAAC,gBAAgB;IAK/B,OAAO,CAAC,MAAM,CAAC,YAAY;IA0C3B;;;OAGG;IACH,OAAO,CAAC,MAAM,CAAC,cAAc;IAQ7B;;;OAGG;IACH,OAAO,CAAC,MAAM,CAAC,oBAAoB;IA6CnC;;OAEG;IACH,OAAO,CAAC,MAAM,CAAC,eAAe;IAgB9B;;OAEG;IACH,OAAO,CAAC,MAAM,CAAC,mBAAmB;CAcrC"}
|
|
@@ -0,0 +1,182 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview Token-aware text chunking with sentence boundary detection.
|
|
3
|
+
*
|
|
4
|
+
* Provides configurable text splitting for use in vectorization, autotagging,
|
|
5
|
+
* and any pipeline that needs to break text into embedder-friendly chunks.
|
|
6
|
+
*
|
|
7
|
+
* @module @memberjunction/ai-vectors
|
|
8
|
+
*/
|
|
9
|
+
/**
|
|
10
|
+
* Token-aware text chunker that respects natural boundaries.
|
|
11
|
+
*
|
|
12
|
+
* Supports three strategies:
|
|
13
|
+
* - **sentence**: Splits on sentence boundaries (`.`, `!`, `?`), never mid-sentence.
|
|
14
|
+
* Best for prose and natural language text.
|
|
15
|
+
* - **paragraph**: Splits on paragraph boundaries (`\n\n`). Best for structured documents.
|
|
16
|
+
* - **fixed**: Splits on whitespace boundaries at the token limit. Fastest but least semantic.
|
|
17
|
+
*/
|
|
18
|
+
export class TextChunker {
|
|
19
|
+
/**
|
|
20
|
+
* Split text into chunks that fit within the token limit.
|
|
21
|
+
*/
|
|
22
|
+
static ChunkText(params) {
|
|
23
|
+
const text = params.Text;
|
|
24
|
+
if (!text || text.trim().length === 0) {
|
|
25
|
+
return [];
|
|
26
|
+
}
|
|
27
|
+
const maxTokens = params.MaxChunkTokens ?? 512;
|
|
28
|
+
const overlapTokens = params.OverlapTokens ?? Math.floor(maxTokens * 0.1);
|
|
29
|
+
const strategy = params.Strategy ?? 'sentence';
|
|
30
|
+
switch (strategy) {
|
|
31
|
+
case 'sentence':
|
|
32
|
+
return TextChunker.chunkBySentence(text, maxTokens, overlapTokens);
|
|
33
|
+
case 'paragraph':
|
|
34
|
+
return TextChunker.chunkByParagraph(text, maxTokens, overlapTokens);
|
|
35
|
+
case 'fixed':
|
|
36
|
+
return TextChunker.chunkByFixed(text, maxTokens, overlapTokens);
|
|
37
|
+
default:
|
|
38
|
+
return TextChunker.chunkBySentence(text, maxTokens, overlapTokens);
|
|
39
|
+
}
|
|
40
|
+
}
|
|
41
|
+
/**
|
|
42
|
+
* Estimate token count using whitespace splitting.
|
|
43
|
+
* This is a fast approximation; for production accuracy, use tiktoken.
|
|
44
|
+
*/
|
|
45
|
+
static EstimateTokenCount(text) {
|
|
46
|
+
if (!text || text.trim().length === 0)
|
|
47
|
+
return 0;
|
|
48
|
+
// Rough approximation: ~4 characters per token for English text
|
|
49
|
+
return Math.ceil(text.length / 4);
|
|
50
|
+
}
|
|
51
|
+
// ─────────────────────────────────────────────
|
|
52
|
+
// Strategy Implementations
|
|
53
|
+
// ─────────────────────────────────────────────
|
|
54
|
+
static chunkBySentence(text, maxTokens, overlapTokens) {
|
|
55
|
+
const sentences = TextChunker.splitSentences(text);
|
|
56
|
+
return TextChunker.mergeUnitsIntoChunks(sentences, text, maxTokens, overlapTokens);
|
|
57
|
+
}
|
|
58
|
+
static chunkByParagraph(text, maxTokens, overlapTokens) {
|
|
59
|
+
const paragraphs = text.split(/\n\n+/).filter((p) => p.trim().length > 0);
|
|
60
|
+
return TextChunker.mergeUnitsIntoChunks(paragraphs, text, maxTokens, overlapTokens);
|
|
61
|
+
}
|
|
62
|
+
static chunkByFixed(text, maxTokens, overlapTokens) {
|
|
63
|
+
const words = text.split(/\s+/);
|
|
64
|
+
const maxChars = maxTokens * 4; // rough token-to-char estimate
|
|
65
|
+
const overlapChars = overlapTokens * 4;
|
|
66
|
+
const chunks = [];
|
|
67
|
+
let startCharOffset = 0;
|
|
68
|
+
let chunkIndex = 0;
|
|
69
|
+
while (startCharOffset < text.length) {
|
|
70
|
+
let endCharOffset = Math.min(startCharOffset + maxChars, text.length);
|
|
71
|
+
// Back up to last whitespace if not at end
|
|
72
|
+
if (endCharOffset < text.length) {
|
|
73
|
+
const lastSpace = text.lastIndexOf(' ', endCharOffset);
|
|
74
|
+
if (lastSpace > startCharOffset) {
|
|
75
|
+
endCharOffset = lastSpace;
|
|
76
|
+
}
|
|
77
|
+
}
|
|
78
|
+
const chunkText = text.slice(startCharOffset, endCharOffset).trim();
|
|
79
|
+
if (chunkText.length > 0) {
|
|
80
|
+
chunks.push({
|
|
81
|
+
Text: chunkText,
|
|
82
|
+
StartOffset: startCharOffset,
|
|
83
|
+
EndOffset: endCharOffset,
|
|
84
|
+
TokenCount: TextChunker.EstimateTokenCount(chunkText),
|
|
85
|
+
Index: chunkIndex++,
|
|
86
|
+
});
|
|
87
|
+
}
|
|
88
|
+
startCharOffset = endCharOffset - overlapChars;
|
|
89
|
+
if (startCharOffset >= text.length)
|
|
90
|
+
break;
|
|
91
|
+
if (endCharOffset >= text.length)
|
|
92
|
+
break;
|
|
93
|
+
}
|
|
94
|
+
return chunks;
|
|
95
|
+
}
|
|
96
|
+
// ─────────────────────────────────────────────
|
|
97
|
+
// Utility Methods
|
|
98
|
+
// ─────────────────────────────────────────────
|
|
99
|
+
/**
|
|
100
|
+
* Split text into sentences using common sentence-ending punctuation.
|
|
101
|
+
* Handles abbreviations, decimals, and common edge cases.
|
|
102
|
+
*/
|
|
103
|
+
static splitSentences(text) {
|
|
104
|
+
// Split on sentence-ending punctuation followed by space or end of string
|
|
105
|
+
const sentenceRegex = /[^.!?]*[.!?]+(?:\s|$)|[^.!?]+$/g;
|
|
106
|
+
const matches = text.match(sentenceRegex);
|
|
107
|
+
if (!matches)
|
|
108
|
+
return [text];
|
|
109
|
+
return matches.map((s) => s.trim()).filter((s) => s.length > 0);
|
|
110
|
+
}
|
|
111
|
+
/**
|
|
112
|
+
* Merge small text units (sentences or paragraphs) into chunks that fit within
|
|
113
|
+
* the token limit, with overlap between consecutive chunks.
|
|
114
|
+
*/
|
|
115
|
+
static mergeUnitsIntoChunks(units, originalText, maxTokens, overlapTokens) {
|
|
116
|
+
const chunks = [];
|
|
117
|
+
let currentUnits = [];
|
|
118
|
+
let currentTokens = 0;
|
|
119
|
+
let chunkIndex = 0;
|
|
120
|
+
for (const unit of units) {
|
|
121
|
+
const unitTokens = TextChunker.EstimateTokenCount(unit);
|
|
122
|
+
// If a single unit exceeds the max, emit it as its own chunk
|
|
123
|
+
if (unitTokens > maxTokens) {
|
|
124
|
+
// Flush current buffer first
|
|
125
|
+
if (currentUnits.length > 0) {
|
|
126
|
+
chunks.push(TextChunker.buildChunkFromUnits(currentUnits, originalText, chunkIndex++));
|
|
127
|
+
currentUnits = TextChunker.getOverlapUnits(currentUnits, overlapTokens);
|
|
128
|
+
currentTokens = currentUnits.reduce((sum, u) => sum + TextChunker.EstimateTokenCount(u), 0);
|
|
129
|
+
}
|
|
130
|
+
// Emit the oversized unit
|
|
131
|
+
chunks.push(TextChunker.buildChunkFromUnits([unit], originalText, chunkIndex++));
|
|
132
|
+
continue;
|
|
133
|
+
}
|
|
134
|
+
if (currentTokens + unitTokens > maxTokens && currentUnits.length > 0) {
|
|
135
|
+
chunks.push(TextChunker.buildChunkFromUnits(currentUnits, originalText, chunkIndex++));
|
|
136
|
+
currentUnits = TextChunker.getOverlapUnits(currentUnits, overlapTokens);
|
|
137
|
+
currentTokens = currentUnits.reduce((sum, u) => sum + TextChunker.EstimateTokenCount(u), 0);
|
|
138
|
+
}
|
|
139
|
+
currentUnits.push(unit);
|
|
140
|
+
currentTokens += unitTokens;
|
|
141
|
+
}
|
|
142
|
+
// Flush remaining
|
|
143
|
+
if (currentUnits.length > 0) {
|
|
144
|
+
chunks.push(TextChunker.buildChunkFromUnits(currentUnits, originalText, chunkIndex));
|
|
145
|
+
}
|
|
146
|
+
return chunks;
|
|
147
|
+
}
|
|
148
|
+
/**
|
|
149
|
+
* Get the trailing units that fit within the overlap token budget.
|
|
150
|
+
*/
|
|
151
|
+
static getOverlapUnits(units, overlapTokens) {
|
|
152
|
+
if (overlapTokens <= 0)
|
|
153
|
+
return [];
|
|
154
|
+
const overlapUnits = [];
|
|
155
|
+
let tokens = 0;
|
|
156
|
+
for (let i = units.length - 1; i >= 0; i--) {
|
|
157
|
+
const unitTokens = TextChunker.EstimateTokenCount(units[i]);
|
|
158
|
+
if (tokens + unitTokens > overlapTokens)
|
|
159
|
+
break;
|
|
160
|
+
overlapUnits.unshift(units[i]);
|
|
161
|
+
tokens += unitTokens;
|
|
162
|
+
}
|
|
163
|
+
return overlapUnits;
|
|
164
|
+
}
|
|
165
|
+
/**
|
|
166
|
+
* Build a TextChunk from a list of text units, finding their offset in the original text.
|
|
167
|
+
*/
|
|
168
|
+
static buildChunkFromUnits(units, originalText, index) {
|
|
169
|
+
const text = units.join(' ');
|
|
170
|
+
const startOffset = originalText.indexOf(units[0]);
|
|
171
|
+
const lastUnit = units[units.length - 1];
|
|
172
|
+
const endOffset = originalText.indexOf(lastUnit, startOffset) + lastUnit.length;
|
|
173
|
+
return {
|
|
174
|
+
Text: text,
|
|
175
|
+
StartOffset: Math.max(0, startOffset),
|
|
176
|
+
EndOffset: Math.min(endOffset, originalText.length),
|
|
177
|
+
TokenCount: TextChunker.EstimateTokenCount(text),
|
|
178
|
+
Index: index,
|
|
179
|
+
};
|
|
180
|
+
}
|
|
181
|
+
}
|
|
182
|
+
//# sourceMappingURL=TextChunker.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"TextChunker.js","sourceRoot":"","sources":["../../src/generic/TextChunker.ts"],"names":[],"mappings":"AAAA;;;;;;;GAOG;AAgCH;;;;;;;;GAQG;AACH,MAAM,OAAO,WAAW;IACpB;;OAEG;IACI,MAAM,CAAC,SAAS,CAAC,MAAuB;QAC3C,MAAM,IAAI,GAAG,MAAM,CAAC,IAAI,CAAC;QACzB,IAAI,CAAC,IAAI,IAAI,IAAI,CAAC,IAAI,EAAE,CAAC,MAAM,KAAK,CAAC,EAAE,CAAC;YACpC,OAAO,EAAE,CAAC;QACd,CAAC;QAED,MAAM,SAAS,GAAG,MAAM,CAAC,cAAc,IAAI,GAAG,CAAC;QAC/C,MAAM,aAAa,GAAG,MAAM,CAAC,aAAa,IAAI,IAAI,CAAC,KAAK,CAAC,SAAS,GAAG,GAAG,CAAC,CAAC;QAC1E,MAAM,QAAQ,GAAG,MAAM,CAAC,QAAQ,IAAI,UAAU,CAAC;QAE/C,QAAQ,QAAQ,EAAE,CAAC;YACf,KAAK,UAAU;gBACX,OAAO,WAAW,CAAC,eAAe,CAAC,IAAI,EAAE,SAAS,EAAE,aAAa,CAAC,CAAC;YACvE,KAAK,WAAW;gBACZ,OAAO,WAAW,CAAC,gBAAgB,CAAC,IAAI,EAAE,SAAS,EAAE,aAAa,CAAC,CAAC;YACxE,KAAK,OAAO;gBACR,OAAO,WAAW,CAAC,YAAY,CAAC,IAAI,EAAE,SAAS,EAAE,aAAa,CAAC,CAAC;YACpE;gBACI,OAAO,WAAW,CAAC,eAAe,CAAC,IAAI,EAAE,SAAS,EAAE,aAAa,CAAC,CAAC;QAC3E,CAAC;IACL,CAAC;IAED;;;OAGG;IACI,MAAM,CAAC,kBAAkB,CAAC,IAAY;QACzC,IAAI,CAAC,IAAI,IAAI,IAAI,CAAC,IAAI,EAAE,CAAC,MAAM,KAAK,CAAC;YAAE,OAAO,CAAC,CAAC;QAChD,gEAAgE;QAChE,OAAO,IAAI,CAAC,IAAI,CAAC,IAAI,CAAC,MAAM,GAAG,CAAC,CAAC,CAAC;IACtC,CAAC;IAED,gDAAgD;IAChD,2BAA2B;IAC3B,gDAAgD;IAExC,MAAM,CAAC,eAAe,CAAC,IAAY,EAAE,SAAiB,EAAE,aAAqB;QACjF,MAAM,SAAS,GAAG,WAAW,CAAC,cAAc,CAAC,IAAI,CAAC,CAAC;QACnD,OAAO,WAAW,CAAC,oBAAoB,CAAC,SAAS,EAAE,IAAI,EAAE,SAAS,EAAE,aAAa,CAAC,CAAC;IACvF,CAAC;IAEO,MAAM,CAAC,gBAAgB,CAAC,IAAY,EAAE,SAAiB,EAAE,aAAqB;QAClF,MAAM,UAAU,GAAG,IAAI,CAAC,KAAK,CAAC,OAAO,CAAC,CAAC,MAAM,CAAC,CAAC,CAAC,EAAE,EAAE,CAAC,CAAC,CAAC,IAAI,EAAE,CAAC,MAAM,GAAG,CAAC,CAAC,CAAC;QAC1E,OAAO,WAAW,CAAC,oBAAoB,CAAC,UAAU,EAAE,IAAI,EAAE,SAAS,EAAE,aAAa,CAAC,CAAC;IACxF,CAAC;IAEO,MAAM,CAAC,YAAY,CAAC,IAAY,EAAE,SAAiB,EAAE,aAAqB;QAC9E,MAAM,KAAK,GAAG,IAAI,CAAC,KAAK,CAAC,KAAK,CAAC,CAAC;QAChC,MAAM,QAAQ,GAAG,SAAS,GAAG,CAAC,CAAC,CAAC,+BAA+B;QAC/D,MAAM,YAAY,GAAG,aAAa,GAAG,CAAC,CAAC;QACvC,MAAM,MAAM,GAAgB,EAAE,CAAC;QAC/B,IAAI,eAAe,GAAG,CAAC,CAAC;QACxB,IAAI,UAAU,GAAG,CAAC,CAAC;QAEnB,OAAO,eAAe,GAAG,IAAI,CAAC,MAAM,EAAE,CAAC;YACnC,IAAI,aAAa,GAAG,IAAI,CAAC,GAAG,CAAC,eAAe,GAAG,QAAQ,EAAE,IAAI,CAAC,MAAM,CAAC,CAAC;YAEtE,2CAA2C;YAC3C,IAAI,aAAa,GAAG,IAAI,CAAC,MAAM,EAAE,CAAC;gBAC9B,MAAM,SAAS,GAAG,IAAI,CAAC,WAAW,CAAC,GAAG,EAAE,aAAa,CAAC,CAAC;gBACvD,IAAI,SAAS,GAAG,eAAe,EAAE,CAAC;oBAC9B,aAAa,GAAG,SAAS,CAAC;gBAC9B,CAAC;YACL,CAAC;YAED,MAAM,SAAS,GAAG,IAAI,CAAC,KAAK,CAAC,eAAe,EAAE,aAAa,CAAC,CAAC,IAAI,EAAE,CAAC;YACpE,IAAI,SAAS,CAAC,MAAM,GAAG,CAAC,EAAE,CAAC;gBACvB,MAAM,CAAC,IAAI,CAAC;oBACR,IAAI,EAAE,SAAS;oBACf,WAAW,EAAE,eAAe;oBAC5B,SAAS,EAAE,aAAa;oBACxB,UAAU,EAAE,WAAW,CAAC,kBAAkB,CAAC,SAAS,CAAC;oBACrD,KAAK,EAAE,UAAU,EAAE;iBACtB,CAAC,CAAC;YACP,CAAC;YAED,eAAe,GAAG,aAAa,GAAG,YAAY,CAAC;YAC/C,IAAI,eAAe,IAAI,IAAI,CAAC,MAAM;gBAAE,MAAM;YAC1C,IAAI,aAAa,IAAI,IAAI,CAAC,MAAM;gBAAE,MAAM;QAC5C,CAAC;QAED,OAAO,MAAM,CAAC;IAClB,CAAC;IAED,gDAAgD;IAChD,kBAAkB;IAClB,gDAAgD;IAEhD;;;OAGG;IACK,MAAM,CAAC,cAAc,CAAC,IAAY;QACtC,0EAA0E;QAC1E,MAAM,aAAa,GAAG,iCAAiC,CAAC;QACxD,MAAM,OAAO,GAAG,IAAI,CAAC,KAAK,CAAC,aAAa,CAAC,CAAC;QAC1C,IAAI,CAAC,OAAO;YAAE,OAAO,CAAC,IAAI,CAAC,CAAC;QAC5B,OAAO,OAAO,CAAC,GAAG,CAAC,CAAC,CAAC,EAAE,EAAE,CAAC,CAAC,CAAC,IAAI,EAAE,CAAC,CAAC,MAAM,CAAC,CAAC,CAAC,EAAE,EAAE,CAAC,CAAC,CAAC,MAAM,GAAG,CAAC,CAAC,CAAC;IACpE,CAAC;IAED;;;OAGG;IACK,MAAM,CAAC,oBAAoB,CAC/B,KAAe,EACf,YAAoB,EACpB,SAAiB,EACjB,aAAqB;QAErB,MAAM,MAAM,GAAgB,EAAE,CAAC;QAC/B,IAAI,YAAY,GAAa,EAAE,CAAC;QAChC,IAAI,aAAa,GAAG,CAAC,CAAC;QACtB,IAAI,UAAU,GAAG,CAAC,CAAC;QAEnB,KAAK,MAAM,IAAI,IAAI,KAAK,EAAE,CAAC;YACvB,MAAM,UAAU,GAAG,WAAW,CAAC,kBAAkB,CAAC,IAAI,CAAC,CAAC;YAExD,6DAA6D;YAC7D,IAAI,UAAU,GAAG,SAAS,EAAE,CAAC;gBACzB,6BAA6B;gBAC7B,IAAI,YAAY,CAAC,MAAM,GAAG,CAAC,EAAE,CAAC;oBAC1B,MAAM,CAAC,IAAI,CAAC,WAAW,CAAC,mBAAmB,CAAC,YAAY,EAAE,YAAY,EAAE,UAAU,EAAE,CAAC,CAAC,CAAC;oBACvF,YAAY,GAAG,WAAW,CAAC,eAAe,CAAC,YAAY,EAAE,aAAa,CAAC,CAAC;oBACxE,aAAa,GAAG,YAAY,CAAC,MAAM,CAAC,CAAC,GAAG,EAAE,CAAC,EAAE,EAAE,CAAC,GAAG,GAAG,WAAW,CAAC,kBAAkB,CAAC,CAAC,CAAC,EAAE,CAAC,CAAC,CAAC;gBAChG,CAAC;gBACD,0BAA0B;gBAC1B,MAAM,CAAC,IAAI,CAAC,WAAW,CAAC,mBAAmB,CAAC,CAAC,IAAI,CAAC,EAAE,YAAY,EAAE,UAAU,EAAE,CAAC,CAAC,CAAC;gBACjF,SAAS;YACb,CAAC;YAED,IAAI,aAAa,GAAG,UAAU,GAAG,SAAS,IAAI,YAAY,CAAC,MAAM,GAAG,CAAC,EAAE,CAAC;gBACpE,MAAM,CAAC,IAAI,CAAC,WAAW,CAAC,mBAAmB,CAAC,YAAY,EAAE,YAAY,EAAE,UAAU,EAAE,CAAC,CAAC,CAAC;gBACvF,YAAY,GAAG,WAAW,CAAC,eAAe,CAAC,YAAY,EAAE,aAAa,CAAC,CAAC;gBACxE,aAAa,GAAG,YAAY,CAAC,MAAM,CAAC,CAAC,GAAG,EAAE,CAAC,EAAE,EAAE,CAAC,GAAG,GAAG,WAAW,CAAC,kBAAkB,CAAC,CAAC,CAAC,EAAE,CAAC,CAAC,CAAC;YAChG,CAAC;YAED,YAAY,CAAC,IAAI,CAAC,IAAI,CAAC,CAAC;YACxB,aAAa,IAAI,UAAU,CAAC;QAChC,CAAC;QAED,kBAAkB;QAClB,IAAI,YAAY,CAAC,MAAM,GAAG,CAAC,EAAE,CAAC;YAC1B,MAAM,CAAC,IAAI,CAAC,WAAW,CAAC,mBAAmB,CAAC,YAAY,EAAE,YAAY,EAAE,UAAU,CAAC,CAAC,CAAC;QACzF,CAAC;QAED,OAAO,MAAM,CAAC;IAClB,CAAC;IAED;;OAEG;IACK,MAAM,CAAC,eAAe,CAAC,KAAe,EAAE,aAAqB;QACjE,IAAI,aAAa,IAAI,CAAC;YAAE,OAAO,EAAE,CAAC;QAElC,MAAM,YAAY,GAAa,EAAE,CAAC;QAClC,IAAI,MAAM,GAAG,CAAC,CAAC;QAEf,KAAK,IAAI,CAAC,GAAG,KAAK,CAAC,MAAM,GAAG,CAAC,EAAE,CAAC,IAAI,CAAC,EAAE,CAAC,EAAE,EAAE,CAAC;YACzC,MAAM,UAAU,GAAG,WAAW,CAAC,kBAAkB,CAAC,KAAK,CAAC,CAAC,CAAC,CAAC,CAAC;YAC5D,IAAI,MAAM,GAAG,UAAU,GAAG,aAAa;gBAAE,MAAM;YAC/C,YAAY,CAAC,OAAO,CAAC,KAAK,CAAC,CAAC,CAAC,CAAC,CAAC;YAC/B,MAAM,IAAI,UAAU,CAAC;QACzB,CAAC;QAED,OAAO,YAAY,CAAC;IACxB,CAAC;IAED;;OAEG;IACK,MAAM,CAAC,mBAAmB,CAAC,KAAe,EAAE,YAAoB,EAAE,KAAa;QACnF,MAAM,IAAI,GAAG,KAAK,CAAC,IAAI,CAAC,GAAG,CAAC,CAAC;QAC7B,MAAM,WAAW,GAAG,YAAY,CAAC,OAAO,CAAC,KAAK,CAAC,CAAC,CAAC,CAAC,CAAC;QACnD,MAAM,QAAQ,GAAG,KAAK,CAAC,KAAK,CAAC,MAAM,GAAG,CAAC,CAAC,CAAC;QACzC,MAAM,SAAS,GAAG,YAAY,CAAC,OAAO,CAAC,QAAQ,EAAE,WAAW,CAAC,GAAG,QAAQ,CAAC,MAAM,CAAC;QAEhF,OAAO;YACH,IAAI,EAAE,IAAI;YACV,WAAW,EAAE,IAAI,CAAC,GAAG,CAAC,CAAC,EAAE,WAAW,CAAC;YACrC,SAAS,EAAE,IAAI,CAAC,GAAG,CAAC,SAAS,EAAE,YAAY,CAAC,MAAM,CAAC;YACnD,UAAU,EAAE,WAAW,CAAC,kBAAkB,CAAC,IAAI,CAAC;YAChD,KAAK,EAAE,KAAK;SACf,CAAC;IACN,CAAC;CACJ"}
|
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview Shared text extraction utilities for various content formats.
|
|
3
|
+
*
|
|
4
|
+
* Provides a unified interface for extracting plain text from HTML, plain text,
|
|
5
|
+
* and other content types. Used by both the vectorization and autotagging pipelines.
|
|
6
|
+
*
|
|
7
|
+
* @module @memberjunction/ai-vectors
|
|
8
|
+
*/
|
|
9
|
+
/**
|
|
10
|
+
* Shared text extraction utilities.
|
|
11
|
+
*
|
|
12
|
+
* Provides static methods for extracting clean text from various content formats.
|
|
13
|
+
* These utilities are intentionally dependency-light — they use regex-based extraction
|
|
14
|
+
* rather than heavy DOM parsing libraries, making them suitable for server-side use
|
|
15
|
+
* without browser dependencies.
|
|
16
|
+
*/
|
|
17
|
+
export declare class TextExtractor {
|
|
18
|
+
/**
|
|
19
|
+
* Extract readable text from HTML content.
|
|
20
|
+
* Strips tags, decodes entities, normalizes whitespace.
|
|
21
|
+
*/
|
|
22
|
+
static ExtractFromHTML(html: string): string;
|
|
23
|
+
/**
|
|
24
|
+
* Extract and normalize plain text.
|
|
25
|
+
* Trims, normalizes whitespace, and removes control characters.
|
|
26
|
+
*/
|
|
27
|
+
static ExtractFromPlainText(text: string): string;
|
|
28
|
+
/**
|
|
29
|
+
* Detect content type from a MIME type string and extract text accordingly.
|
|
30
|
+
*
|
|
31
|
+
* Currently supports:
|
|
32
|
+
* - text/html → ExtractFromHTML
|
|
33
|
+
* - text/plain → ExtractFromPlainText
|
|
34
|
+
* - text/* → ExtractFromPlainText (fallback)
|
|
35
|
+
*
|
|
36
|
+
* For binary formats (PDF, DOCX, etc.), callers should use dedicated libraries
|
|
37
|
+
* (pdf-parse, officeparser) and then pass the extracted text through ExtractFromPlainText.
|
|
38
|
+
*/
|
|
39
|
+
static ExtractByMimeType(content: string, mimeType: string): string;
|
|
40
|
+
/**
|
|
41
|
+
* Truncate text to a maximum token count (estimated).
|
|
42
|
+
* Truncates at the last whitespace boundary before the limit.
|
|
43
|
+
*/
|
|
44
|
+
static TruncateToTokenLimit(text: string, maxTokens: number): string;
|
|
45
|
+
private static decodeHTMLEntities;
|
|
46
|
+
private static normalizeWhitespace;
|
|
47
|
+
}
|
|
48
|
+
//# sourceMappingURL=TextExtractor.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"TextExtractor.d.ts","sourceRoot":"","sources":["../../src/generic/TextExtractor.ts"],"names":[],"mappings":"AAAA;;;;;;;GAOG;AAEH;;;;;;;GAOG;AACH,qBAAa,aAAa;IACtB;;;OAGG;WACW,eAAe,CAAC,IAAI,EAAE,MAAM,GAAG,MAAM;IA0BnD;;;OAGG;WACW,oBAAoB,CAAC,IAAI,EAAE,MAAM,GAAG,MAAM;IAUxD;;;;;;;;;;OAUG;WACW,iBAAiB,CAAC,OAAO,EAAE,MAAM,EAAE,QAAQ,EAAE,MAAM,GAAG,MAAM;IAa1E;;;OAGG;WACW,oBAAoB,CAAC,IAAI,EAAE,MAAM,EAAE,SAAS,EAAE,MAAM,GAAG,MAAM;IAe3E,OAAO,CAAC,MAAM,CAAC,kBAAkB;IA6BjC,OAAO,CAAC,MAAM,CAAC,mBAAmB;CASrC"}
|
|
@@ -0,0 +1,126 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview Shared text extraction utilities for various content formats.
|
|
3
|
+
*
|
|
4
|
+
* Provides a unified interface for extracting plain text from HTML, plain text,
|
|
5
|
+
* and other content types. Used by both the vectorization and autotagging pipelines.
|
|
6
|
+
*
|
|
7
|
+
* @module @memberjunction/ai-vectors
|
|
8
|
+
*/
|
|
9
|
+
/**
|
|
10
|
+
* Shared text extraction utilities.
|
|
11
|
+
*
|
|
12
|
+
* Provides static methods for extracting clean text from various content formats.
|
|
13
|
+
* These utilities are intentionally dependency-light — they use regex-based extraction
|
|
14
|
+
* rather than heavy DOM parsing libraries, making them suitable for server-side use
|
|
15
|
+
* without browser dependencies.
|
|
16
|
+
*/
|
|
17
|
+
export class TextExtractor {
|
|
18
|
+
/**
|
|
19
|
+
* Extract readable text from HTML content.
|
|
20
|
+
* Strips tags, decodes entities, normalizes whitespace.
|
|
21
|
+
*/
|
|
22
|
+
static ExtractFromHTML(html) {
|
|
23
|
+
if (!html || html.trim().length === 0)
|
|
24
|
+
return '';
|
|
25
|
+
let text = html;
|
|
26
|
+
// Remove script and style elements entirely
|
|
27
|
+
text = text.replace(/<script[^>]*>[\s\S]*?<\/script>/gi, '');
|
|
28
|
+
text = text.replace(/<style[^>]*>[\s\S]*?<\/style>/gi, '');
|
|
29
|
+
// Replace block-level elements with newlines
|
|
30
|
+
text = text.replace(/<\/(p|div|h[1-6]|li|tr|br|blockquote|pre)>/gi, '\n');
|
|
31
|
+
text = text.replace(/<br\s*\/?>/gi, '\n');
|
|
32
|
+
text = text.replace(/<(p|div|h[1-6]|li|tr|blockquote|pre)[^>]*>/gi, '\n');
|
|
33
|
+
// Remove remaining HTML tags
|
|
34
|
+
text = text.replace(/<[^>]+>/g, ' ');
|
|
35
|
+
// Decode common HTML entities
|
|
36
|
+
text = TextExtractor.decodeHTMLEntities(text);
|
|
37
|
+
// Normalize whitespace
|
|
38
|
+
text = TextExtractor.normalizeWhitespace(text);
|
|
39
|
+
return text.trim();
|
|
40
|
+
}
|
|
41
|
+
/**
|
|
42
|
+
* Extract and normalize plain text.
|
|
43
|
+
* Trims, normalizes whitespace, and removes control characters.
|
|
44
|
+
*/
|
|
45
|
+
static ExtractFromPlainText(text) {
|
|
46
|
+
if (!text || text.trim().length === 0)
|
|
47
|
+
return '';
|
|
48
|
+
// Remove control characters except newlines and tabs
|
|
49
|
+
let cleaned = text.replace(/[\x00-\x08\x0B\x0C\x0E-\x1F\x7F]/g, '');
|
|
50
|
+
cleaned = TextExtractor.normalizeWhitespace(cleaned);
|
|
51
|
+
return cleaned.trim();
|
|
52
|
+
}
|
|
53
|
+
/**
|
|
54
|
+
* Detect content type from a MIME type string and extract text accordingly.
|
|
55
|
+
*
|
|
56
|
+
* Currently supports:
|
|
57
|
+
* - text/html → ExtractFromHTML
|
|
58
|
+
* - text/plain → ExtractFromPlainText
|
|
59
|
+
* - text/* → ExtractFromPlainText (fallback)
|
|
60
|
+
*
|
|
61
|
+
* For binary formats (PDF, DOCX, etc.), callers should use dedicated libraries
|
|
62
|
+
* (pdf-parse, officeparser) and then pass the extracted text through ExtractFromPlainText.
|
|
63
|
+
*/
|
|
64
|
+
static ExtractByMimeType(content, mimeType) {
|
|
65
|
+
if (!content)
|
|
66
|
+
return '';
|
|
67
|
+
const normalizedMime = mimeType.toLowerCase().trim();
|
|
68
|
+
if (normalizedMime.includes('html')) {
|
|
69
|
+
return TextExtractor.ExtractFromHTML(content);
|
|
70
|
+
}
|
|
71
|
+
// All other text types
|
|
72
|
+
return TextExtractor.ExtractFromPlainText(content);
|
|
73
|
+
}
|
|
74
|
+
/**
|
|
75
|
+
* Truncate text to a maximum token count (estimated).
|
|
76
|
+
* Truncates at the last whitespace boundary before the limit.
|
|
77
|
+
*/
|
|
78
|
+
static TruncateToTokenLimit(text, maxTokens) {
|
|
79
|
+
if (!text)
|
|
80
|
+
return '';
|
|
81
|
+
const estimatedChars = maxTokens * 4; // rough estimate: ~4 chars per token
|
|
82
|
+
if (text.length <= estimatedChars)
|
|
83
|
+
return text;
|
|
84
|
+
const truncated = text.slice(0, estimatedChars);
|
|
85
|
+
const lastSpace = truncated.lastIndexOf(' ');
|
|
86
|
+
return lastSpace > 0 ? truncated.slice(0, lastSpace) : truncated;
|
|
87
|
+
}
|
|
88
|
+
// ─────────────────────────────────────────────
|
|
89
|
+
// Private Helpers
|
|
90
|
+
// ─────────────────────────────────────────────
|
|
91
|
+
static decodeHTMLEntities(text) {
|
|
92
|
+
const entityMap = {
|
|
93
|
+
'&': '&',
|
|
94
|
+
'<': '<',
|
|
95
|
+
'>': '>',
|
|
96
|
+
'"': '"',
|
|
97
|
+
''': "'",
|
|
98
|
+
''': "'",
|
|
99
|
+
' ': ' ',
|
|
100
|
+
'—': '—',
|
|
101
|
+
'–': '–',
|
|
102
|
+
'…': '…',
|
|
103
|
+
'©': '©',
|
|
104
|
+
'®': '®',
|
|
105
|
+
'™': '™',
|
|
106
|
+
};
|
|
107
|
+
let decoded = text;
|
|
108
|
+
for (const [entity, char] of Object.entries(entityMap)) {
|
|
109
|
+
decoded = decoded.split(entity).join(char);
|
|
110
|
+
}
|
|
111
|
+
// Decode numeric entities (decimal and hex)
|
|
112
|
+
decoded = decoded.replace(/&#(\d+);/g, (_, code) => String.fromCharCode(parseInt(code, 10)));
|
|
113
|
+
decoded = decoded.replace(/&#x([0-9a-fA-F]+);/g, (_, code) => String.fromCharCode(parseInt(code, 16)));
|
|
114
|
+
return decoded;
|
|
115
|
+
}
|
|
116
|
+
static normalizeWhitespace(text) {
|
|
117
|
+
// Collapse multiple spaces to single space
|
|
118
|
+
let normalized = text.replace(/[ \t]+/g, ' ');
|
|
119
|
+
// Collapse 3+ newlines to 2
|
|
120
|
+
normalized = normalized.replace(/\n{3,}/g, '\n\n');
|
|
121
|
+
// Remove spaces at beginning of lines
|
|
122
|
+
normalized = normalized.replace(/\n /g, '\n');
|
|
123
|
+
return normalized;
|
|
124
|
+
}
|
|
125
|
+
}
|
|
126
|
+
//# sourceMappingURL=TextExtractor.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"TextExtractor.js","sourceRoot":"","sources":["../../src/generic/TextExtractor.ts"],"names":[],"mappings":"AAAA;;;;;;;GAOG;AAEH;;;;;;;GAOG;AACH,MAAM,OAAO,aAAa;IACtB;;;OAGG;IACI,MAAM,CAAC,eAAe,CAAC,IAAY;QACtC,IAAI,CAAC,IAAI,IAAI,IAAI,CAAC,IAAI,EAAE,CAAC,MAAM,KAAK,CAAC;YAAE,OAAO,EAAE,CAAC;QAEjD,IAAI,IAAI,GAAG,IAAI,CAAC;QAEhB,4CAA4C;QAC5C,IAAI,GAAG,IAAI,CAAC,OAAO,CAAC,mCAAmC,EAAE,EAAE,CAAC,CAAC;QAC7D,IAAI,GAAG,IAAI,CAAC,OAAO,CAAC,iCAAiC,EAAE,EAAE,CAAC,CAAC;QAE3D,6CAA6C;QAC7C,IAAI,GAAG,IAAI,CAAC,OAAO,CAAC,8CAA8C,EAAE,IAAI,CAAC,CAAC;QAC1E,IAAI,GAAG,IAAI,CAAC,OAAO,CAAC,cAAc,EAAE,IAAI,CAAC,CAAC;QAC1C,IAAI,GAAG,IAAI,CAAC,OAAO,CAAC,8CAA8C,EAAE,IAAI,CAAC,CAAC;QAE1E,6BAA6B;QAC7B,IAAI,GAAG,IAAI,CAAC,OAAO,CAAC,UAAU,EAAE,GAAG,CAAC,CAAC;QAErC,8BAA8B;QAC9B,IAAI,GAAG,aAAa,CAAC,kBAAkB,CAAC,IAAI,CAAC,CAAC;QAE9C,uBAAuB;QACvB,IAAI,GAAG,aAAa,CAAC,mBAAmB,CAAC,IAAI,CAAC,CAAC;QAE/C,OAAO,IAAI,CAAC,IAAI,EAAE,CAAC;IACvB,CAAC;IAED;;;OAGG;IACI,MAAM,CAAC,oBAAoB,CAAC,IAAY;QAC3C,IAAI,CAAC,IAAI,IAAI,IAAI,CAAC,IAAI,EAAE,CAAC,MAAM,KAAK,CAAC;YAAE,OAAO,EAAE,CAAC;QAEjD,qDAAqD;QACrD,IAAI,OAAO,GAAG,IAAI,CAAC,OAAO,CAAC,mCAAmC,EAAE,EAAE,CAAC,CAAC;QAEpE,OAAO,GAAG,aAAa,CAAC,mBAAmB,CAAC,OAAO,CAAC,CAAC;QACrD,OAAO,OAAO,CAAC,IAAI,EAAE,CAAC;IAC1B,CAAC;IAED;;;;;;;;;;OAUG;IACI,MAAM,CAAC,iBAAiB,CAAC,OAAe,EAAE,QAAgB;QAC7D,IAAI,CAAC,OAAO;YAAE,OAAO,EAAE,CAAC;QAExB,MAAM,cAAc,GAAG,QAAQ,CAAC,WAAW,EAAE,CAAC,IAAI,EAAE,CAAC;QAErD,IAAI,cAAc,CAAC,QAAQ,CAAC,MAAM,CAAC,EAAE,CAAC;YAClC,OAAO,aAAa,CAAC,eAAe,CAAC,OAAO,CAAC,CAAC;QAClD,CAAC;QAED,uBAAuB;QACvB,OAAO,aAAa,CAAC,oBAAoB,CAAC,OAAO,CAAC,CAAC;IACvD,CAAC;IAED;;;OAGG;IACI,MAAM,CAAC,oBAAoB,CAAC,IAAY,EAAE,SAAiB;QAC9D,IAAI,CAAC,IAAI;YAAE,OAAO,EAAE,CAAC;QAErB,MAAM,cAAc,GAAG,SAAS,GAAG,CAAC,CAAC,CAAC,qCAAqC;QAC3E,IAAI,IAAI,CAAC,MAAM,IAAI,cAAc;YAAE,OAAO,IAAI,CAAC;QAE/C,MAAM,SAAS,GAAG,IAAI,CAAC,KAAK,CAAC,CAAC,EAAE,cAAc,CAAC,CAAC;QAChD,MAAM,SAAS,GAAG,SAAS,CAAC,WAAW,CAAC,GAAG,CAAC,CAAC;QAC7C,OAAO,SAAS,GAAG,CAAC,CAAC,CAAC,CAAC,SAAS,CAAC,KAAK,CAAC,CAAC,EAAE,SAAS,CAAC,CAAC,CAAC,CAAC,SAAS,CAAC;IACrE,CAAC;IAED,gDAAgD;IAChD,kBAAkB;IAClB,gDAAgD;IAExC,MAAM,CAAC,kBAAkB,CAAC,IAAY;QAC1C,MAAM,SAAS,GAA2B;YACtC,OAAO,EAAE,GAAG;YACZ,MAAM,EAAE,GAAG;YACX,MAAM,EAAE,GAAG;YACX,QAAQ,EAAE,GAAG;YACb,OAAO,EAAE,GAAG;YACZ,QAAQ,EAAE,GAAG;YACb,QAAQ,EAAE,GAAG;YACb,SAAS,EAAE,GAAG;YACd,SAAS,EAAE,GAAG;YACd,UAAU,EAAE,GAAG;YACf,QAAQ,EAAE,GAAG;YACb,OAAO,EAAE,GAAG;YACZ,SAAS,EAAE,GAAG;SACjB,CAAC;QAEF,IAAI,OAAO,GAAG,IAAI,CAAC;QACnB,KAAK,MAAM,CAAC,MAAM,EAAE,IAAI,CAAC,IAAI,MAAM,CAAC,OAAO,CAAC,SAAS,CAAC,EAAE,CAAC;YACrD,OAAO,GAAG,OAAO,CAAC,KAAK,CAAC,MAAM,CAAC,CAAC,IAAI,CAAC,IAAI,CAAC,CAAC;QAC/C,CAAC;QAED,4CAA4C;QAC5C,OAAO,GAAG,OAAO,CAAC,OAAO,CAAC,WAAW,EAAE,CAAC,CAAC,EAAE,IAAI,EAAE,EAAE,CAAC,MAAM,CAAC,YAAY,CAAC,QAAQ,CAAC,IAAI,EAAE,EAAE,CAAC,CAAC,CAAC,CAAC;QAC7F,OAAO,GAAG,OAAO,CAAC,OAAO,CAAC,qBAAqB,EAAE,CAAC,CAAC,EAAE,IAAI,EAAE,EAAE,CAAC,MAAM,CAAC,YAAY,CAAC,QAAQ,CAAC,IAAI,EAAE,EAAE,CAAC,CAAC,CAAC,CAAC;QAEvG,OAAO,OAAO,CAAC;IACnB,CAAC;IAEO,MAAM,CAAC,mBAAmB,CAAC,IAAY;QAC3C,2CAA2C;QAC3C,IAAI,UAAU,GAAG,IAAI,CAAC,OAAO,CAAC,SAAS,EAAE,GAAG,CAAC,CAAC;QAC9C,4BAA4B;QAC5B,UAAU,GAAG,UAAU,CAAC,OAAO,CAAC,SAAS,EAAE,MAAM,CAAC,CAAC;QACnD,sCAAsC;QACtC,UAAU,GAAG,UAAU,CAAC,OAAO,CAAC,MAAM,EAAE,IAAI,CAAC,CAAC;QAC9C,OAAO,UAAU,CAAC;IACtB,CAAC;CACJ"}
|
package/dist/index.d.ts
CHANGED
|
@@ -3,4 +3,7 @@ export * from './generic/IVectorIndex.js';
|
|
|
3
3
|
export * from './generic/IEmbedding.js';
|
|
4
4
|
export * from './models/VectorBase.js';
|
|
5
5
|
export * from './generic/VectorCore.types.js';
|
|
6
|
+
export * from './generic/TextChunker.js';
|
|
7
|
+
export * from './generic/TextExtractor.js';
|
|
8
|
+
export * from './generic/SharedIndexMetadata.js';
|
|
6
9
|
//# sourceMappingURL=index.d.ts.map
|
package/dist/index.d.ts.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"index.d.ts","sourceRoot":"","sources":["../src/index.ts"],"names":[],"mappings":"AAAA,cAAc,2BAA2B,CAAC;AAC1C,cAAc,wBAAwB,CAAC;AACvC,cAAc,sBAAsB,CAAC;AACrC,cAAc,qBAAqB,CAAC;AACpC,cAAc,4BAA4B,CAAC"}
|
|
1
|
+
{"version":3,"file":"index.d.ts","sourceRoot":"","sources":["../src/index.ts"],"names":[],"mappings":"AAAA,cAAc,2BAA2B,CAAC;AAC1C,cAAc,wBAAwB,CAAC;AACvC,cAAc,sBAAsB,CAAC;AACrC,cAAc,qBAAqB,CAAC;AACpC,cAAc,4BAA4B,CAAC;AAC3C,cAAc,uBAAuB,CAAC;AACtC,cAAc,yBAAyB,CAAC;AACxC,cAAc,+BAA+B,CAAC"}
|
package/dist/index.js
CHANGED
|
@@ -3,4 +3,7 @@ export * from './generic/IVectorIndex.js';
|
|
|
3
3
|
export * from './generic/IEmbedding.js';
|
|
4
4
|
export * from './models/VectorBase.js';
|
|
5
5
|
export * from './generic/VectorCore.types.js';
|
|
6
|
+
export * from './generic/TextChunker.js';
|
|
7
|
+
export * from './generic/TextExtractor.js';
|
|
8
|
+
export * from './generic/SharedIndexMetadata.js';
|
|
6
9
|
//# sourceMappingURL=index.js.map
|
package/dist/index.js.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"index.js","sourceRoot":"","sources":["../src/index.ts"],"names":[],"mappings":"AAAA,cAAc,2BAA2B,CAAC;AAC1C,cAAc,wBAAwB,CAAC;AACvC,cAAc,sBAAsB,CAAC;AACrC,cAAc,qBAAqB,CAAC;AACpC,cAAc,4BAA4B,CAAC"}
|
|
1
|
+
{"version":3,"file":"index.js","sourceRoot":"","sources":["../src/index.ts"],"names":[],"mappings":"AAAA,cAAc,2BAA2B,CAAC;AAC1C,cAAc,wBAAwB,CAAC;AACvC,cAAc,sBAAsB,CAAC;AACrC,cAAc,qBAAqB,CAAC;AACpC,cAAc,4BAA4B,CAAC;AAC3C,cAAc,uBAAuB,CAAC;AACtC,cAAc,yBAAyB,CAAC;AACxC,cAAc,+BAA+B,CAAC"}
|
|
@@ -13,7 +13,10 @@ export declare class VectorBase {
|
|
|
13
13
|
set CurrentUser(user: UserInfo);
|
|
14
14
|
protected GetRecordsByEntityID(entityID: string, recordIDs?: CompositeKey[]): Promise<BaseEntity[]>;
|
|
15
15
|
protected PageRecordsByEntityID<T>(params: PageRecordsParams): Promise<T[]>;
|
|
16
|
-
|
|
16
|
+
/**
|
|
17
|
+
* Builds a SQL filter from composite keys. Values are sanitized to prevent SQL injection.
|
|
18
|
+
*/
|
|
19
|
+
protected BuildExtraFilter(compositeKeys: CompositeKey[]): string;
|
|
17
20
|
protected GetAIModel(id?: string): MJAIModelEntityExtended;
|
|
18
21
|
protected GetVectorDatabase(id?: string): MJVectorDatabaseEntity;
|
|
19
22
|
protected RunViewForSingleValue<T extends BaseEntity>(entityName: string, extraFilter: string): Promise<T | null>;
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"VectorBase.d.ts","sourceRoot":"","sources":["../../src/models/VectorBase.ts"],"names":[],"mappings":"AACA,OAAO,EAAE,UAAU,EAAE,QAAQ,EAAE,YAAY,EAAE,OAAO,EAAE,QAAQ,EAAuC,MAAM,sBAAsB,CAAC;AAClI,OAAO,EAAE,sBAAsB,EAAE,MAAM,+BAA+B,CAAC;AAEvE,OAAO,EAAE,iBAAiB,EAAE,MAAM,6BAA6B,CAAC;AAChE,OAAO,EAAE,uBAAuB,EAAE,MAAM,8BAA8B,CAAC;AAEvE,qBAAa,UAAU;IACnB,QAAQ,EAAE,OAAO,CAAC;IAClB,SAAS,EAAE,QAAQ,CAAC;IACpB,YAAY,EAAE,QAAQ,CAAC;;IAQvB,IAAW,QAAQ,IAAI,QAAQ,CAA2B;IAC1D,IAAW,OAAO,IAAI,OAAO,CAA0B;IACvD,IAAW,WAAW,IAAI,QAAQ,CAA8B;IAChE,IAAW,WAAW,CAAC,IAAI,EAAE,QAAQ,EAA+B;cAEpD,oBAAoB,CAAC,QAAQ,EAAE,MAAM,EAAE,SAAS,CAAC,EAAE,YAAY,EAAE,GAAG,OAAO,CAAC,UAAU,EAAE,CAAC;cAqBzF,qBAAqB,CAAC,CAAC,EAAE,MAAM,EAAE,iBAAiB,GAAG,OAAO,CAAC,CAAC,EAAE,CAAC;IAqBjF,SAAS,CAAC,gBAAgB,CAAC,
|
|
1
|
+
{"version":3,"file":"VectorBase.d.ts","sourceRoot":"","sources":["../../src/models/VectorBase.ts"],"names":[],"mappings":"AACA,OAAO,EAAE,UAAU,EAAE,QAAQ,EAAE,YAAY,EAAE,OAAO,EAAE,QAAQ,EAAuC,MAAM,sBAAsB,CAAC;AAClI,OAAO,EAAE,sBAAsB,EAAE,MAAM,+BAA+B,CAAC;AAEvE,OAAO,EAAE,iBAAiB,EAAE,MAAM,6BAA6B,CAAC;AAChE,OAAO,EAAE,uBAAuB,EAAE,MAAM,8BAA8B,CAAC;AAEvE,qBAAa,UAAU;IACnB,QAAQ,EAAE,OAAO,CAAC;IAClB,SAAS,EAAE,QAAQ,CAAC;IACpB,YAAY,EAAE,QAAQ,CAAC;;IAQvB,IAAW,QAAQ,IAAI,QAAQ,CAA2B;IAC1D,IAAW,OAAO,IAAI,OAAO,CAA0B;IACvD,IAAW,WAAW,IAAI,QAAQ,CAA8B;IAChE,IAAW,WAAW,CAAC,IAAI,EAAE,QAAQ,EAA+B;cAEpD,oBAAoB,CAAC,QAAQ,EAAE,MAAM,EAAE,SAAS,CAAC,EAAE,YAAY,EAAE,GAAG,OAAO,CAAC,UAAU,EAAE,CAAC;cAqBzF,qBAAqB,CAAC,CAAC,EAAE,MAAM,EAAE,iBAAiB,GAAG,OAAO,CAAC,CAAC,EAAE,CAAC;IAqBjF;;OAEG;IACH,SAAS,CAAC,gBAAgB,CAAC,aAAa,EAAE,YAAY,EAAE,GAAG,MAAM;IASjE,SAAS,CAAC,UAAU,CAAC,EAAE,CAAC,EAAE,MAAM,GAAG,uBAAuB;IAe1D,SAAS,CAAC,iBAAiB,CAAC,EAAE,CAAC,EAAE,MAAM,GAAG,sBAAsB;cAgBhD,qBAAqB,CAAC,CAAC,SAAS,UAAU,EAAE,UAAU,EAAE,MAAM,EAAE,WAAW,EAAE,MAAM,GAAG,OAAO,CAAC,CAAC,GAAG,IAAI,CAAC;IAgBvH;;;QAGI;cACY,UAAU,CAAC,MAAM,EAAE,UAAU,GAAG,OAAO,CAAC,OAAO,CAAC;CAInE"}
|
|
@@ -45,10 +45,14 @@ export class VectorBase {
|
|
|
45
45
|
}
|
|
46
46
|
return rvResult.Results;
|
|
47
47
|
}
|
|
48
|
-
|
|
49
|
-
|
|
48
|
+
/**
|
|
49
|
+
* Builds a SQL filter from composite keys. Values are sanitized to prevent SQL injection.
|
|
50
|
+
*/
|
|
51
|
+
BuildExtraFilter(compositeKeys) {
|
|
52
|
+
return compositeKeys.map((keyValue) => {
|
|
50
53
|
return keyValue.KeyValuePairs.map((keys) => {
|
|
51
|
-
|
|
54
|
+
const sanitizedValue = String(keys.Value).replace(/'/g, "''");
|
|
55
|
+
return `${keys.FieldName} = '${sanitizedValue}'`;
|
|
52
56
|
}).join(" AND ");
|
|
53
57
|
}).join("\n OR ");
|
|
54
58
|
}
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"VectorBase.js","sourceRoot":"","sources":["../../src/models/VectorBase.ts"],"names":[],"mappings":"AAAA,OAAO,EAAE,QAAQ,EAAE,MAAM,0BAA0B,CAAC;AACpD,OAAO,EAAc,QAAQ,EAAgB,OAAO,EAAuC,QAAQ,EAAE,MAAM,sBAAsB,CAAC;AAElI,OAAO,EAAE,UAAU,EAAE,MAAM,wBAAwB,CAAC;AAIpD,MAAM,OAAO,UAAU;IAKnB;QACI,IAAI,CAAC,QAAQ,GAAG,IAAI,OAAO,EAAE,CAAC;QAC9B,IAAI,CAAC,SAAS,GAAG,IAAI,QAAQ,EAAE,CAAC;QAChC,IAAI,CAAC,YAAY,GAAG,IAAI,CAAC,SAAS,CAAC,WAAW,CAAC;IACnD,CAAC;IAED,IAAW,QAAQ,KAAe,OAAO,IAAI,CAAC,SAAS,CAAC,CAAC,CAAC;IAC1D,IAAW,OAAO,KAAc,OAAO,IAAI,CAAC,QAAQ,CAAC,CAAC,CAAC;IACvD,IAAW,WAAW,KAAe,OAAO,IAAI,CAAC,YAAY,CAAC,CAAC,CAAC;IAChE,IAAW,WAAW,CAAC,IAAc,IAAI,IAAI,CAAC,YAAY,GAAG,IAAI,CAAC,CAAC,CAAC;IAE1D,KAAK,CAAC,oBAAoB,CAAC,QAAgB,EAAE,SAA0B;QAC7E,MAAM,EAAE,GAAG,IAAI,QAAQ,EAAE,CAAC;QAC1B,MAAM,MAAM,GAAG,EAAE,CAAC,QAAQ,CAAC,IAAI,CAAC,CAAC,CAAC,EAAE,CAAC,UAAU,CAAC,CAAC,CAAC,EAAE,EAAE,QAAQ,CAAC,CAAC,CAAC;QACjE,IAAI,CAAC,MAAM,EAAC,CAAC;YACT,MAAM,IAAI,KAAK,CAAC,kBAAkB,QAAQ,aAAa,CAAC,CAAC;QAC7D,CAAC;QAED,MAAM,QAAQ,GAAG,MAAM,IAAI,CAAC,QAAQ,CAAC,OAAO,CAAa;YACrD,UAAU,EAAE,MAAM,CAAC,IAAI;YACvB,WAAW,EAAE,SAAS,CAAC,CAAC,CAAC,IAAI,CAAC,gBAAgB,CAAC,SAAS,CAAC,CAAA,CAAC,CAAC,SAAS;YACpE,UAAU,EAAE,eAAe;YAC3B,aAAa,EAAE,IAAI;SACtB,EAAE,IAAI,CAAC,WAAW,CAAC,CAAC;QAErB,IAAG,CAAC,QAAQ,CAAC,OAAO,EAAC,CAAC;YAClB,MAAM,IAAI,KAAK,CAAC,QAAQ,CAAC,YAAY,CAAC,CAAC;QAC3C,CAAC;QAED,OAAO,QAAQ,CAAC,OAAO,CAAC;IAC5B,CAAC;IAES,KAAK,CAAC,qBAAqB,CAAI,MAAyB;QAC9D,MAAM,MAAM,GAA2B,IAAI,CAAC,QAAQ,CAAC,QAAQ,CAAC,IAAI,CAAC,CAAC,CAAC,EAAE,EAAE,CAAC,UAAU,CAAC,CAAC,CAAC,EAAE,EAAE,MAAM,CAAC,QAAkB,CAAC,CAAC,CAAC;QACvH,IAAI,CAAC,MAAM,EAAE,CAAC;YACZ,MAAM,IAAI,KAAK,CAAC,kBAAkB,MAAM,CAAC,QAAQ,aAAa,CAAC,CAAC;QAClE,CAAC;QAED,MAAM,QAAQ,GAAqB,MAAM,IAAI,CAAC,QAAQ,CAAC,OAAO,CAAI;YAC9D,UAAU,EAAE,MAAM,CAAC,IAAI;YACvB,UAAU,EAAE,MAAM,CAAC,UAAU;YAC7B,OAAO,EAAE,MAAM,CAAC,QAAQ;YACxB,QAAQ,EAAE,IAAI,CAAC,GAAG,CAAC,CAAC,EAAE,CAAC,MAAM,CAAC,UAAU,GAAG,CAAC,CAAC,GAAG,MAAM,CAAC,QAAQ,CAAC;YAChE,WAAW,EAAE,MAAM,CAAC,MAAM;SAC7B,EAAE,IAAI,CAAC,WAAW,CAAC,CAAC;QAErB,IAAI,CAAC,QAAQ,CAAC,OAAO,EAAE,CAAC;YACtB,MAAM,IAAI,KAAK,CAAC,QAAQ,CAAC,YAAY,CAAC,CAAC;QACzC,CAAC;QAED,OAAO,QAAQ,CAAC,OAAO,CAAC;IAC5B,CAAC;
|
|
1
|
+
{"version":3,"file":"VectorBase.js","sourceRoot":"","sources":["../../src/models/VectorBase.ts"],"names":[],"mappings":"AAAA,OAAO,EAAE,QAAQ,EAAE,MAAM,0BAA0B,CAAC;AACpD,OAAO,EAAc,QAAQ,EAAgB,OAAO,EAAuC,QAAQ,EAAE,MAAM,sBAAsB,CAAC;AAElI,OAAO,EAAE,UAAU,EAAE,MAAM,wBAAwB,CAAC;AAIpD,MAAM,OAAO,UAAU;IAKnB;QACI,IAAI,CAAC,QAAQ,GAAG,IAAI,OAAO,EAAE,CAAC;QAC9B,IAAI,CAAC,SAAS,GAAG,IAAI,QAAQ,EAAE,CAAC;QAChC,IAAI,CAAC,YAAY,GAAG,IAAI,CAAC,SAAS,CAAC,WAAW,CAAC;IACnD,CAAC;IAED,IAAW,QAAQ,KAAe,OAAO,IAAI,CAAC,SAAS,CAAC,CAAC,CAAC;IAC1D,IAAW,OAAO,KAAc,OAAO,IAAI,CAAC,QAAQ,CAAC,CAAC,CAAC;IACvD,IAAW,WAAW,KAAe,OAAO,IAAI,CAAC,YAAY,CAAC,CAAC,CAAC;IAChE,IAAW,WAAW,CAAC,IAAc,IAAI,IAAI,CAAC,YAAY,GAAG,IAAI,CAAC,CAAC,CAAC;IAE1D,KAAK,CAAC,oBAAoB,CAAC,QAAgB,EAAE,SAA0B;QAC7E,MAAM,EAAE,GAAG,IAAI,QAAQ,EAAE,CAAC;QAC1B,MAAM,MAAM,GAAG,EAAE,CAAC,QAAQ,CAAC,IAAI,CAAC,CAAC,CAAC,EAAE,CAAC,UAAU,CAAC,CAAC,CAAC,EAAE,EAAE,QAAQ,CAAC,CAAC,CAAC;QACjE,IAAI,CAAC,MAAM,EAAC,CAAC;YACT,MAAM,IAAI,KAAK,CAAC,kBAAkB,QAAQ,aAAa,CAAC,CAAC;QAC7D,CAAC;QAED,MAAM,QAAQ,GAAG,MAAM,IAAI,CAAC,QAAQ,CAAC,OAAO,CAAa;YACrD,UAAU,EAAE,MAAM,CAAC,IAAI;YACvB,WAAW,EAAE,SAAS,CAAC,CAAC,CAAC,IAAI,CAAC,gBAAgB,CAAC,SAAS,CAAC,CAAA,CAAC,CAAC,SAAS;YACpE,UAAU,EAAE,eAAe;YAC3B,aAAa,EAAE,IAAI;SACtB,EAAE,IAAI,CAAC,WAAW,CAAC,CAAC;QAErB,IAAG,CAAC,QAAQ,CAAC,OAAO,EAAC,CAAC;YAClB,MAAM,IAAI,KAAK,CAAC,QAAQ,CAAC,YAAY,CAAC,CAAC;QAC3C,CAAC;QAED,OAAO,QAAQ,CAAC,OAAO,CAAC;IAC5B,CAAC;IAES,KAAK,CAAC,qBAAqB,CAAI,MAAyB;QAC9D,MAAM,MAAM,GAA2B,IAAI,CAAC,QAAQ,CAAC,QAAQ,CAAC,IAAI,CAAC,CAAC,CAAC,EAAE,EAAE,CAAC,UAAU,CAAC,CAAC,CAAC,EAAE,EAAE,MAAM,CAAC,QAAkB,CAAC,CAAC,CAAC;QACvH,IAAI,CAAC,MAAM,EAAE,CAAC;YACZ,MAAM,IAAI,KAAK,CAAC,kBAAkB,MAAM,CAAC,QAAQ,aAAa,CAAC,CAAC;QAClE,CAAC;QAED,MAAM,QAAQ,GAAqB,MAAM,IAAI,CAAC,QAAQ,CAAC,OAAO,CAAI;YAC9D,UAAU,EAAE,MAAM,CAAC,IAAI;YACvB,UAAU,EAAE,MAAM,CAAC,UAAU;YAC7B,OAAO,EAAE,MAAM,CAAC,QAAQ;YACxB,QAAQ,EAAE,IAAI,CAAC,GAAG,CAAC,CAAC,EAAE,CAAC,MAAM,CAAC,UAAU,GAAG,CAAC,CAAC,GAAG,MAAM,CAAC,QAAQ,CAAC;YAChE,WAAW,EAAE,MAAM,CAAC,MAAM;SAC7B,EAAE,IAAI,CAAC,WAAW,CAAC,CAAC;QAErB,IAAI,CAAC,QAAQ,CAAC,OAAO,EAAE,CAAC;YACtB,MAAM,IAAI,KAAK,CAAC,QAAQ,CAAC,YAAY,CAAC,CAAC;QACzC,CAAC;QAED,OAAO,QAAQ,CAAC,OAAO,CAAC;IAC5B,CAAC;IAED;;OAEG;IACO,gBAAgB,CAAC,aAA6B;QACpD,OAAO,aAAa,CAAC,GAAG,CAAC,CAAC,QAAQ,EAAE,EAAE;YAClC,OAAO,QAAQ,CAAC,aAAa,CAAC,GAAG,CAAC,CAAC,IAAI,EAAE,EAAE;gBACvC,MAAM,cAAc,GAAG,MAAM,CAAC,IAAI,CAAC,KAAK,CAAC,CAAC,OAAO,CAAC,IAAI,EAAE,IAAI,CAAC,CAAC;gBAC9D,OAAO,GAAG,IAAI,CAAC,SAAS,OAAO,cAAc,GAAG,CAAC;YACrD,CAAC,CAAC,CAAC,IAAI,CAAC,OAAO,CAAC,CAAC;QACrB,CAAC,CAAC,CAAC,IAAI,CAAC,QAAQ,CAAC,CAAC;IACtB,CAAC;IAES,UAAU,CAAC,EAAW;QAC5B,IAAI,KAA8B,CAAC;QACnC,IAAG,EAAE,EAAC,CAAC;YACH,KAAK,GAAG,QAAQ,CAAC,QAAQ,CAAC,MAAM,CAAC,IAAI,CAAC,CAAC,CAAC,EAAE,CAAC,CAAC,CAAC,WAAW,KAAK,YAAY,IAAI,UAAU,CAAC,CAAC,CAAC,EAAE,EAAE,EAAE,CAAC,CAAC,CAAC;QACvG,CAAC;aACG,CAAC;YACD,KAAK,GAAG,QAAQ,CAAC,QAAQ,CAAC,MAAM,CAAC,IAAI,CAAC,CAAC,CAAC,EAAE,CAAC,CAAC,CAAC,WAAW,KAAK,YAAY,CAAC,CAAC;QAC/E,CAAC;QAED,IAAG,CAAC,KAAK,EAAC,CAAC;YACP,MAAM,IAAI,KAAK,CAAC,0BAA0B,CAAC,CAAC;QAChD,CAAC;QACD,OAAO,KAAK,CAAC;IACjB,CAAC;IAES,iBAAiB,CAAC,EAAW;QACnC,IAAG,QAAQ,CAAC,QAAQ,CAAC,eAAe,CAAC,MAAM,GAAG,CAAC,EAAC,CAAC;YAC7C,IAAG,EAAE,EAAC,CAAC;gBACH,IAAI,QAAQ,GAAG,QAAQ,CAAC,QAAQ,CAAC,eAAe,CAAC,IAAI,CAAC,EAAE,CAAC,EAAE,CAAC,UAAU,CAAC,EAAE,CAAC,EAAE,EAAE,EAAE,CAAC,CAAC,CAAC;gBACnF,IAAG,QAAQ,EAAC,CAAC;oBACT,OAAO,QAAQ,CAAC;gBACpB,CAAC;YACL,CAAC;iBACG,CAAC;gBACD,OAAO,QAAQ,CAAC,QAAQ,CAAC,eAAe,CAAC,CAAC,CAAC,CAAC;YAChD,CAAC;QACL,CAAC;QAED,MAAM,IAAI,KAAK,CAAC,iCAAiC,CAAC,CAAC;IACvD,CAAC;IAES,KAAK,CAAC,qBAAqB,CAAuB,UAAkB,EAAE,WAAmB;QAC/F,MAAM,QAAQ,GAAG,MAAM,IAAI,CAAC,QAAQ,CAAC,OAAO,CAAC;YACzC,UAAU,EAAE,UAAU;YACtB,WAAW,EAAE,WAAW;YACxB,UAAU,EAAE,eAAe;SAC9B,EAAE,IAAI,CAAC,WAAW,CAAC,CAAC;QAErB,IAAG,QAAQ,CAAC,OAAO,EAAC,CAAC;YACjB,OAAO,QAAQ,CAAC,QAAQ,GAAG,CAAC,CAAC,CAAC,CAAC,QAAQ,CAAC,OAAO,CAAC,CAAC,CAAM,CAAA,CAAC,CAAC,IAAI,CAAC;QAClE,CAAC;aACG,CAAC;YACD,QAAQ,CAAC,QAAQ,CAAC,YAAY,CAAC,CAAC;YAChC,OAAO,IAAI,CAAC;QAChB,CAAC;IACL,CAAC;IAED;;;QAGI;IACM,KAAK,CAAC,UAAU,CAAC,MAAkB;QACzC,MAAM,CAAC,kBAAkB,GAAG,IAAI,CAAC,WAAW,CAAC;QAC7C,OAAO,MAAM,MAAM,CAAC,IAAI,EAAE,CAAC;IAC/B,CAAC;CACJ"}
|