@memberjunction/ai-segmentation 0.0.0 → 5.51.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +177 -43
- package/dist/generic/AdaptiveBoundarySegmenter.d.ts +98 -0
- package/dist/generic/AdaptiveBoundarySegmenter.d.ts.map +1 -0
- package/dist/generic/AdaptiveBoundarySegmenter.js +177 -0
- package/dist/generic/AdaptiveBoundarySegmenter.js.map +1 -0
- package/dist/generic/BaseContentCleaner.d.ts +101 -0
- package/dist/generic/BaseContentCleaner.d.ts.map +1 -0
- package/dist/generic/BaseContentCleaner.js +114 -0
- package/dist/generic/BaseContentCleaner.js.map +1 -0
- package/dist/generic/BaseSegmenter.d.ts +106 -0
- package/dist/generic/BaseSegmenter.d.ts.map +1 -0
- package/dist/generic/BaseSegmenter.js +260 -0
- package/dist/generic/BaseSegmenter.js.map +1 -0
- package/dist/generic/FixedWindowSegmenter.d.ts +49 -0
- package/dist/generic/FixedWindowSegmenter.d.ts.map +1 -0
- package/dist/generic/FixedWindowSegmenter.js +106 -0
- package/dist/generic/FixedWindowSegmenter.js.map +1 -0
- package/dist/generic/HtmlContentCleaner.d.ts +61 -0
- package/dist/generic/HtmlContentCleaner.d.ts.map +1 -0
- package/dist/generic/HtmlContentCleaner.js +130 -0
- package/dist/generic/HtmlContentCleaner.js.map +1 -0
- package/dist/generic/PagedContentSegmenter.d.ts +55 -0
- package/dist/generic/PagedContentSegmenter.d.ts.map +1 -0
- package/dist/generic/PagedContentSegmenter.js +99 -0
- package/dist/generic/PagedContentSegmenter.js.map +1 -0
- package/dist/generic/PlainTextContentCleaner.d.ts +22 -0
- package/dist/generic/PlainTextContentCleaner.d.ts.map +1 -0
- package/dist/generic/PlainTextContentCleaner.js +37 -0
- package/dist/generic/PlainTextContentCleaner.js.map +1 -0
- package/dist/generic/Segmentation.types.d.ts +198 -0
- package/dist/generic/Segmentation.types.d.ts.map +1 -0
- package/dist/generic/Segmentation.types.js +19 -0
- package/dist/generic/Segmentation.types.js.map +1 -0
- package/dist/generic/SegmentationResolver.d.ts +43 -0
- package/dist/generic/SegmentationResolver.d.ts.map +1 -0
- package/dist/generic/SegmentationResolver.js +83 -0
- package/dist/generic/SegmentationResolver.js.map +1 -0
- package/dist/generic/SemanticTextSegmenter.d.ts +80 -0
- package/dist/generic/SemanticTextSegmenter.d.ts.map +1 -0
- package/dist/generic/SemanticTextSegmenter.js +201 -0
- package/dist/generic/SemanticTextSegmenter.js.map +1 -0
- package/dist/generic/StructuralTextSegmenter.d.ts +63 -0
- package/dist/generic/StructuralTextSegmenter.d.ts.map +1 -0
- package/dist/generic/StructuralTextSegmenter.js +177 -0
- package/dist/generic/StructuralTextSegmenter.js.map +1 -0
- package/dist/generic/TranscriptSegmenter.d.ts +78 -0
- package/dist/generic/TranscriptSegmenter.d.ts.map +1 -0
- package/dist/generic/TranscriptSegmenter.js +194 -0
- package/dist/generic/TranscriptSegmenter.js.map +1 -0
- package/dist/index.d.ts +33 -0
- package/dist/index.d.ts.map +1 -0
- package/dist/index.js +36 -0
- package/dist/index.js.map +1 -0
- package/package.json +33 -7
package/README.md
CHANGED
|
@@ -1,45 +1,179 @@
|
|
|
1
1
|
# @memberjunction/ai-segmentation
|
|
2
2
|
|
|
3
|
-
|
|
4
|
-
|
|
5
|
-
|
|
6
|
-
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
-
|
|
33
|
-
-
|
|
34
|
-
-
|
|
35
|
-
-
|
|
36
|
-
|
|
37
|
-
##
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
3
|
+
Pluggable **content segmentation** for MemberJunction's RAG ingestion — the step that decides *what*
|
|
4
|
+
gets embedded, before an embedding model decides *how*.
|
|
5
|
+
|
|
6
|
+
## Why this package exists
|
|
7
|
+
|
|
8
|
+
Retrieval quality is capped by segmentation quality. An embedding is only as good as the span of
|
|
9
|
+
content it represents: split a document on an arbitrary token boundary and you get vectors that
|
|
10
|
+
straddle two topics and match neither query well. Feed a 60-minute recording to a multimodal model in
|
|
11
|
+
one call and you get a mush vector, because those models sample a bounded window regardless of clip
|
|
12
|
+
length.
|
|
13
|
+
|
|
14
|
+
Before this package, chunking was a private helper inside whichever pipeline needed it — which meant
|
|
15
|
+
every new strategy would have been another bespoke branch in someone else's engine. Segmentation is
|
|
16
|
+
now a **registered, swappable strategy**, selected the same way `BaseEmbeddings` and `VectorDBBase`
|
|
17
|
+
providers already are.
|
|
18
|
+
|
|
19
|
+
## Layering
|
|
20
|
+
|
|
21
|
+
```
|
|
22
|
+
@memberjunction/ai-vectors TextChunker — "split this string to fit a token budget"
|
|
23
|
+
▲
|
|
24
|
+
│ uses
|
|
25
|
+
@memberjunction/ai-segmentation BaseSegmenter — "what are the meaningful units of this content"
|
|
26
|
+
▲
|
|
27
|
+
│ consumed by
|
|
28
|
+
ingestion pipelines (content autotagging, vector sync, knowledge pipeline)
|
|
29
|
+
```
|
|
30
|
+
|
|
31
|
+
`TextChunker` stays a low-level string primitive in `ai-vectors`. Segmenters live one layer up,
|
|
32
|
+
where `@memberjunction/ai-prompts` is available — which is what lets `SemanticText` run a real,
|
|
33
|
+
tracked `MJ: AI Prompt Run` rather than an untracked side-channel LLM call. (`ai-vectors` cannot
|
|
34
|
+
depend on `ai-prompts`: `ai-prompts → templates → ai-provider-bundle → ai-vectors-pinecone →
|
|
35
|
+
ai-vectors` would make it circular.)
|
|
36
|
+
|
|
37
|
+
## Built-in segmenters
|
|
38
|
+
|
|
39
|
+
| Key | Class | Best for |
|
|
40
|
+
|---|---|---|
|
|
41
|
+
| `StructuralText` | `StructuralTextSegmenter` | Documents with headings — markdown, HTML, converted PDFs. **Recommended text default.** |
|
|
42
|
+
| `SemanticText` | `SemanticTextSegmenter` | Prose with no structure — transcripts, reports, long articles. Uses an LLM to find latent topic boundaries. |
|
|
43
|
+
| `Transcript` | `TranscriptSegmenter` | Audio/video with a timed transcript. Produces time-windowed **chapters** (and optional per-speaker sub-chapters). |
|
|
44
|
+
| `FixedWindow` | `FixedWindowSegmenter` | Universal fallback — token windows for text, fixed-duration windows for untranscribed media. |
|
|
45
|
+
|
|
46
|
+
The ordering in `SuggestSegmenterKey` encodes the quality hierarchy: a real transcript beats document
|
|
47
|
+
structure, which beats uniform windows.
|
|
48
|
+
|
|
49
|
+
## Usage
|
|
50
|
+
|
|
51
|
+
```typescript
|
|
52
|
+
import { ResolveSegmenter, SuggestSegmenterKey } from '@memberjunction/ai-segmentation';
|
|
53
|
+
|
|
54
|
+
const params = { Text: extractedText, MimeType: 'text/markdown', ContextUser: contextUser };
|
|
55
|
+
const segmenter = ResolveSegmenter(contentType.SegmenterKey, SuggestSegmenterKey(params));
|
|
56
|
+
|
|
57
|
+
const result = await segmenter.Segment({ ...params, Options: { MaxSegmentTokens: 512 } });
|
|
58
|
+
if (!result.Success) {
|
|
59
|
+
LogError(`Segmentation failed: ${result.ErrorMessage}`);
|
|
60
|
+
return;
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
for (const segment of result.Segments) {
|
|
64
|
+
// segment.Text -> embed into the text index
|
|
65
|
+
// segment.Media -> embed via EmbedContent into a multimodal index
|
|
66
|
+
// segment.StartMs/EndMs, StartOffset/EndOffset -> persist as chunk provenance
|
|
67
|
+
// segment.ParentSequence -> chapter / sub-chapter hierarchy
|
|
68
|
+
}
|
|
69
|
+
```
|
|
70
|
+
|
|
71
|
+
Segmenters **never throw** for content-shaped problems; they return `Success: false` with an
|
|
72
|
+
`ErrorMessage`, matching the convention used by `RunView` and `BaseEntity.Save()`.
|
|
73
|
+
|
|
74
|
+
### Bootstrap
|
|
75
|
+
|
|
76
|
+
Segmenters are resolved dynamically through the class factory, so bundlers may tree-shake them. Call
|
|
77
|
+
the load-prevention export once from your application bootstrap:
|
|
78
|
+
|
|
79
|
+
```typescript
|
|
80
|
+
import { LoadContentSegmenters } from '@memberjunction/ai-segmentation';
|
|
81
|
+
LoadContentSegmenters();
|
|
82
|
+
```
|
|
83
|
+
|
|
84
|
+
Skipping this doesn't produce an error — `ResolveSegmenter` degrades to the fallback strategy — so
|
|
85
|
+
content still gets chunked, just by the wrong strategy and silently.
|
|
86
|
+
|
|
87
|
+
### Audio/video chapters
|
|
88
|
+
|
|
89
|
+
```typescript
|
|
90
|
+
const result = await ResolveSegmenter('Transcript').Segment({
|
|
91
|
+
Media: { URL: 'https://cdn/session-1428.mp4', MimeType: 'video/mp4' },
|
|
92
|
+
Cues: timedTranscriptCues, // from ASR, or MJ realtime session capture
|
|
93
|
+
Options: { MaxChapterMs: 300_000, BoundaryGapMs: 4_000 },
|
|
94
|
+
});
|
|
95
|
+
```
|
|
96
|
+
|
|
97
|
+
Each chapter carries **both** a media reference with a `StartMs`/`EndMs` window *and* the transcript
|
|
98
|
+
text for that window. That dual payload is the point: the media reference is what a multimodal model
|
|
99
|
+
embeds for retrieval, while the transcript is what an agent can read, what a cross-encoder can rerank,
|
|
100
|
+
and what keyword search can match.
|
|
101
|
+
|
|
102
|
+
This is **one vector per chunk**, not two — the native multimodal vector — with the description and
|
|
103
|
+
transcript stored as text alongside it. A short summary plus structured fields (`SegmentTitle`,
|
|
104
|
+
`StartMs`/`EndMs`, `Speaker`, `Modality`) may be mirrored into the vector record's metadata for
|
|
105
|
+
filtering and display; a full transcript should not be, since metadata is size-capped and is
|
|
106
|
+
filter/payload rather than ranked full-text.
|
|
107
|
+
|
|
108
|
+
### Making media findable from a text query
|
|
109
|
+
|
|
110
|
+
| Path | Strong at | Weak at |
|
|
111
|
+
|---|---|---|
|
|
112
|
+
| Native text→media similarity | visual/audio semantics | runs colder than text→text |
|
|
113
|
+
| Lexical (FTS/BM25) over the description | names, acronyms, jargon, titles | paraphrase |
|
|
114
|
+
| A description *vector* (text→text) | paraphrase | rare proper nouns |
|
|
115
|
+
|
|
116
|
+
Ship the first two; add a description vector selectively, once measurement shows paraphrase misses.
|
|
117
|
+
When you do, add it as a **sibling text chunk row** pointing at the media chunk rather than a second
|
|
118
|
+
vector on the same row — that keeps one vector per row and lets retrieval collapse duplicates on the
|
|
119
|
+
parent key. For speech-dominant archives, also price transcript-only ingestion first: text chapters
|
|
120
|
+
alone give full semantic search at zero multimodal embedding spend.
|
|
121
|
+
|
|
122
|
+
See the [Content Segmentation Guide](../../../guides/CONTENT_SEGMENTATION_GUIDE.md) for the full
|
|
123
|
+
rationale and the cross-package picture.
|
|
124
|
+
|
|
125
|
+
## Adding a new strategy
|
|
126
|
+
|
|
127
|
+
Implement one method and register the class:
|
|
128
|
+
|
|
129
|
+
```typescript
|
|
130
|
+
@RegisterClass(BaseSegmenter, 'MyStrategy')
|
|
131
|
+
export class MySegmenter extends BaseSegmenter {
|
|
132
|
+
public get Key(): string { return 'MyStrategy'; }
|
|
133
|
+
public get SupportedModalities(): ContentModality[] { return ['text']; }
|
|
134
|
+
|
|
135
|
+
protected async SegmentCore(params: SegmentationParams): Promise<RawSegment[]> {
|
|
136
|
+
return findBoundaries(params.Text ?? '').map(b => ({
|
|
137
|
+
Modality: 'text', Text: b.Text, StartOffset: b.Start, EndOffset: b.End,
|
|
138
|
+
}));
|
|
139
|
+
}
|
|
140
|
+
}
|
|
141
|
+
```
|
|
142
|
+
|
|
143
|
+
`BaseSegmenter` handles everything else: input validation, enforcing the token ceiling (splitting
|
|
144
|
+
oversized segments while preserving titles and rebasing offsets), merging undersized segments,
|
|
145
|
+
sequence numbering, remapping `ParentIndex` → `ParentSequence` after splits, cycle-safe depth
|
|
146
|
+
calculation, and provenance stamping. Subclasses reference sibling segments by their index in the
|
|
147
|
+
array they just returned and never reason about post-split numbering.
|
|
148
|
+
|
|
149
|
+
## Options
|
|
150
|
+
|
|
151
|
+
Common to every segmenter (`SegmentationOptions`):
|
|
152
|
+
|
|
153
|
+
| Option | Default | Purpose |
|
|
154
|
+
|---|---|---|
|
|
155
|
+
| `MaxSegmentTokens` | 512 | Hard ceiling; the base class splits anything larger. |
|
|
156
|
+
| `OverlapTokens` | 10% of max | Overlap applied when an oversized segment is split. |
|
|
157
|
+
| `MinSegmentTokens` | 0 (off) | Merge adjacent text segments below this size. |
|
|
158
|
+
|
|
159
|
+
Each segmenter extends these with its own strongly-typed options — see
|
|
160
|
+
`StructuralTextSegmentationOptions`, `SemanticTextSegmentationOptions`,
|
|
161
|
+
`TranscriptSegmentationOptions`, and `FixedWindowSegmentationOptions`.
|
|
162
|
+
|
|
163
|
+
## Cost posture
|
|
164
|
+
|
|
165
|
+
Segmentation runs on every ingested item, so the defaults are deliberately cheap:
|
|
166
|
+
|
|
167
|
+
- `StructuralText` and `FixedWindow` make **no** model calls.
|
|
168
|
+
- `SemanticText` skips the LLM entirely for documents under `MinTokensForLLM` (default 750 tokens),
|
|
169
|
+
truncates each block to a preview before prompting, asks the model for **block indices** rather than
|
|
170
|
+
character offsets (models are unreliable at arithmetic over long strings, and a bad offset would
|
|
171
|
+
silently corrupt chunk provenance), and degrades to `StructuralText` on any failure.
|
|
172
|
+
- `Transcript` emits sub-chapters only when `EmitSubChapters` is enabled, since they double the
|
|
173
|
+
embedding count for a chapter.
|
|
174
|
+
|
|
175
|
+
## Testing
|
|
176
|
+
|
|
177
|
+
```bash
|
|
178
|
+
cd packages/AI/Segmentation && npm run test
|
|
179
|
+
```
|
|
@@ -0,0 +1,98 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview Target-size segmenter with an escalating boundary preference.
|
|
3
|
+
*
|
|
4
|
+
* @module @memberjunction/ai-segmentation
|
|
5
|
+
*/
|
|
6
|
+
import { BaseSegmenter } from './BaseSegmenter.js';
|
|
7
|
+
import { ContentModality, RawSegment, SegmentationOptions, SegmentationParams } from './Segmentation.types.js';
|
|
8
|
+
/** Registration key for {@link AdaptiveBoundarySegmenter}. */
|
|
9
|
+
export declare const ADAPTIVE_BOUNDARY_SEGMENTER_KEY = "AdaptiveBoundary";
|
|
10
|
+
/** Options specific to {@link AdaptiveBoundarySegmenter}. */
|
|
11
|
+
export interface AdaptiveBoundarySegmentationOptions extends SegmentationOptions {
|
|
12
|
+
/**
|
|
13
|
+
* Desired segment size in tokens.
|
|
14
|
+
*
|
|
15
|
+
* **Size this to your queries, not to your embedding model.** The model's context
|
|
16
|
+
* window is an upper bound, not a target — a chunk should be about as much content as
|
|
17
|
+
* a good answer to a typical query, so that a matching chunk is mostly signal. If
|
|
18
|
+
* queries are short paraphrases, smaller chunks retrieve better; if downstream
|
|
19
|
+
* summarization wants context, larger ones do. Default: 512.
|
|
20
|
+
*/
|
|
21
|
+
TargetTokens?: number;
|
|
22
|
+
/**
|
|
23
|
+
* How far *below* target (percent) the segmenter may close on a good boundary.
|
|
24
|
+
* Entering this band is what makes segment sizes vary in service of clean breaks.
|
|
25
|
+
* Default: 20.
|
|
26
|
+
*/
|
|
27
|
+
UndershootPercent?: number;
|
|
28
|
+
/**
|
|
29
|
+
* How far *above* target (percent) it keeps looking for a sentence or word boundary
|
|
30
|
+
* before giving up and cutting at the hard ceiling. Default: 20.
|
|
31
|
+
*/
|
|
32
|
+
OvershootPercent?: number;
|
|
33
|
+
/**
|
|
34
|
+
* If the whole text is within this percent above target, emit it as ONE segment
|
|
35
|
+
* rather than splitting it into a large piece plus a small remainder. Default: 40.
|
|
36
|
+
*/
|
|
37
|
+
NoSplitPercent?: number;
|
|
38
|
+
}
|
|
39
|
+
/**
|
|
40
|
+
* Splits text toward a **target** size, closing on the best available natural boundary
|
|
41
|
+
* near that target rather than cutting at a fixed offset.
|
|
42
|
+
*
|
|
43
|
+
* ## Why this beats a fixed window
|
|
44
|
+
*
|
|
45
|
+
* A fixed window cuts wherever the budget runs out, which routinely lands mid-paragraph
|
|
46
|
+
* — the chunk then straddles two ideas and matches neither query well. This segmenter
|
|
47
|
+
* treats the target as a goal with a tolerance band and escalates through boundary
|
|
48
|
+
* quality as it goes:
|
|
49
|
+
*
|
|
50
|
+
* 1. Once within `UndershootPercent` of target, close on a **paragraph** break.
|
|
51
|
+
* 2. Past target, accept a **sentence** break.
|
|
52
|
+
* 3. Past `OvershootPercent`, accept a **word** break.
|
|
53
|
+
* 4. Failing all of those, cut at the hard `MaxSegmentTokens` ceiling.
|
|
54
|
+
*
|
|
55
|
+
* Segment sizes therefore vary — deliberately. A slightly short segment that ends at a
|
|
56
|
+
* paragraph is worth more at retrieval time than an exactly-sized one that ends mid-clause.
|
|
57
|
+
*
|
|
58
|
+
* It also declines to split at all when the whole text is only modestly over target
|
|
59
|
+
* (`NoSplitPercent`), which avoids the common pathology of one full-size chunk followed by
|
|
60
|
+
* a runt carrying two sentences and no context.
|
|
61
|
+
*
|
|
62
|
+
* This is the recommended default for prose when document structure isn't available;
|
|
63
|
+
* prefer `StructuralText` when the content has headings, since an authored boundary beats
|
|
64
|
+
* an inferred one.
|
|
65
|
+
*/
|
|
66
|
+
export declare class AdaptiveBoundarySegmenter extends BaseSegmenter {
|
|
67
|
+
get Key(): string;
|
|
68
|
+
get SupportedModalities(): ContentModality[];
|
|
69
|
+
protected SegmentCore(params: SegmentationParams<AdaptiveBoundarySegmentationOptions>): RawSegment[];
|
|
70
|
+
/** True when the whole text is close enough to target that splitting would only make a runt. */
|
|
71
|
+
private fitsWithoutSplitting;
|
|
72
|
+
/** Walk the text emitting one segment per located boundary. */
|
|
73
|
+
private walk;
|
|
74
|
+
/**
|
|
75
|
+
* Locate this segment's end by escalating through boundary quality.
|
|
76
|
+
* Ranges are half-open [from, to).
|
|
77
|
+
*/
|
|
78
|
+
private findBoundary;
|
|
79
|
+
/**
|
|
80
|
+
* End offset of the last match of `pattern` starting within [from, to), or null.
|
|
81
|
+
* The returned offset is the END of the match, so the delimiter stays with the
|
|
82
|
+
* segment it terminates rather than opening the next one.
|
|
83
|
+
*/
|
|
84
|
+
private lastMatch;
|
|
85
|
+
/** Merge caller options with adaptive-specific defaults. */
|
|
86
|
+
private resolveAdaptiveOptions;
|
|
87
|
+
/** Keep a percentage option inside a sane range. */
|
|
88
|
+
private clampPercent;
|
|
89
|
+
/** Target size expressed in characters (TextChunker's ~4 chars/token estimate). */
|
|
90
|
+
private targetChars;
|
|
91
|
+
/** Hard ceiling expressed in characters. */
|
|
92
|
+
private hardChars;
|
|
93
|
+
/** Overlap in characters, capped at half the target so segments always advance. */
|
|
94
|
+
private overlapChars;
|
|
95
|
+
/** Estimated tokens for a string — exposed for callers reasoning about sizing. */
|
|
96
|
+
EstimateTokens(text: string): number;
|
|
97
|
+
}
|
|
98
|
+
//# sourceMappingURL=AdaptiveBoundarySegmenter.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"AdaptiveBoundarySegmenter.d.ts","sourceRoot":"","sources":["../../src/generic/AdaptiveBoundarySegmenter.ts"],"names":[],"mappings":"AAAA;;;;GAIG;AAIH,OAAO,EAAE,aAAa,EAAE,MAAM,iBAAiB,CAAC;AAChD,OAAO,EAAE,eAAe,EAAE,UAAU,EAAE,mBAAmB,EAAE,kBAAkB,EAAE,MAAM,sBAAsB,CAAC;AAE5G,8DAA8D;AAC9D,eAAO,MAAM,+BAA+B,qBAAqB,CAAC;AAElE,6DAA6D;AAC7D,MAAM,WAAW,mCAAoC,SAAQ,mBAAmB;IAC5E;;;;;;;;OAQG;IACH,YAAY,CAAC,EAAE,MAAM,CAAC;IACtB;;;;OAIG;IACH,iBAAiB,CAAC,EAAE,MAAM,CAAC;IAC3B;;;OAGG;IACH,gBAAgB,CAAC,EAAE,MAAM,CAAC;IAC1B;;;OAGG;IACH,cAAc,CAAC,EAAE,MAAM,CAAC;CAC3B;AAUD;;;;;;;;;;;;;;;;;;;;;;;;;;GA0BG;AACH,qBACa,yBAA0B,SAAQ,aAAa;IACxD,IAAW,GAAG,IAAI,MAAM,CAEvB;IAED,IAAW,mBAAmB,IAAI,eAAe,EAAE,CAElD;IAED,SAAS,CAAC,WAAW,CAAC,MAAM,EAAE,kBAAkB,CAAC,mCAAmC,CAAC,GAAG,UAAU,EAAE;IAYpG,gGAAgG;IAChG,OAAO,CAAC,oBAAoB;IAK5B,+DAA+D;IAC/D,OAAO,CAAC,IAAI;IAoBZ;;;OAGG;IACH,OAAO,CAAC,YAAY;IA6BpB;;;;OAIG;IACH,OAAO,CAAC,SAAS;IAmBjB,4DAA4D;IAC5D,OAAO,CAAC,sBAAsB;IAiB9B,oDAAoD;IACpD,OAAO,CAAC,YAAY;IAOpB,mFAAmF;IACnF,OAAO,CAAC,WAAW;IAInB,4CAA4C;IAC5C,OAAO,CAAC,SAAS;IAIjB,mFAAmF;IACnF,OAAO,CAAC,YAAY;IAIpB,kFAAkF;IAC3E,cAAc,CAAC,IAAI,EAAE,MAAM,GAAG,MAAM;CAG9C"}
|
|
@@ -0,0 +1,177 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview Target-size segmenter with an escalating boundary preference.
|
|
3
|
+
*
|
|
4
|
+
* @module @memberjunction/ai-segmentation
|
|
5
|
+
*/
|
|
6
|
+
var __decorate = (this && this.__decorate) || function (decorators, target, key, desc) {
|
|
7
|
+
var c = arguments.length, r = c < 3 ? target : desc === null ? desc = Object.getOwnPropertyDescriptor(target, key) : desc, d;
|
|
8
|
+
if (typeof Reflect === "object" && typeof Reflect.decorate === "function") r = Reflect.decorate(decorators, target, key, desc);
|
|
9
|
+
else for (var i = decorators.length - 1; i >= 0; i--) if (d = decorators[i]) r = (c < 3 ? d(r) : c > 3 ? d(target, key, r) : d(target, key)) || r;
|
|
10
|
+
return c > 3 && r && Object.defineProperty(target, key, r), r;
|
|
11
|
+
};
|
|
12
|
+
import { RegisterClass } from '@memberjunction/global';
|
|
13
|
+
import { TextChunker } from '@memberjunction/ai-vectors';
|
|
14
|
+
import { BaseSegmenter } from './BaseSegmenter.js';
|
|
15
|
+
/** Registration key for {@link AdaptiveBoundarySegmenter}. */
|
|
16
|
+
export const ADAPTIVE_BOUNDARY_SEGMENTER_KEY = 'AdaptiveBoundary';
|
|
17
|
+
/**
|
|
18
|
+
* Splits text toward a **target** size, closing on the best available natural boundary
|
|
19
|
+
* near that target rather than cutting at a fixed offset.
|
|
20
|
+
*
|
|
21
|
+
* ## Why this beats a fixed window
|
|
22
|
+
*
|
|
23
|
+
* A fixed window cuts wherever the budget runs out, which routinely lands mid-paragraph
|
|
24
|
+
* — the chunk then straddles two ideas and matches neither query well. This segmenter
|
|
25
|
+
* treats the target as a goal with a tolerance band and escalates through boundary
|
|
26
|
+
* quality as it goes:
|
|
27
|
+
*
|
|
28
|
+
* 1. Once within `UndershootPercent` of target, close on a **paragraph** break.
|
|
29
|
+
* 2. Past target, accept a **sentence** break.
|
|
30
|
+
* 3. Past `OvershootPercent`, accept a **word** break.
|
|
31
|
+
* 4. Failing all of those, cut at the hard `MaxSegmentTokens` ceiling.
|
|
32
|
+
*
|
|
33
|
+
* Segment sizes therefore vary — deliberately. A slightly short segment that ends at a
|
|
34
|
+
* paragraph is worth more at retrieval time than an exactly-sized one that ends mid-clause.
|
|
35
|
+
*
|
|
36
|
+
* It also declines to split at all when the whole text is only modestly over target
|
|
37
|
+
* (`NoSplitPercent`), which avoids the common pathology of one full-size chunk followed by
|
|
38
|
+
* a runt carrying two sentences and no context.
|
|
39
|
+
*
|
|
40
|
+
* This is the recommended default for prose when document structure isn't available;
|
|
41
|
+
* prefer `StructuralText` when the content has headings, since an authored boundary beats
|
|
42
|
+
* an inferred one.
|
|
43
|
+
*/
|
|
44
|
+
let AdaptiveBoundarySegmenter = class AdaptiveBoundarySegmenter extends BaseSegmenter {
|
|
45
|
+
get Key() {
|
|
46
|
+
return ADAPTIVE_BOUNDARY_SEGMENTER_KEY;
|
|
47
|
+
}
|
|
48
|
+
get SupportedModalities() {
|
|
49
|
+
return ['text'];
|
|
50
|
+
}
|
|
51
|
+
SegmentCore(params) {
|
|
52
|
+
const text = params.Text ?? '';
|
|
53
|
+
if (text.trim().length === 0) {
|
|
54
|
+
return [];
|
|
55
|
+
}
|
|
56
|
+
const settings = this.resolveAdaptiveOptions(params.Options);
|
|
57
|
+
if (this.fitsWithoutSplitting(text, settings)) {
|
|
58
|
+
return [{ Modality: 'text', Text: text.trim(), StartOffset: 0, EndOffset: text.length }];
|
|
59
|
+
}
|
|
60
|
+
return this.walk(text, settings);
|
|
61
|
+
}
|
|
62
|
+
/** True when the whole text is close enough to target that splitting would only make a runt. */
|
|
63
|
+
fitsWithoutSplitting(text, settings) {
|
|
64
|
+
const limit = this.targetChars(settings) * (1 + settings.NoSplitPercent / 100);
|
|
65
|
+
return text.length <= limit;
|
|
66
|
+
}
|
|
67
|
+
/** Walk the text emitting one segment per located boundary. */
|
|
68
|
+
walk(text, settings) {
|
|
69
|
+
const segments = [];
|
|
70
|
+
const overlapChars = this.overlapChars(settings);
|
|
71
|
+
let cursor = 0;
|
|
72
|
+
while (cursor < text.length) {
|
|
73
|
+
const boundary = this.findBoundary(text, cursor, settings);
|
|
74
|
+
const body = text.slice(cursor, boundary.End).trim();
|
|
75
|
+
if (body.length > 0) {
|
|
76
|
+
segments.push({ Modality: 'text', Text: body, StartOffset: cursor, EndOffset: boundary.End });
|
|
77
|
+
}
|
|
78
|
+
if (boundary.End >= text.length) {
|
|
79
|
+
break;
|
|
80
|
+
}
|
|
81
|
+
// Always advance, even when the overlap would otherwise stall the cursor.
|
|
82
|
+
cursor = Math.max(boundary.End - overlapChars, cursor + 1);
|
|
83
|
+
}
|
|
84
|
+
return segments;
|
|
85
|
+
}
|
|
86
|
+
/**
|
|
87
|
+
* Locate this segment's end by escalating through boundary quality.
|
|
88
|
+
* Ranges are half-open [from, to).
|
|
89
|
+
*/
|
|
90
|
+
findBoundary(text, cursor, settings) {
|
|
91
|
+
const target = this.targetChars(settings);
|
|
92
|
+
const softMin = cursor + Math.floor(target * (1 - settings.UndershootPercent / 100));
|
|
93
|
+
const softMax = cursor + Math.floor(target * (1 + settings.OvershootPercent / 100));
|
|
94
|
+
const hardMax = cursor + this.hardChars(settings);
|
|
95
|
+
if (text.length <= softMax) {
|
|
96
|
+
return { End: text.length, Kind: 'end' };
|
|
97
|
+
}
|
|
98
|
+
const paragraph = this.lastMatch(text, /\n\s*\n/g, softMin, softMax);
|
|
99
|
+
if (paragraph !== null) {
|
|
100
|
+
return { End: paragraph, Kind: 'paragraph' };
|
|
101
|
+
}
|
|
102
|
+
const sentence = this.lastMatch(text, /[.!?]["')\]]?\s/g, softMin, softMax);
|
|
103
|
+
if (sentence !== null) {
|
|
104
|
+
return { End: sentence, Kind: 'sentence' };
|
|
105
|
+
}
|
|
106
|
+
const word = this.lastMatch(text, /\s+/g, softMin, Math.min(hardMax, text.length));
|
|
107
|
+
if (word !== null) {
|
|
108
|
+
return { End: word, Kind: 'word' };
|
|
109
|
+
}
|
|
110
|
+
return { End: Math.min(hardMax, text.length), Kind: 'hard' };
|
|
111
|
+
}
|
|
112
|
+
/**
|
|
113
|
+
* End offset of the last match of `pattern` starting within [from, to), or null.
|
|
114
|
+
* The returned offset is the END of the match, so the delimiter stays with the
|
|
115
|
+
* segment it terminates rather than opening the next one.
|
|
116
|
+
*/
|
|
117
|
+
lastMatch(text, pattern, from, to) {
|
|
118
|
+
if (to <= from) {
|
|
119
|
+
return null;
|
|
120
|
+
}
|
|
121
|
+
const scan = new RegExp(pattern.source, 'g');
|
|
122
|
+
scan.lastIndex = Math.max(from, 0);
|
|
123
|
+
let found = null;
|
|
124
|
+
let match = scan.exec(text);
|
|
125
|
+
while (match !== null && match.index < to) {
|
|
126
|
+
found = match.index + match[0].length;
|
|
127
|
+
match = scan.exec(text);
|
|
128
|
+
}
|
|
129
|
+
return found;
|
|
130
|
+
}
|
|
131
|
+
// ─────────────────────────────────────────────
|
|
132
|
+
// Settings
|
|
133
|
+
// ─────────────────────────────────────────────
|
|
134
|
+
/** Merge caller options with adaptive-specific defaults. */
|
|
135
|
+
resolveAdaptiveOptions(options) {
|
|
136
|
+
const base = this.resolveOptions(options);
|
|
137
|
+
const target = Math.max(options?.TargetTokens ?? base.MaxSegmentTokens, 1);
|
|
138
|
+
return {
|
|
139
|
+
...base,
|
|
140
|
+
// The hard ceiling can never sit below the target, or every segment would be
|
|
141
|
+
// cut by the ceiling before a boundary was ever considered.
|
|
142
|
+
MaxSegmentTokens: Math.max(base.MaxSegmentTokens, target),
|
|
143
|
+
TargetTokens: target,
|
|
144
|
+
UndershootPercent: this.clampPercent(options?.UndershootPercent ?? 20, 0, 90),
|
|
145
|
+
OvershootPercent: this.clampPercent(options?.OvershootPercent ?? 20, 0, 200),
|
|
146
|
+
NoSplitPercent: this.clampPercent(options?.NoSplitPercent ?? 40, 0, 500),
|
|
147
|
+
};
|
|
148
|
+
}
|
|
149
|
+
/** Keep a percentage option inside a sane range. */
|
|
150
|
+
clampPercent(value, min, max) {
|
|
151
|
+
if (!Number.isFinite(value)) {
|
|
152
|
+
return min;
|
|
153
|
+
}
|
|
154
|
+
return Math.min(Math.max(value, min), max);
|
|
155
|
+
}
|
|
156
|
+
/** Target size expressed in characters (TextChunker's ~4 chars/token estimate). */
|
|
157
|
+
targetChars(settings) {
|
|
158
|
+
return settings.TargetTokens * 4;
|
|
159
|
+
}
|
|
160
|
+
/** Hard ceiling expressed in characters. */
|
|
161
|
+
hardChars(settings) {
|
|
162
|
+
return settings.MaxSegmentTokens * 4;
|
|
163
|
+
}
|
|
164
|
+
/** Overlap in characters, capped at half the target so segments always advance. */
|
|
165
|
+
overlapChars(settings) {
|
|
166
|
+
return Math.min(settings.OverlapTokens * 4, Math.floor(this.targetChars(settings) / 2));
|
|
167
|
+
}
|
|
168
|
+
/** Estimated tokens for a string — exposed for callers reasoning about sizing. */
|
|
169
|
+
EstimateTokens(text) {
|
|
170
|
+
return TextChunker.EstimateTokenCount(text);
|
|
171
|
+
}
|
|
172
|
+
};
|
|
173
|
+
AdaptiveBoundarySegmenter = __decorate([
|
|
174
|
+
RegisterClass(BaseSegmenter, ADAPTIVE_BOUNDARY_SEGMENTER_KEY)
|
|
175
|
+
], AdaptiveBoundarySegmenter);
|
|
176
|
+
export { AdaptiveBoundarySegmenter };
|
|
177
|
+
//# sourceMappingURL=AdaptiveBoundarySegmenter.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"AdaptiveBoundarySegmenter.js","sourceRoot":"","sources":["../../src/generic/AdaptiveBoundarySegmenter.ts"],"names":[],"mappings":"AAAA;;;;GAIG;;;;;;;AAEH,OAAO,EAAE,aAAa,EAAE,MAAM,wBAAwB,CAAC;AACvD,OAAO,EAAE,WAAW,EAAE,MAAM,4BAA4B,CAAC;AACzD,OAAO,EAAE,aAAa,EAAE,MAAM,iBAAiB,CAAC;AAGhD,8DAA8D;AAC9D,MAAM,CAAC,MAAM,+BAA+B,GAAG,kBAAkB,CAAC;AAwClE;;;;;;;;;;;;;;;;;;;;;;;;;;GA0BG;AAEI,IAAM,yBAAyB,GAA/B,MAAM,yBAA0B,SAAQ,aAAa;IACxD,IAAW,GAAG;QACV,OAAO,+BAA+B,CAAC;IAC3C,CAAC;IAED,IAAW,mBAAmB;QAC1B,OAAO,CAAC,MAAM,CAAC,CAAC;IACpB,CAAC;IAES,WAAW,CAAC,MAA+D;QACjF,MAAM,IAAI,GAAG,MAAM,CAAC,IAAI,IAAI,EAAE,CAAC;QAC/B,IAAI,IAAI,CAAC,IAAI,EAAE,CAAC,MAAM,KAAK,CAAC,EAAE,CAAC;YAC3B,OAAO,EAAE,CAAC;QACd,CAAC;QACD,MAAM,QAAQ,GAAG,IAAI,CAAC,sBAAsB,CAAC,MAAM,CAAC,OAAO,CAAC,CAAC;QAC7D,IAAI,IAAI,CAAC,oBAAoB,CAAC,IAAI,EAAE,QAAQ,CAAC,EAAE,CAAC;YAC5C,OAAO,CAAC,EAAE,QAAQ,EAAE,MAAM,EAAE,IAAI,EAAE,IAAI,CAAC,IAAI,EAAE,EAAE,WAAW,EAAE,CAAC,EAAE,SAAS,EAAE,IAAI,CAAC,MAAM,EAAE,CAAC,CAAC;QAC7F,CAAC;QACD,OAAO,IAAI,CAAC,IAAI,CAAC,IAAI,EAAE,QAAQ,CAAC,CAAC;IACrC,CAAC;IAED,gGAAgG;IACxF,oBAAoB,CAAC,IAAY,EAAE,QAAuD;QAC9F,MAAM,KAAK,GAAG,IAAI,CAAC,WAAW,CAAC,QAAQ,CAAC,GAAG,CAAC,CAAC,GAAG,QAAQ,CAAC,cAAc,GAAG,GAAG,CAAC,CAAC;QAC/E,OAAO,IAAI,CAAC,MAAM,IAAI,KAAK,CAAC;IAChC,CAAC;IAED,+DAA+D;IACvD,IAAI,CAAC,IAAY,EAAE,QAAuD;QAC9E,MAAM,QAAQ,GAAiB,EAAE,CAAC;QAClC,MAAM,YAAY,GAAG,IAAI,CAAC,YAAY,CAAC,QAAQ,CAAC,CAAC;QACjD,IAAI,MAAM,GAAG,CAAC,CAAC;QAEf,OAAO,MAAM,GAAG,IAAI,CAAC,MAAM,EAAE,CAAC;YAC1B,MAAM,QAAQ,GAAG,IAAI,CAAC,YAAY,CAAC,IAAI,EAAE,MAAM,EAAE,QAAQ,CAAC,CAAC;YAC3D,MAAM,IAAI,GAAG,IAAI,CAAC,KAAK,CAAC,MAAM,EAAE,QAAQ,CAAC,GAAG,CAAC,CAAC,IAAI,EAAE,CAAC;YACrD,IAAI,IAAI,CAAC,MAAM,GAAG,CAAC,EAAE,CAAC;gBAClB,QAAQ,CAAC,IAAI,CAAC,EAAE,QAAQ,EAAE,MAAM,EAAE,IAAI,EAAE,IAAI,EAAE,WAAW,EAAE,MAAM,EAAE,SAAS,EAAE,QAAQ,CAAC,GAAG,EAAE,CAAC,CAAC;YAClG,CAAC;YACD,IAAI,QAAQ,CAAC,GAAG,IAAI,IAAI,CAAC,MAAM,EAAE,CAAC;gBAC9B,MAAM;YACV,CAAC;YACD,0EAA0E;YAC1E,MAAM,GAAG,IAAI,CAAC,GAAG,CAAC,QAAQ,CAAC,GAAG,GAAG,YAAY,EAAE,MAAM,GAAG,CAAC,CAAC,CAAC;QAC/D,CAAC;QACD,OAAO,QAAQ,CAAC;IACpB,CAAC;IAED;;;OAGG;IACK,YAAY,CAChB,IAAY,EACZ,MAAc,EACd,QAAuD;QAEvD,MAAM,MAAM,GAAG,IAAI,CAAC,WAAW,CAAC,QAAQ,CAAC,CAAC;QAC1C,MAAM,OAAO,GAAG,MAAM,GAAG,IAAI,CAAC,KAAK,CAAC,MAAM,GAAG,CAAC,CAAC,GAAG,QAAQ,CAAC,iBAAiB,GAAG,GAAG,CAAC,CAAC,CAAC;QACrF,MAAM,OAAO,GAAG,MAAM,GAAG,IAAI,CAAC,KAAK,CAAC,MAAM,GAAG,CAAC,CAAC,GAAG,QAAQ,CAAC,gBAAgB,GAAG,GAAG,CAAC,CAAC,CAAC;QACpF,MAAM,OAAO,GAAG,MAAM,GAAG,IAAI,CAAC,SAAS,CAAC,QAAQ,CAAC,CAAC;QAElD,IAAI,IAAI,CAAC,MAAM,IAAI,OAAO,EAAE,CAAC;YACzB,OAAO,EAAE,GAAG,EAAE,IAAI,CAAC,MAAM,EAAE,IAAI,EAAE,KAAK,EAAE,CAAC;QAC7C,CAAC;QAED,MAAM,SAAS,GAAG,IAAI,CAAC,SAAS,CAAC,IAAI,EAAE,UAAU,EAAE,OAAO,EAAE,OAAO,CAAC,CAAC;QACrE,IAAI,SAAS,KAAK,IAAI,EAAE,CAAC;YACrB,OAAO,EAAE,GAAG,EAAE,SAAS,EAAE,IAAI,EAAE,WAAW,EAAE,CAAC;QACjD,CAAC;QACD,MAAM,QAAQ,GAAG,IAAI,CAAC,SAAS,CAAC,IAAI,EAAE,kBAAkB,EAAE,OAAO,EAAE,OAAO,CAAC,CAAC;QAC5E,IAAI,QAAQ,KAAK,IAAI,EAAE,CAAC;YACpB,OAAO,EAAE,GAAG,EAAE,QAAQ,EAAE,IAAI,EAAE,UAAU,EAAE,CAAC;QAC/C,CAAC;QACD,MAAM,IAAI,GAAG,IAAI,CAAC,SAAS,CAAC,IAAI,EAAE,MAAM,EAAE,OAAO,EAAE,IAAI,CAAC,GAAG,CAAC,OAAO,EAAE,IAAI,CAAC,MAAM,CAAC,CAAC,CAAC;QACnF,IAAI,IAAI,KAAK,IAAI,EAAE,CAAC;YAChB,OAAO,EAAE,GAAG,EAAE,IAAI,EAAE,IAAI,EAAE,MAAM,EAAE,CAAC;QACvC,CAAC;QACD,OAAO,EAAE,GAAG,EAAE,IAAI,CAAC,GAAG,CAAC,OAAO,EAAE,IAAI,CAAC,MAAM,CAAC,EAAE,IAAI,EAAE,MAAM,EAAE,CAAC;IACjE,CAAC;IAED;;;;OAIG;IACK,SAAS,CAAC,IAAY,EAAE,OAAe,EAAE,IAAY,EAAE,EAAU;QACrE,IAAI,EAAE,IAAI,IAAI,EAAE,CAAC;YACb,OAAO,IAAI,CAAC;QAChB,CAAC;QACD,MAAM,IAAI,GAAG,IAAI,MAAM,CAAC,OAAO,CAAC,MAAM,EAAE,GAAG,CAAC,CAAC;QAC7C,IAAI,CAAC,SAAS,GAAG,IAAI,CAAC,GAAG,CAAC,IAAI,EAAE,CAAC,CAAC,CAAC;QACnC,IAAI,KAAK,GAAkB,IAAI,CAAC;QAChC,IAAI,KAAK,GAAG,IAAI,CAAC,IAAI,CAAC,IAAI,CAAC,CAAC;QAC5B,OAAO,KAAK,KAAK,IAAI,IAAI,KAAK,CAAC,KAAK,GAAG,EAAE,EAAE,CAAC;YACxC,KAAK,GAAG,KAAK,CAAC,KAAK,GAAG,KAAK,CAAC,CAAC,CAAC,CAAC,MAAM,CAAC;YACtC,KAAK,GAAG,IAAI,CAAC,IAAI,CAAC,IAAI,CAAC,CAAC;QAC5B,CAAC;QACD,OAAO,KAAK,CAAC;IACjB,CAAC;IAED,gDAAgD;IAChD,WAAW;IACX,gDAAgD;IAEhD,4DAA4D;IACpD,sBAAsB,CAC1B,OAA6C;QAE7C,MAAM,IAAI,GAAG,IAAI,CAAC,cAAc,CAAC,OAAO,CAAC,CAAC;QAC1C,MAAM,MAAM,GAAG,IAAI,CAAC,GAAG,CAAC,OAAO,EAAE,YAAY,IAAI,IAAI,CAAC,gBAAgB,EAAE,CAAC,CAAC,CAAC;QAC3E,OAAO;YACH,GAAG,IAAI;YACP,6EAA6E;YAC7E,4DAA4D;YAC5D,gBAAgB,EAAE,IAAI,CAAC,GAAG,CAAC,IAAI,CAAC,gBAAgB,EAAE,MAAM,CAAC;YACzD,YAAY,EAAE,MAAM;YACpB,iBAAiB,EAAE,IAAI,CAAC,YAAY,CAAC,OAAO,EAAE,iBAAiB,IAAI,EAAE,EAAE,CAAC,EAAE,EAAE,CAAC;YAC7E,gBAAgB,EAAE,IAAI,CAAC,YAAY,CAAC,OAAO,EAAE,gBAAgB,IAAI,EAAE,EAAE,CAAC,EAAE,GAAG,CAAC;YAC5E,cAAc,EAAE,IAAI,CAAC,YAAY,CAAC,OAAO,EAAE,cAAc,IAAI,EAAE,EAAE,CAAC,EAAE,GAAG,CAAC;SAC3E,CAAC;IACN,CAAC;IAED,oDAAoD;IAC5C,YAAY,CAAC,KAAa,EAAE,GAAW,EAAE,GAAW;QACxD,IAAI,CAAC,MAAM,CAAC,QAAQ,CAAC,KAAK,CAAC,EAAE,CAAC;YAC1B,OAAO,GAAG,CAAC;QACf,CAAC;QACD,OAAO,IAAI,CAAC,GAAG,CAAC,IAAI,CAAC,GAAG,CAAC,KAAK,EAAE,GAAG,CAAC,EAAE,GAAG,CAAC,CAAC;IAC/C,CAAC;IAED,mFAAmF;IAC3E,WAAW,CAAC,QAAuD;QACvE,OAAO,QAAQ,CAAC,YAAY,GAAG,CAAC,CAAC;IACrC,CAAC;IAED,4CAA4C;IACpC,SAAS,CAAC,QAAuD;QACrE,OAAO,QAAQ,CAAC,gBAAgB,GAAG,CAAC,CAAC;IACzC,CAAC;IAED,mFAAmF;IAC3E,YAAY,CAAC,QAAuD;QACxE,OAAO,IAAI,CAAC,GAAG,CAAC,QAAQ,CAAC,aAAa,GAAG,CAAC,EAAE,IAAI,CAAC,KAAK,CAAC,IAAI,CAAC,WAAW,CAAC,QAAQ,CAAC,GAAG,CAAC,CAAC,CAAC,CAAC;IAC5F,CAAC;IAED,kFAAkF;IAC3E,cAAc,CAAC,IAAY;QAC9B,OAAO,WAAW,CAAC,kBAAkB,CAAC,IAAI,CAAC,CAAC;IAChD,CAAC;CACJ,CAAA;AAtJY,yBAAyB;IADrC,aAAa,CAAC,aAAa,EAAE,+BAA+B,CAAC;GACjD,yBAAyB,CAsJrC"}
|
|
@@ -0,0 +1,101 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview Pluggable content cleaning — the stage that runs before segmentation.
|
|
3
|
+
*
|
|
4
|
+
* @module @memberjunction/ai-segmentation
|
|
5
|
+
*/
|
|
6
|
+
/** Options common to every cleaner. */
|
|
7
|
+
export interface ContentCleaningOptions {
|
|
8
|
+
/**
|
|
9
|
+
* CSS selectors whose content is the ONLY content to keep. When set, everything
|
|
10
|
+
* outside these selectors is discarded before any exclusion is applied.
|
|
11
|
+
*
|
|
12
|
+
* This is the highest-leverage knob for messy sources: pointing at `.article-body`
|
|
13
|
+
* removes navigation, sidebars, and advertising in one stroke, without having to
|
|
14
|
+
* enumerate every block you *don't* want.
|
|
15
|
+
*/
|
|
16
|
+
IncludeSelectors?: string[];
|
|
17
|
+
/**
|
|
18
|
+
* CSS selectors to remove. Applied after `IncludeSelectors`. Use for the blocks that
|
|
19
|
+
* survive the include (inline ad slots, share widgets, cookie banners).
|
|
20
|
+
*/
|
|
21
|
+
ExcludeSelectors?: string[];
|
|
22
|
+
/** Collapse runs of whitespace and blank lines. Default: true. */
|
|
23
|
+
NormalizeWhitespace?: boolean;
|
|
24
|
+
/** Maximum characters to retain; content beyond this is dropped. Unset = no limit. */
|
|
25
|
+
MaxLength?: number;
|
|
26
|
+
}
|
|
27
|
+
/** Input to a cleaning pass. */
|
|
28
|
+
export interface ContentCleaningParams<TOptions extends ContentCleaningOptions = ContentCleaningOptions> {
|
|
29
|
+
/** Raw content — markup or plain text. */
|
|
30
|
+
Content: string;
|
|
31
|
+
/** Mime type of `Content`, when known; lets a cleaner pick its parser. */
|
|
32
|
+
MimeType?: string;
|
|
33
|
+
/** Cleaner-specific options. */
|
|
34
|
+
Options?: TOptions;
|
|
35
|
+
}
|
|
36
|
+
/** Result of a cleaning pass. Never throws for content-shaped problems. */
|
|
37
|
+
export interface ContentCleaningResult {
|
|
38
|
+
Success: boolean;
|
|
39
|
+
/** Cleaned text. Equal to the input when `Success` is false. */
|
|
40
|
+
Content: string;
|
|
41
|
+
/** Registration key of the cleaner that ran. */
|
|
42
|
+
CleanerKey: string;
|
|
43
|
+
ErrorMessage?: string;
|
|
44
|
+
Warnings: string[];
|
|
45
|
+
/** Characters removed by cleaning — a cheap signal that a selector is wrong. */
|
|
46
|
+
CharactersRemoved: number;
|
|
47
|
+
}
|
|
48
|
+
/**
|
|
49
|
+
* Base class for content cleaning strategies.
|
|
50
|
+
*
|
|
51
|
+
* Cleaning is deliberately a **separate stage from segmentation**, and a separate
|
|
52
|
+
* plug-in point. The two answer different questions — cleaning asks *which text is
|
|
53
|
+
* actually content*, segmentation asks *where that content divides* — and they change for
|
|
54
|
+
* different reasons: a new CMS template needs new selectors, not a new chunking strategy.
|
|
55
|
+
* Splitting them also means the cleaning rules apply once and benefit every downstream
|
|
56
|
+
* consumer (embedding chunks, tagging chunks, full-text indexing) instead of being
|
|
57
|
+
* reimplemented per pipeline.
|
|
58
|
+
*
|
|
59
|
+
* Garbage that survives this stage is expensive: it gets embedded, stored, retrieved, and
|
|
60
|
+
* eventually shown to a user or an agent. Navigation chrome repeated across a thousand
|
|
61
|
+
* pages produces a thousand near-identical vectors that crowd out real answers.
|
|
62
|
+
*
|
|
63
|
+
* ```typescript
|
|
64
|
+
* @RegisterClass(BaseContentCleaner, 'MyCleaner')
|
|
65
|
+
* export class MyCleaner extends BaseContentCleaner {
|
|
66
|
+
* public get Key(): string { return 'MyCleaner'; }
|
|
67
|
+
* protected CleanCore(params: ContentCleaningParams): string { return strip(params.Content); }
|
|
68
|
+
* }
|
|
69
|
+
* ```
|
|
70
|
+
*/
|
|
71
|
+
export declare abstract class BaseContentCleaner {
|
|
72
|
+
/** Registration key; must match the key passed to `@RegisterClass`. */
|
|
73
|
+
abstract get Key(): string;
|
|
74
|
+
/** Perform the cleaning. The base class handles validation, whitespace, and truncation. */
|
|
75
|
+
protected abstract CleanCore(params: ContentCleaningParams): string;
|
|
76
|
+
/**
|
|
77
|
+
* Clean content ahead of segmentation.
|
|
78
|
+
*
|
|
79
|
+
* Never throws for content-shaped problems — inspect `Success`/`ErrorMessage`. On
|
|
80
|
+
* failure the ORIGINAL content is returned rather than an empty string, so a bad
|
|
81
|
+
* selector degrades to "not cleaned" instead of silently discarding the document.
|
|
82
|
+
*/
|
|
83
|
+
Clean(params: ContentCleaningParams): ContentCleaningResult;
|
|
84
|
+
/**
|
|
85
|
+
* Resolve a registered cleaner by key.
|
|
86
|
+
*
|
|
87
|
+
* Uses `TryCreateInstance` because `CreateInstance` never returns null for an unknown
|
|
88
|
+
* key — it silently yields a hollow base instance whose abstract members are undefined.
|
|
89
|
+
*/
|
|
90
|
+
static Resolve(key: string): BaseContentCleaner | null;
|
|
91
|
+
/** Whitespace normalization and truncation, shared by every cleaner. */
|
|
92
|
+
protected applyCommonRules(content: string, options?: ContentCleaningOptions, warnings?: string[]): string;
|
|
93
|
+
/**
|
|
94
|
+
* Collapse horizontal whitespace and runs of blank lines, while preserving the single
|
|
95
|
+
* blank line that marks a paragraph break — segmenters rely on it as a boundary signal.
|
|
96
|
+
*/
|
|
97
|
+
protected normalizeWhitespace(content: string): string;
|
|
98
|
+
/** Build a successful result with the removal delta computed. */
|
|
99
|
+
private buildResult;
|
|
100
|
+
}
|
|
101
|
+
//# sourceMappingURL=BaseContentCleaner.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"BaseContentCleaner.d.ts","sourceRoot":"","sources":["../../src/generic/BaseContentCleaner.ts"],"names":[],"mappings":"AAAA;;;;GAIG;AAKH,uCAAuC;AACvC,MAAM,WAAW,sBAAsB;IACnC;;;;;;;OAOG;IACH,gBAAgB,CAAC,EAAE,MAAM,EAAE,CAAC;IAC5B;;;OAGG;IACH,gBAAgB,CAAC,EAAE,MAAM,EAAE,CAAC;IAC5B,kEAAkE;IAClE,mBAAmB,CAAC,EAAE,OAAO,CAAC;IAC9B,sFAAsF;IACtF,SAAS,CAAC,EAAE,MAAM,CAAC;CACtB;AAED,gCAAgC;AAChC,MAAM,WAAW,qBAAqB,CAAC,QAAQ,SAAS,sBAAsB,GAAG,sBAAsB;IACnG,0CAA0C;IAC1C,OAAO,EAAE,MAAM,CAAC;IAChB,0EAA0E;IAC1E,QAAQ,CAAC,EAAE,MAAM,CAAC;IAClB,gCAAgC;IAChC,OAAO,CAAC,EAAE,QAAQ,CAAC;CACtB;AAED,2EAA2E;AAC3E,MAAM,WAAW,qBAAqB;IAClC,OAAO,EAAE,OAAO,CAAC;IACjB,gEAAgE;IAChE,OAAO,EAAE,MAAM,CAAC;IAChB,gDAAgD;IAChD,UAAU,EAAE,MAAM,CAAC;IACnB,YAAY,CAAC,EAAE,MAAM,CAAC;IACtB,QAAQ,EAAE,MAAM,EAAE,CAAC;IACnB,gFAAgF;IAChF,iBAAiB,EAAE,MAAM,CAAC;CAC7B;AAED;;;;;;;;;;;;;;;;;;;;;;GAsBG;AACH,8BAAsB,kBAAkB;IACpC,uEAAuE;IACvE,aAAoB,GAAG,IAAI,MAAM,CAAC;IAElC,2FAA2F;IAC3F,SAAS,CAAC,QAAQ,CAAC,SAAS,CAAC,MAAM,EAAE,qBAAqB,GAAG,MAAM;IAEnE;;;;;;OAMG;IACI,KAAK,CAAC,MAAM,EAAE,qBAAqB,GAAG,qBAAqB;IA6BlE;;;;;OAKG;WACW,OAAO,CAAC,GAAG,EAAE,MAAM,GAAG,kBAAkB,GAAG,IAAI;IAW7D,wEAAwE;IACxE,SAAS,CAAC,gBAAgB,CAAC,OAAO,EAAE,MAAM,EAAE,OAAO,CAAC,EAAE,sBAAsB,EAAE,QAAQ,CAAC,EAAE,MAAM,EAAE,GAAG,MAAM;IAY1G;;;OAGG;IACH,SAAS,CAAC,mBAAmB,CAAC,OAAO,EAAE,MAAM,GAAG,MAAM;IAQtD,iEAAiE;IACjE,OAAO,CAAC,WAAW;CAStB"}
|