@memberjunction/ai-segmentation 0.0.0 → 5.50.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (54) hide show
  1. package/README.md +177 -43
  2. package/dist/generic/AdaptiveBoundarySegmenter.d.ts +98 -0
  3. package/dist/generic/AdaptiveBoundarySegmenter.d.ts.map +1 -0
  4. package/dist/generic/AdaptiveBoundarySegmenter.js +177 -0
  5. package/dist/generic/AdaptiveBoundarySegmenter.js.map +1 -0
  6. package/dist/generic/BaseContentCleaner.d.ts +101 -0
  7. package/dist/generic/BaseContentCleaner.d.ts.map +1 -0
  8. package/dist/generic/BaseContentCleaner.js +114 -0
  9. package/dist/generic/BaseContentCleaner.js.map +1 -0
  10. package/dist/generic/BaseSegmenter.d.ts +106 -0
  11. package/dist/generic/BaseSegmenter.d.ts.map +1 -0
  12. package/dist/generic/BaseSegmenter.js +260 -0
  13. package/dist/generic/BaseSegmenter.js.map +1 -0
  14. package/dist/generic/FixedWindowSegmenter.d.ts +49 -0
  15. package/dist/generic/FixedWindowSegmenter.d.ts.map +1 -0
  16. package/dist/generic/FixedWindowSegmenter.js +106 -0
  17. package/dist/generic/FixedWindowSegmenter.js.map +1 -0
  18. package/dist/generic/HtmlContentCleaner.d.ts +61 -0
  19. package/dist/generic/HtmlContentCleaner.d.ts.map +1 -0
  20. package/dist/generic/HtmlContentCleaner.js +130 -0
  21. package/dist/generic/HtmlContentCleaner.js.map +1 -0
  22. package/dist/generic/PagedContentSegmenter.d.ts +55 -0
  23. package/dist/generic/PagedContentSegmenter.d.ts.map +1 -0
  24. package/dist/generic/PagedContentSegmenter.js +99 -0
  25. package/dist/generic/PagedContentSegmenter.js.map +1 -0
  26. package/dist/generic/PlainTextContentCleaner.d.ts +22 -0
  27. package/dist/generic/PlainTextContentCleaner.d.ts.map +1 -0
  28. package/dist/generic/PlainTextContentCleaner.js +37 -0
  29. package/dist/generic/PlainTextContentCleaner.js.map +1 -0
  30. package/dist/generic/Segmentation.types.d.ts +198 -0
  31. package/dist/generic/Segmentation.types.d.ts.map +1 -0
  32. package/dist/generic/Segmentation.types.js +19 -0
  33. package/dist/generic/Segmentation.types.js.map +1 -0
  34. package/dist/generic/SegmentationResolver.d.ts +43 -0
  35. package/dist/generic/SegmentationResolver.d.ts.map +1 -0
  36. package/dist/generic/SegmentationResolver.js +83 -0
  37. package/dist/generic/SegmentationResolver.js.map +1 -0
  38. package/dist/generic/SemanticTextSegmenter.d.ts +80 -0
  39. package/dist/generic/SemanticTextSegmenter.d.ts.map +1 -0
  40. package/dist/generic/SemanticTextSegmenter.js +201 -0
  41. package/dist/generic/SemanticTextSegmenter.js.map +1 -0
  42. package/dist/generic/StructuralTextSegmenter.d.ts +63 -0
  43. package/dist/generic/StructuralTextSegmenter.d.ts.map +1 -0
  44. package/dist/generic/StructuralTextSegmenter.js +177 -0
  45. package/dist/generic/StructuralTextSegmenter.js.map +1 -0
  46. package/dist/generic/TranscriptSegmenter.d.ts +78 -0
  47. package/dist/generic/TranscriptSegmenter.d.ts.map +1 -0
  48. package/dist/generic/TranscriptSegmenter.js +194 -0
  49. package/dist/generic/TranscriptSegmenter.js.map +1 -0
  50. package/dist/index.d.ts +33 -0
  51. package/dist/index.d.ts.map +1 -0
  52. package/dist/index.js +36 -0
  53. package/dist/index.js.map +1 -0
  54. package/package.json +33 -7
package/README.md CHANGED
@@ -1,45 +1,179 @@
1
1
  # @memberjunction/ai-segmentation
2
2
 
3
- ## ⚠️ IMPORTANT NOTICE ⚠️
4
-
5
- **This package is created solely for the purpose of setting up OIDC (OpenID Connect) trusted publishing with npm.**
6
-
7
- This is **NOT** a functional package and contains **NO** code or functionality beyond the OIDC setup configuration.
8
-
9
- ## Purpose
10
-
11
- This package exists to:
12
- 1. Configure OIDC trusted publishing for the package name `@memberjunction/ai-segmentation`
13
- 2. Enable secure, token-less publishing from CI/CD workflows
14
- 3. Establish provenance for packages published under this name
15
-
16
- ## What is OIDC Trusted Publishing?
17
-
18
- OIDC trusted publishing allows package maintainers to publish packages directly from their CI/CD workflows without needing to manage npm access tokens. Instead, it uses OpenID Connect to establish trust between the CI/CD provider (like GitHub Actions) and npm.
19
-
20
- ## Setup Instructions
21
-
22
- To properly configure OIDC trusted publishing for this package:
23
-
24
- 1. Go to [npmjs.com](https://www.npmjs.com/) and navigate to your package settings
25
- 2. Configure the trusted publisher (e.g., GitHub Actions)
26
- 3. Specify the repository and workflow that should be allowed to publish
27
- 4. Use the configured workflow to publish your actual package
28
-
29
- ## DO NOT USE THIS PACKAGE
30
-
31
- This package is a placeholder for OIDC configuration only. It:
32
- - Contains no executable code
33
- - Provides no functionality
34
- - Should not be installed as a dependency
35
- - Exists only for administrative purposes
36
-
37
- ## More Information
38
-
39
- For more details about npm's trusted publishing feature, see:
40
- - [npm Trusted Publishing Documentation](https://docs.npmjs.com/generating-provenance-statements)
41
- - [GitHub Actions OIDC Documentation](https://docs.github.com/en/actions/deployment/security-hardening-your-deployments/about-security-hardening-with-openid-connect)
42
-
43
- ---
44
-
45
- **Maintained for OIDC setup purposes only**
3
+ Pluggable **content segmentation** for MemberJunction's RAG ingestion — the step that decides *what*
4
+ gets embedded, before an embedding model decides *how*.
5
+
6
+ ## Why this package exists
7
+
8
+ Retrieval quality is capped by segmentation quality. An embedding is only as good as the span of
9
+ content it represents: split a document on an arbitrary token boundary and you get vectors that
10
+ straddle two topics and match neither query well. Feed a 60-minute recording to a multimodal model in
11
+ one call and you get a mush vector, because those models sample a bounded window regardless of clip
12
+ length.
13
+
14
+ Before this package, chunking was a private helper inside whichever pipeline needed it — which meant
15
+ every new strategy would have been another bespoke branch in someone else's engine. Segmentation is
16
+ now a **registered, swappable strategy**, selected the same way `BaseEmbeddings` and `VectorDBBase`
17
+ providers already are.
18
+
19
+ ## Layering
20
+
21
+ ```
22
+ @memberjunction/ai-vectors TextChunker — "split this string to fit a token budget"
23
+ ▲
24
+ │ uses
25
+ @memberjunction/ai-segmentation BaseSegmenter — "what are the meaningful units of this content"
26
+ ▲
27
+ │ consumed by
28
+ ingestion pipelines (content autotagging, vector sync, knowledge pipeline)
29
+ ```
30
+
31
+ `TextChunker` stays a low-level string primitive in `ai-vectors`. Segmenters live one layer up,
32
+ where `@memberjunction/ai-prompts` is available — which is what lets `SemanticText` run a real,
33
+ tracked `MJ: AI Prompt Run` rather than an untracked side-channel LLM call. (`ai-vectors` cannot
34
+ depend on `ai-prompts`: `ai-prompts → templates → ai-provider-bundle → ai-vectors-pinecone →
35
+ ai-vectors` would make it circular.)
36
+
37
+ ## Built-in segmenters
38
+
39
+ | Key | Class | Best for |
40
+ |---|---|---|
41
+ | `StructuralText` | `StructuralTextSegmenter` | Documents with headings — markdown, HTML, converted PDFs. **Recommended text default.** |
42
+ | `SemanticText` | `SemanticTextSegmenter` | Prose with no structure — transcripts, reports, long articles. Uses an LLM to find latent topic boundaries. |
43
+ | `Transcript` | `TranscriptSegmenter` | Audio/video with a timed transcript. Produces time-windowed **chapters** (and optional per-speaker sub-chapters). |
44
+ | `FixedWindow` | `FixedWindowSegmenter` | Universal fallback — token windows for text, fixed-duration windows for untranscribed media. |
45
+
46
+ The ordering in `SuggestSegmenterKey` encodes the quality hierarchy: a real transcript beats document
47
+ structure, which beats uniform windows.
48
+
49
+ ## Usage
50
+
51
+ ```typescript
52
+ import { ResolveSegmenter, SuggestSegmenterKey } from '@memberjunction/ai-segmentation';
53
+
54
+ const params = { Text: extractedText, MimeType: 'text/markdown', ContextUser: contextUser };
55
+ const segmenter = ResolveSegmenter(contentType.SegmenterKey, SuggestSegmenterKey(params));
56
+
57
+ const result = await segmenter.Segment({ ...params, Options: { MaxSegmentTokens: 512 } });
58
+ if (!result.Success) {
59
+ LogError(`Segmentation failed: ${result.ErrorMessage}`);
60
+ return;
61
+ }
62
+
63
+ for (const segment of result.Segments) {
64
+ // segment.Text -> embed into the text index
65
+ // segment.Media -> embed via EmbedContent into a multimodal index
66
+ // segment.StartMs/EndMs, StartOffset/EndOffset -> persist as chunk provenance
67
+ // segment.ParentSequence -> chapter / sub-chapter hierarchy
68
+ }
69
+ ```
70
+
71
+ Segmenters **never throw** for content-shaped problems; they return `Success: false` with an
72
+ `ErrorMessage`, matching the convention used by `RunView` and `BaseEntity.Save()`.
73
+
74
+ ### Bootstrap
75
+
76
+ Segmenters are resolved dynamically through the class factory, so bundlers may tree-shake them. Call
77
+ the load-prevention export once from your application bootstrap:
78
+
79
+ ```typescript
80
+ import { LoadContentSegmenters } from '@memberjunction/ai-segmentation';
81
+ LoadContentSegmenters();
82
+ ```
83
+
84
+ Skipping this doesn't produce an error — `ResolveSegmenter` degrades to the fallback strategy — so
85
+ content still gets chunked, just by the wrong strategy and silently.
86
+
87
+ ### Audio/video chapters
88
+
89
+ ```typescript
90
+ const result = await ResolveSegmenter('Transcript').Segment({
91
+ Media: { URL: 'https://cdn/session-1428.mp4', MimeType: 'video/mp4' },
92
+ Cues: timedTranscriptCues, // from ASR, or MJ realtime session capture
93
+ Options: { MaxChapterMs: 300_000, BoundaryGapMs: 4_000 },
94
+ });
95
+ ```
96
+
97
+ Each chapter carries **both** a media reference with a `StartMs`/`EndMs` window *and* the transcript
98
+ text for that window. That dual payload is the point: the media reference is what a multimodal model
99
+ embeds for retrieval, while the transcript is what an agent can read, what a cross-encoder can rerank,
100
+ and what keyword search can match.
101
+
102
+ This is **one vector per chunk**, not two — the native multimodal vector — with the description and
103
+ transcript stored as text alongside it. A short summary plus structured fields (`SegmentTitle`,
104
+ `StartMs`/`EndMs`, `Speaker`, `Modality`) may be mirrored into the vector record's metadata for
105
+ filtering and display; a full transcript should not be, since metadata is size-capped and is
106
+ filter/payload rather than ranked full-text.
107
+
108
+ ### Making media findable from a text query
109
+
110
+ | Path | Strong at | Weak at |
111
+ |---|---|---|
112
+ | Native text→media similarity | visual/audio semantics | runs colder than text→text |
113
+ | Lexical (FTS/BM25) over the description | names, acronyms, jargon, titles | paraphrase |
114
+ | A description *vector* (text→text) | paraphrase | rare proper nouns |
115
+
116
+ Ship the first two; add a description vector selectively, once measurement shows paraphrase misses.
117
+ When you do, add it as a **sibling text chunk row** pointing at the media chunk rather than a second
118
+ vector on the same row — that keeps one vector per row and lets retrieval collapse duplicates on the
119
+ parent key. For speech-dominant archives, also price transcript-only ingestion first: text chapters
120
+ alone give full semantic search at zero multimodal embedding spend.
121
+
122
+ See the [Content Segmentation Guide](../../../guides/CONTENT_SEGMENTATION_GUIDE.md) for the full
123
+ rationale and the cross-package picture.
124
+
125
+ ## Adding a new strategy
126
+
127
+ Implement one method and register the class:
128
+
129
+ ```typescript
130
+ @RegisterClass(BaseSegmenter, 'MyStrategy')
131
+ export class MySegmenter extends BaseSegmenter {
132
+ public get Key(): string { return 'MyStrategy'; }
133
+ public get SupportedModalities(): ContentModality[] { return ['text']; }
134
+
135
+ protected async SegmentCore(params: SegmentationParams): Promise<RawSegment[]> {
136
+ return findBoundaries(params.Text ?? '').map(b => ({
137
+ Modality: 'text', Text: b.Text, StartOffset: b.Start, EndOffset: b.End,
138
+ }));
139
+ }
140
+ }
141
+ ```
142
+
143
+ `BaseSegmenter` handles everything else: input validation, enforcing the token ceiling (splitting
144
+ oversized segments while preserving titles and rebasing offsets), merging undersized segments,
145
+ sequence numbering, remapping `ParentIndex` → `ParentSequence` after splits, cycle-safe depth
146
+ calculation, and provenance stamping. Subclasses reference sibling segments by their index in the
147
+ array they just returned and never reason about post-split numbering.
148
+
149
+ ## Options
150
+
151
+ Common to every segmenter (`SegmentationOptions`):
152
+
153
+ | Option | Default | Purpose |
154
+ |---|---|---|
155
+ | `MaxSegmentTokens` | 512 | Hard ceiling; the base class splits anything larger. |
156
+ | `OverlapTokens` | 10% of max | Overlap applied when an oversized segment is split. |
157
+ | `MinSegmentTokens` | 0 (off) | Merge adjacent text segments below this size. |
158
+
159
+ Each segmenter extends these with its own strongly-typed options — see
160
+ `StructuralTextSegmentationOptions`, `SemanticTextSegmentationOptions`,
161
+ `TranscriptSegmentationOptions`, and `FixedWindowSegmentationOptions`.
162
+
163
+ ## Cost posture
164
+
165
+ Segmentation runs on every ingested item, so the defaults are deliberately cheap:
166
+
167
+ - `StructuralText` and `FixedWindow` make **no** model calls.
168
+ - `SemanticText` skips the LLM entirely for documents under `MinTokensForLLM` (default 750 tokens),
169
+ truncates each block to a preview before prompting, asks the model for **block indices** rather than
170
+ character offsets (models are unreliable at arithmetic over long strings, and a bad offset would
171
+ silently corrupt chunk provenance), and degrades to `StructuralText` on any failure.
172
+ - `Transcript` emits sub-chapters only when `EmitSubChapters` is enabled, since they double the
173
+ embedding count for a chapter.
174
+
175
+ ## Testing
176
+
177
+ ```bash
178
+ cd packages/AI/Segmentation && npm run test
179
+ ```
@@ -0,0 +1,98 @@
1
+ /**
2
+ * @fileoverview Target-size segmenter with an escalating boundary preference.
3
+ *
4
+ * @module @memberjunction/ai-segmentation
5
+ */
6
+ import { BaseSegmenter } from './BaseSegmenter.js';
7
+ import { ContentModality, RawSegment, SegmentationOptions, SegmentationParams } from './Segmentation.types.js';
8
+ /** Registration key for {@link AdaptiveBoundarySegmenter}. */
9
+ export declare const ADAPTIVE_BOUNDARY_SEGMENTER_KEY = "AdaptiveBoundary";
10
+ /** Options specific to {@link AdaptiveBoundarySegmenter}. */
11
+ export interface AdaptiveBoundarySegmentationOptions extends SegmentationOptions {
12
+ /**
13
+ * Desired segment size in tokens.
14
+ *
15
+ * **Size this to your queries, not to your embedding model.** The model's context
16
+ * window is an upper bound, not a target — a chunk should be about as much content as
17
+ * a good answer to a typical query, so that a matching chunk is mostly signal. If
18
+ * queries are short paraphrases, smaller chunks retrieve better; if downstream
19
+ * summarization wants context, larger ones do. Default: 512.
20
+ */
21
+ TargetTokens?: number;
22
+ /**
23
+ * How far *below* target (percent) the segmenter may close on a good boundary.
24
+ * Entering this band is what makes segment sizes vary in service of clean breaks.
25
+ * Default: 20.
26
+ */
27
+ UndershootPercent?: number;
28
+ /**
29
+ * How far *above* target (percent) it keeps looking for a sentence or word boundary
30
+ * before giving up and cutting at the hard ceiling. Default: 20.
31
+ */
32
+ OvershootPercent?: number;
33
+ /**
34
+ * If the whole text is within this percent above target, emit it as ONE segment
35
+ * rather than splitting it into a large piece plus a small remainder. Default: 40.
36
+ */
37
+ NoSplitPercent?: number;
38
+ }
39
+ /**
40
+ * Splits text toward a **target** size, closing on the best available natural boundary
41
+ * near that target rather than cutting at a fixed offset.
42
+ *
43
+ * ## Why this beats a fixed window
44
+ *
45
+ * A fixed window cuts wherever the budget runs out, which routinely lands mid-paragraph
46
+ * — the chunk then straddles two ideas and matches neither query well. This segmenter
47
+ * treats the target as a goal with a tolerance band and escalates through boundary
48
+ * quality as it goes:
49
+ *
50
+ * 1. Once within `UndershootPercent` of target, close on a **paragraph** break.
51
+ * 2. Past target, accept a **sentence** break.
52
+ * 3. Past `OvershootPercent`, accept a **word** break.
53
+ * 4. Failing all of those, cut at the hard `MaxSegmentTokens` ceiling.
54
+ *
55
+ * Segment sizes therefore vary — deliberately. A slightly short segment that ends at a
56
+ * paragraph is worth more at retrieval time than an exactly-sized one that ends mid-clause.
57
+ *
58
+ * It also declines to split at all when the whole text is only modestly over target
59
+ * (`NoSplitPercent`), which avoids the common pathology of one full-size chunk followed by
60
+ * a runt carrying two sentences and no context.
61
+ *
62
+ * This is the recommended default for prose when document structure isn't available;
63
+ * prefer `StructuralText` when the content has headings, since an authored boundary beats
64
+ * an inferred one.
65
+ */
66
+ export declare class AdaptiveBoundarySegmenter extends BaseSegmenter {
67
+ get Key(): string;
68
+ get SupportedModalities(): ContentModality[];
69
+ protected SegmentCore(params: SegmentationParams<AdaptiveBoundarySegmentationOptions>): RawSegment[];
70
+ /** True when the whole text is close enough to target that splitting would only make a runt. */
71
+ private fitsWithoutSplitting;
72
+ /** Walk the text emitting one segment per located boundary. */
73
+ private walk;
74
+ /**
75
+ * Locate this segment's end by escalating through boundary quality.
76
+ * Ranges are half-open [from, to).
77
+ */
78
+ private findBoundary;
79
+ /**
80
+ * End offset of the last match of `pattern` starting within [from, to), or null.
81
+ * The returned offset is the END of the match, so the delimiter stays with the
82
+ * segment it terminates rather than opening the next one.
83
+ */
84
+ private lastMatch;
85
+ /** Merge caller options with adaptive-specific defaults. */
86
+ private resolveAdaptiveOptions;
87
+ /** Keep a percentage option inside a sane range. */
88
+ private clampPercent;
89
+ /** Target size expressed in characters (TextChunker's ~4 chars/token estimate). */
90
+ private targetChars;
91
+ /** Hard ceiling expressed in characters. */
92
+ private hardChars;
93
+ /** Overlap in characters, capped at half the target so segments always advance. */
94
+ private overlapChars;
95
+ /** Estimated tokens for a string — exposed for callers reasoning about sizing. */
96
+ EstimateTokens(text: string): number;
97
+ }
98
+ //# sourceMappingURL=AdaptiveBoundarySegmenter.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"AdaptiveBoundarySegmenter.d.ts","sourceRoot":"","sources":["../../src/generic/AdaptiveBoundarySegmenter.ts"],"names":[],"mappings":"AAAA;;;;GAIG;AAIH,OAAO,EAAE,aAAa,EAAE,MAAM,iBAAiB,CAAC;AAChD,OAAO,EAAE,eAAe,EAAE,UAAU,EAAE,mBAAmB,EAAE,kBAAkB,EAAE,MAAM,sBAAsB,CAAC;AAE5G,8DAA8D;AAC9D,eAAO,MAAM,+BAA+B,qBAAqB,CAAC;AAElE,6DAA6D;AAC7D,MAAM,WAAW,mCAAoC,SAAQ,mBAAmB;IAC5E;;;;;;;;OAQG;IACH,YAAY,CAAC,EAAE,MAAM,CAAC;IACtB;;;;OAIG;IACH,iBAAiB,CAAC,EAAE,MAAM,CAAC;IAC3B;;;OAGG;IACH,gBAAgB,CAAC,EAAE,MAAM,CAAC;IAC1B;;;OAGG;IACH,cAAc,CAAC,EAAE,MAAM,CAAC;CAC3B;AAUD;;;;;;;;;;;;;;;;;;;;;;;;;;GA0BG;AACH,qBACa,yBAA0B,SAAQ,aAAa;IACxD,IAAW,GAAG,IAAI,MAAM,CAEvB;IAED,IAAW,mBAAmB,IAAI,eAAe,EAAE,CAElD;IAED,SAAS,CAAC,WAAW,CAAC,MAAM,EAAE,kBAAkB,CAAC,mCAAmC,CAAC,GAAG,UAAU,EAAE;IAYpG,gGAAgG;IAChG,OAAO,CAAC,oBAAoB;IAK5B,+DAA+D;IAC/D,OAAO,CAAC,IAAI;IAoBZ;;;OAGG;IACH,OAAO,CAAC,YAAY;IA6BpB;;;;OAIG;IACH,OAAO,CAAC,SAAS;IAmBjB,4DAA4D;IAC5D,OAAO,CAAC,sBAAsB;IAiB9B,oDAAoD;IACpD,OAAO,CAAC,YAAY;IAOpB,mFAAmF;IACnF,OAAO,CAAC,WAAW;IAInB,4CAA4C;IAC5C,OAAO,CAAC,SAAS;IAIjB,mFAAmF;IACnF,OAAO,CAAC,YAAY;IAIpB,kFAAkF;IAC3E,cAAc,CAAC,IAAI,EAAE,MAAM,GAAG,MAAM;CAG9C"}
@@ -0,0 +1,177 @@
1
+ /**
2
+ * @fileoverview Target-size segmenter with an escalating boundary preference.
3
+ *
4
+ * @module @memberjunction/ai-segmentation
5
+ */
6
+ var __decorate = (this && this.__decorate) || function (decorators, target, key, desc) {
7
+ var c = arguments.length, r = c < 3 ? target : desc === null ? desc = Object.getOwnPropertyDescriptor(target, key) : desc, d;
8
+ if (typeof Reflect === "object" && typeof Reflect.decorate === "function") r = Reflect.decorate(decorators, target, key, desc);
9
+ else for (var i = decorators.length - 1; i >= 0; i--) if (d = decorators[i]) r = (c < 3 ? d(r) : c > 3 ? d(target, key, r) : d(target, key)) || r;
10
+ return c > 3 && r && Object.defineProperty(target, key, r), r;
11
+ };
12
+ import { RegisterClass } from '@memberjunction/global';
13
+ import { TextChunker } from '@memberjunction/ai-vectors';
14
+ import { BaseSegmenter } from './BaseSegmenter.js';
15
+ /** Registration key for {@link AdaptiveBoundarySegmenter}. */
16
+ export const ADAPTIVE_BOUNDARY_SEGMENTER_KEY = 'AdaptiveBoundary';
17
+ /**
18
+ * Splits text toward a **target** size, closing on the best available natural boundary
19
+ * near that target rather than cutting at a fixed offset.
20
+ *
21
+ * ## Why this beats a fixed window
22
+ *
23
+ * A fixed window cuts wherever the budget runs out, which routinely lands mid-paragraph
24
+ * — the chunk then straddles two ideas and matches neither query well. This segmenter
25
+ * treats the target as a goal with a tolerance band and escalates through boundary
26
+ * quality as it goes:
27
+ *
28
+ * 1. Once within `UndershootPercent` of target, close on a **paragraph** break.
29
+ * 2. Past target, accept a **sentence** break.
30
+ * 3. Past `OvershootPercent`, accept a **word** break.
31
+ * 4. Failing all of those, cut at the hard `MaxSegmentTokens` ceiling.
32
+ *
33
+ * Segment sizes therefore vary — deliberately. A slightly short segment that ends at a
34
+ * paragraph is worth more at retrieval time than an exactly-sized one that ends mid-clause.
35
+ *
36
+ * It also declines to split at all when the whole text is only modestly over target
37
+ * (`NoSplitPercent`), which avoids the common pathology of one full-size chunk followed by
38
+ * a runt carrying two sentences and no context.
39
+ *
40
+ * This is the recommended default for prose when document structure isn't available;
41
+ * prefer `StructuralText` when the content has headings, since an authored boundary beats
42
+ * an inferred one.
43
+ */
44
+ let AdaptiveBoundarySegmenter = class AdaptiveBoundarySegmenter extends BaseSegmenter {
45
+ get Key() {
46
+ return ADAPTIVE_BOUNDARY_SEGMENTER_KEY;
47
+ }
48
+ get SupportedModalities() {
49
+ return ['text'];
50
+ }
51
+ SegmentCore(params) {
52
+ const text = params.Text ?? '';
53
+ if (text.trim().length === 0) {
54
+ return [];
55
+ }
56
+ const settings = this.resolveAdaptiveOptions(params.Options);
57
+ if (this.fitsWithoutSplitting(text, settings)) {
58
+ return [{ Modality: 'text', Text: text.trim(), StartOffset: 0, EndOffset: text.length }];
59
+ }
60
+ return this.walk(text, settings);
61
+ }
62
+ /** True when the whole text is close enough to target that splitting would only make a runt. */
63
+ fitsWithoutSplitting(text, settings) {
64
+ const limit = this.targetChars(settings) * (1 + settings.NoSplitPercent / 100);
65
+ return text.length <= limit;
66
+ }
67
+ /** Walk the text emitting one segment per located boundary. */
68
+ walk(text, settings) {
69
+ const segments = [];
70
+ const overlapChars = this.overlapChars(settings);
71
+ let cursor = 0;
72
+ while (cursor < text.length) {
73
+ const boundary = this.findBoundary(text, cursor, settings);
74
+ const body = text.slice(cursor, boundary.End).trim();
75
+ if (body.length > 0) {
76
+ segments.push({ Modality: 'text', Text: body, StartOffset: cursor, EndOffset: boundary.End });
77
+ }
78
+ if (boundary.End >= text.length) {
79
+ break;
80
+ }
81
+ // Always advance, even when the overlap would otherwise stall the cursor.
82
+ cursor = Math.max(boundary.End - overlapChars, cursor + 1);
83
+ }
84
+ return segments;
85
+ }
86
+ /**
87
+ * Locate this segment's end by escalating through boundary quality.
88
+ * Ranges are half-open [from, to).
89
+ */
90
+ findBoundary(text, cursor, settings) {
91
+ const target = this.targetChars(settings);
92
+ const softMin = cursor + Math.floor(target * (1 - settings.UndershootPercent / 100));
93
+ const softMax = cursor + Math.floor(target * (1 + settings.OvershootPercent / 100));
94
+ const hardMax = cursor + this.hardChars(settings);
95
+ if (text.length <= softMax) {
96
+ return { End: text.length, Kind: 'end' };
97
+ }
98
+ const paragraph = this.lastMatch(text, /\n\s*\n/g, softMin, softMax);
99
+ if (paragraph !== null) {
100
+ return { End: paragraph, Kind: 'paragraph' };
101
+ }
102
+ const sentence = this.lastMatch(text, /[.!?]["')\]]?\s/g, softMin, softMax);
103
+ if (sentence !== null) {
104
+ return { End: sentence, Kind: 'sentence' };
105
+ }
106
+ const word = this.lastMatch(text, /\s+/g, softMin, Math.min(hardMax, text.length));
107
+ if (word !== null) {
108
+ return { End: word, Kind: 'word' };
109
+ }
110
+ return { End: Math.min(hardMax, text.length), Kind: 'hard' };
111
+ }
112
+ /**
113
+ * End offset of the last match of `pattern` starting within [from, to), or null.
114
+ * The returned offset is the END of the match, so the delimiter stays with the
115
+ * segment it terminates rather than opening the next one.
116
+ */
117
+ lastMatch(text, pattern, from, to) {
118
+ if (to <= from) {
119
+ return null;
120
+ }
121
+ const scan = new RegExp(pattern.source, 'g');
122
+ scan.lastIndex = Math.max(from, 0);
123
+ let found = null;
124
+ let match = scan.exec(text);
125
+ while (match !== null && match.index < to) {
126
+ found = match.index + match[0].length;
127
+ match = scan.exec(text);
128
+ }
129
+ return found;
130
+ }
131
+ // ─────────────────────────────────────────────
132
+ // Settings
133
+ // ─────────────────────────────────────────────
134
+ /** Merge caller options with adaptive-specific defaults. */
135
+ resolveAdaptiveOptions(options) {
136
+ const base = this.resolveOptions(options);
137
+ const target = Math.max(options?.TargetTokens ?? base.MaxSegmentTokens, 1);
138
+ return {
139
+ ...base,
140
+ // The hard ceiling can never sit below the target, or every segment would be
141
+ // cut by the ceiling before a boundary was ever considered.
142
+ MaxSegmentTokens: Math.max(base.MaxSegmentTokens, target),
143
+ TargetTokens: target,
144
+ UndershootPercent: this.clampPercent(options?.UndershootPercent ?? 20, 0, 90),
145
+ OvershootPercent: this.clampPercent(options?.OvershootPercent ?? 20, 0, 200),
146
+ NoSplitPercent: this.clampPercent(options?.NoSplitPercent ?? 40, 0, 500),
147
+ };
148
+ }
149
+ /** Keep a percentage option inside a sane range. */
150
+ clampPercent(value, min, max) {
151
+ if (!Number.isFinite(value)) {
152
+ return min;
153
+ }
154
+ return Math.min(Math.max(value, min), max);
155
+ }
156
+ /** Target size expressed in characters (TextChunker's ~4 chars/token estimate). */
157
+ targetChars(settings) {
158
+ return settings.TargetTokens * 4;
159
+ }
160
+ /** Hard ceiling expressed in characters. */
161
+ hardChars(settings) {
162
+ return settings.MaxSegmentTokens * 4;
163
+ }
164
+ /** Overlap in characters, capped at half the target so segments always advance. */
165
+ overlapChars(settings) {
166
+ return Math.min(settings.OverlapTokens * 4, Math.floor(this.targetChars(settings) / 2));
167
+ }
168
+ /** Estimated tokens for a string — exposed for callers reasoning about sizing. */
169
+ EstimateTokens(text) {
170
+ return TextChunker.EstimateTokenCount(text);
171
+ }
172
+ };
173
+ AdaptiveBoundarySegmenter = __decorate([
174
+ RegisterClass(BaseSegmenter, ADAPTIVE_BOUNDARY_SEGMENTER_KEY)
175
+ ], AdaptiveBoundarySegmenter);
176
+ export { AdaptiveBoundarySegmenter };
177
+ //# sourceMappingURL=AdaptiveBoundarySegmenter.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"AdaptiveBoundarySegmenter.js","sourceRoot":"","sources":["../../src/generic/AdaptiveBoundarySegmenter.ts"],"names":[],"mappings":"AAAA;;;;GAIG;;;;;;;AAEH,OAAO,EAAE,aAAa,EAAE,MAAM,wBAAwB,CAAC;AACvD,OAAO,EAAE,WAAW,EAAE,MAAM,4BAA4B,CAAC;AACzD,OAAO,EAAE,aAAa,EAAE,MAAM,iBAAiB,CAAC;AAGhD,8DAA8D;AAC9D,MAAM,CAAC,MAAM,+BAA+B,GAAG,kBAAkB,CAAC;AAwClE;;;;;;;;;;;;;;;;;;;;;;;;;;GA0BG;AAEI,IAAM,yBAAyB,GAA/B,MAAM,yBAA0B,SAAQ,aAAa;IACxD,IAAW,GAAG;QACV,OAAO,+BAA+B,CAAC;IAC3C,CAAC;IAED,IAAW,mBAAmB;QAC1B,OAAO,CAAC,MAAM,CAAC,CAAC;IACpB,CAAC;IAES,WAAW,CAAC,MAA+D;QACjF,MAAM,IAAI,GAAG,MAAM,CAAC,IAAI,IAAI,EAAE,CAAC;QAC/B,IAAI,IAAI,CAAC,IAAI,EAAE,CAAC,MAAM,KAAK,CAAC,EAAE,CAAC;YAC3B,OAAO,EAAE,CAAC;QACd,CAAC;QACD,MAAM,QAAQ,GAAG,IAAI,CAAC,sBAAsB,CAAC,MAAM,CAAC,OAAO,CAAC,CAAC;QAC7D,IAAI,IAAI,CAAC,oBAAoB,CAAC,IAAI,EAAE,QAAQ,CAAC,EAAE,CAAC;YAC5C,OAAO,CAAC,EAAE,QAAQ,EAAE,MAAM,EAAE,IAAI,EAAE,IAAI,CAAC,IAAI,EAAE,EAAE,WAAW,EAAE,CAAC,EAAE,SAAS,EAAE,IAAI,CAAC,MAAM,EAAE,CAAC,CAAC;QAC7F,CAAC;QACD,OAAO,IAAI,CAAC,IAAI,CAAC,IAAI,EAAE,QAAQ,CAAC,CAAC;IACrC,CAAC;IAED,gGAAgG;IACxF,oBAAoB,CAAC,IAAY,EAAE,QAAuD;QAC9F,MAAM,KAAK,GAAG,IAAI,CAAC,WAAW,CAAC,QAAQ,CAAC,GAAG,CAAC,CAAC,GAAG,QAAQ,CAAC,cAAc,GAAG,GAAG,CAAC,CAAC;QAC/E,OAAO,IAAI,CAAC,MAAM,IAAI,KAAK,CAAC;IAChC,CAAC;IAED,+DAA+D;IACvD,IAAI,CAAC,IAAY,EAAE,QAAuD;QAC9E,MAAM,QAAQ,GAAiB,EAAE,CAAC;QAClC,MAAM,YAAY,GAAG,IAAI,CAAC,YAAY,CAAC,QAAQ,CAAC,CAAC;QACjD,IAAI,MAAM,GAAG,CAAC,CAAC;QAEf,OAAO,MAAM,GAAG,IAAI,CAAC,MAAM,EAAE,CAAC;YAC1B,MAAM,QAAQ,GAAG,IAAI,CAAC,YAAY,CAAC,IAAI,EAAE,MAAM,EAAE,QAAQ,CAAC,CAAC;YAC3D,MAAM,IAAI,GAAG,IAAI,CAAC,KAAK,CAAC,MAAM,EAAE,QAAQ,CAAC,GAAG,CAAC,CAAC,IAAI,EAAE,CAAC;YACrD,IAAI,IAAI,CAAC,MAAM,GAAG,CAAC,EAAE,CAAC;gBAClB,QAAQ,CAAC,IAAI,CAAC,EAAE,QAAQ,EAAE,MAAM,EAAE,IAAI,EAAE,IAAI,EAAE,WAAW,EAAE,MAAM,EAAE,SAAS,EAAE,QAAQ,CAAC,GAAG,EAAE,CAAC,CAAC;YAClG,CAAC;YACD,IAAI,QAAQ,CAAC,GAAG,IAAI,IAAI,CAAC,MAAM,EAAE,CAAC;gBAC9B,MAAM;YACV,CAAC;YACD,0EAA0E;YAC1E,MAAM,GAAG,IAAI,CAAC,GAAG,CAAC,QAAQ,CAAC,GAAG,GAAG,YAAY,EAAE,MAAM,GAAG,CAAC,CAAC,CAAC;QAC/D,CAAC;QACD,OAAO,QAAQ,CAAC;IACpB,CAAC;IAED;;;OAGG;IACK,YAAY,CAChB,IAAY,EACZ,MAAc,EACd,QAAuD;QAEvD,MAAM,MAAM,GAAG,IAAI,CAAC,WAAW,CAAC,QAAQ,CAAC,CAAC;QAC1C,MAAM,OAAO,GAAG,MAAM,GAAG,IAAI,CAAC,KAAK,CAAC,MAAM,GAAG,CAAC,CAAC,GAAG,QAAQ,CAAC,iBAAiB,GAAG,GAAG,CAAC,CAAC,CAAC;QACrF,MAAM,OAAO,GAAG,MAAM,GAAG,IAAI,CAAC,KAAK,CAAC,MAAM,GAAG,CAAC,CAAC,GAAG,QAAQ,CAAC,gBAAgB,GAAG,GAAG,CAAC,CAAC,CAAC;QACpF,MAAM,OAAO,GAAG,MAAM,GAAG,IAAI,CAAC,SAAS,CAAC,QAAQ,CAAC,CAAC;QAElD,IAAI,IAAI,CAAC,MAAM,IAAI,OAAO,EAAE,CAAC;YACzB,OAAO,EAAE,GAAG,EAAE,IAAI,CAAC,MAAM,EAAE,IAAI,EAAE,KAAK,EAAE,CAAC;QAC7C,CAAC;QAED,MAAM,SAAS,GAAG,IAAI,CAAC,SAAS,CAAC,IAAI,EAAE,UAAU,EAAE,OAAO,EAAE,OAAO,CAAC,CAAC;QACrE,IAAI,SAAS,KAAK,IAAI,EAAE,CAAC;YACrB,OAAO,EAAE,GAAG,EAAE,SAAS,EAAE,IAAI,EAAE,WAAW,EAAE,CAAC;QACjD,CAAC;QACD,MAAM,QAAQ,GAAG,IAAI,CAAC,SAAS,CAAC,IAAI,EAAE,kBAAkB,EAAE,OAAO,EAAE,OAAO,CAAC,CAAC;QAC5E,IAAI,QAAQ,KAAK,IAAI,EAAE,CAAC;YACpB,OAAO,EAAE,GAAG,EAAE,QAAQ,EAAE,IAAI,EAAE,UAAU,EAAE,CAAC;QAC/C,CAAC;QACD,MAAM,IAAI,GAAG,IAAI,CAAC,SAAS,CAAC,IAAI,EAAE,MAAM,EAAE,OAAO,EAAE,IAAI,CAAC,GAAG,CAAC,OAAO,EAAE,IAAI,CAAC,MAAM,CAAC,CAAC,CAAC;QACnF,IAAI,IAAI,KAAK,IAAI,EAAE,CAAC;YAChB,OAAO,EAAE,GAAG,EAAE,IAAI,EAAE,IAAI,EAAE,MAAM,EAAE,CAAC;QACvC,CAAC;QACD,OAAO,EAAE,GAAG,EAAE,IAAI,CAAC,GAAG,CAAC,OAAO,EAAE,IAAI,CAAC,MAAM,CAAC,EAAE,IAAI,EAAE,MAAM,EAAE,CAAC;IACjE,CAAC;IAED;;;;OAIG;IACK,SAAS,CAAC,IAAY,EAAE,OAAe,EAAE,IAAY,EAAE,EAAU;QACrE,IAAI,EAAE,IAAI,IAAI,EAAE,CAAC;YACb,OAAO,IAAI,CAAC;QAChB,CAAC;QACD,MAAM,IAAI,GAAG,IAAI,MAAM,CAAC,OAAO,CAAC,MAAM,EAAE,GAAG,CAAC,CAAC;QAC7C,IAAI,CAAC,SAAS,GAAG,IAAI,CAAC,GAAG,CAAC,IAAI,EAAE,CAAC,CAAC,CAAC;QACnC,IAAI,KAAK,GAAkB,IAAI,CAAC;QAChC,IAAI,KAAK,GAAG,IAAI,CAAC,IAAI,CAAC,IAAI,CAAC,CAAC;QAC5B,OAAO,KAAK,KAAK,IAAI,IAAI,KAAK,CAAC,KAAK,GAAG,EAAE,EAAE,CAAC;YACxC,KAAK,GAAG,KAAK,CAAC,KAAK,GAAG,KAAK,CAAC,CAAC,CAAC,CAAC,MAAM,CAAC;YACtC,KAAK,GAAG,IAAI,CAAC,IAAI,CAAC,IAAI,CAAC,CAAC;QAC5B,CAAC;QACD,OAAO,KAAK,CAAC;IACjB,CAAC;IAED,gDAAgD;IAChD,WAAW;IACX,gDAAgD;IAEhD,4DAA4D;IACpD,sBAAsB,CAC1B,OAA6C;QAE7C,MAAM,IAAI,GAAG,IAAI,CAAC,cAAc,CAAC,OAAO,CAAC,CAAC;QAC1C,MAAM,MAAM,GAAG,IAAI,CAAC,GAAG,CAAC,OAAO,EAAE,YAAY,IAAI,IAAI,CAAC,gBAAgB,EAAE,CAAC,CAAC,CAAC;QAC3E,OAAO;YACH,GAAG,IAAI;YACP,6EAA6E;YAC7E,4DAA4D;YAC5D,gBAAgB,EAAE,IAAI,CAAC,GAAG,CAAC,IAAI,CAAC,gBAAgB,EAAE,MAAM,CAAC;YACzD,YAAY,EAAE,MAAM;YACpB,iBAAiB,EAAE,IAAI,CAAC,YAAY,CAAC,OAAO,EAAE,iBAAiB,IAAI,EAAE,EAAE,CAAC,EAAE,EAAE,CAAC;YAC7E,gBAAgB,EAAE,IAAI,CAAC,YAAY,CAAC,OAAO,EAAE,gBAAgB,IAAI,EAAE,EAAE,CAAC,EAAE,GAAG,CAAC;YAC5E,cAAc,EAAE,IAAI,CAAC,YAAY,CAAC,OAAO,EAAE,cAAc,IAAI,EAAE,EAAE,CAAC,EAAE,GAAG,CAAC;SAC3E,CAAC;IACN,CAAC;IAED,oDAAoD;IAC5C,YAAY,CAAC,KAAa,EAAE,GAAW,EAAE,GAAW;QACxD,IAAI,CAAC,MAAM,CAAC,QAAQ,CAAC,KAAK,CAAC,EAAE,CAAC;YAC1B,OAAO,GAAG,CAAC;QACf,CAAC;QACD,OAAO,IAAI,CAAC,GAAG,CAAC,IAAI,CAAC,GAAG,CAAC,KAAK,EAAE,GAAG,CAAC,EAAE,GAAG,CAAC,CAAC;IAC/C,CAAC;IAED,mFAAmF;IAC3E,WAAW,CAAC,QAAuD;QACvE,OAAO,QAAQ,CAAC,YAAY,GAAG,CAAC,CAAC;IACrC,CAAC;IAED,4CAA4C;IACpC,SAAS,CAAC,QAAuD;QACrE,OAAO,QAAQ,CAAC,gBAAgB,GAAG,CAAC,CAAC;IACzC,CAAC;IAED,mFAAmF;IAC3E,YAAY,CAAC,QAAuD;QACxE,OAAO,IAAI,CAAC,GAAG,CAAC,QAAQ,CAAC,aAAa,GAAG,CAAC,EAAE,IAAI,CAAC,KAAK,CAAC,IAAI,CAAC,WAAW,CAAC,QAAQ,CAAC,GAAG,CAAC,CAAC,CAAC,CAAC;IAC5F,CAAC;IAED,kFAAkF;IAC3E,cAAc,CAAC,IAAY;QAC9B,OAAO,WAAW,CAAC,kBAAkB,CAAC,IAAI,CAAC,CAAC;IAChD,CAAC;CACJ,CAAA;AAtJY,yBAAyB;IADrC,aAAa,CAAC,aAAa,EAAE,+BAA+B,CAAC;GACjD,yBAAyB,CAsJrC"}
@@ -0,0 +1,101 @@
1
+ /**
2
+ * @fileoverview Pluggable content cleaning — the stage that runs before segmentation.
3
+ *
4
+ * @module @memberjunction/ai-segmentation
5
+ */
6
+ /** Options common to every cleaner. */
7
+ export interface ContentCleaningOptions {
8
+ /**
9
+ * CSS selectors whose content is the ONLY content to keep. When set, everything
10
+ * outside these selectors is discarded before any exclusion is applied.
11
+ *
12
+ * This is the highest-leverage knob for messy sources: pointing at `.article-body`
13
+ * removes navigation, sidebars, and advertising in one stroke, without having to
14
+ * enumerate every block you *don't* want.
15
+ */
16
+ IncludeSelectors?: string[];
17
+ /**
18
+ * CSS selectors to remove. Applied after `IncludeSelectors`. Use for the blocks that
19
+ * survive the include (inline ad slots, share widgets, cookie banners).
20
+ */
21
+ ExcludeSelectors?: string[];
22
+ /** Collapse runs of whitespace and blank lines. Default: true. */
23
+ NormalizeWhitespace?: boolean;
24
+ /** Maximum characters to retain; content beyond this is dropped. Unset = no limit. */
25
+ MaxLength?: number;
26
+ }
27
+ /** Input to a cleaning pass. */
28
+ export interface ContentCleaningParams<TOptions extends ContentCleaningOptions = ContentCleaningOptions> {
29
+ /** Raw content — markup or plain text. */
30
+ Content: string;
31
+ /** Mime type of `Content`, when known; lets a cleaner pick its parser. */
32
+ MimeType?: string;
33
+ /** Cleaner-specific options. */
34
+ Options?: TOptions;
35
+ }
36
+ /** Result of a cleaning pass. Never throws for content-shaped problems. */
37
+ export interface ContentCleaningResult {
38
+ Success: boolean;
39
+ /** Cleaned text. Equal to the input when `Success` is false. */
40
+ Content: string;
41
+ /** Registration key of the cleaner that ran. */
42
+ CleanerKey: string;
43
+ ErrorMessage?: string;
44
+ Warnings: string[];
45
+ /** Characters removed by cleaning — a cheap signal that a selector is wrong. */
46
+ CharactersRemoved: number;
47
+ }
48
+ /**
49
+ * Base class for content cleaning strategies.
50
+ *
51
+ * Cleaning is deliberately a **separate stage from segmentation**, and a separate
52
+ * plug-in point. The two answer different questions — cleaning asks *which text is
53
+ * actually content*, segmentation asks *where that content divides* — and they change for
54
+ * different reasons: a new CMS template needs new selectors, not a new chunking strategy.
55
+ * Splitting them also means the cleaning rules apply once and benefit every downstream
56
+ * consumer (embedding chunks, tagging chunks, full-text indexing) instead of being
57
+ * reimplemented per pipeline.
58
+ *
59
+ * Garbage that survives this stage is expensive: it gets embedded, stored, retrieved, and
60
+ * eventually shown to a user or an agent. Navigation chrome repeated across a thousand
61
+ * pages produces a thousand near-identical vectors that crowd out real answers.
62
+ *
63
+ * ```typescript
64
+ * @RegisterClass(BaseContentCleaner, 'MyCleaner')
65
+ * export class MyCleaner extends BaseContentCleaner {
66
+ * public get Key(): string { return 'MyCleaner'; }
67
+ * protected CleanCore(params: ContentCleaningParams): string { return strip(params.Content); }
68
+ * }
69
+ * ```
70
+ */
71
+ export declare abstract class BaseContentCleaner {
72
+ /** Registration key; must match the key passed to `@RegisterClass`. */
73
+ abstract get Key(): string;
74
+ /** Perform the cleaning. The base class handles validation, whitespace, and truncation. */
75
+ protected abstract CleanCore(params: ContentCleaningParams): string;
76
+ /**
77
+ * Clean content ahead of segmentation.
78
+ *
79
+ * Never throws for content-shaped problems — inspect `Success`/`ErrorMessage`. On
80
+ * failure the ORIGINAL content is returned rather than an empty string, so a bad
81
+ * selector degrades to "not cleaned" instead of silently discarding the document.
82
+ */
83
+ Clean(params: ContentCleaningParams): ContentCleaningResult;
84
+ /**
85
+ * Resolve a registered cleaner by key.
86
+ *
87
+ * Uses `TryCreateInstance` because `CreateInstance` never returns null for an unknown
88
+ * key — it silently yields a hollow base instance whose abstract members are undefined.
89
+ */
90
+ static Resolve(key: string): BaseContentCleaner | null;
91
+ /** Whitespace normalization and truncation, shared by every cleaner. */
92
+ protected applyCommonRules(content: string, options?: ContentCleaningOptions, warnings?: string[]): string;
93
+ /**
94
+ * Collapse horizontal whitespace and runs of blank lines, while preserving the single
95
+ * blank line that marks a paragraph break — segmenters rely on it as a boundary signal.
96
+ */
97
+ protected normalizeWhitespace(content: string): string;
98
+ /** Build a successful result with the removal delta computed. */
99
+ private buildResult;
100
+ }
101
+ //# sourceMappingURL=BaseContentCleaner.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"BaseContentCleaner.d.ts","sourceRoot":"","sources":["../../src/generic/BaseContentCleaner.ts"],"names":[],"mappings":"AAAA;;;;GAIG;AAKH,uCAAuC;AACvC,MAAM,WAAW,sBAAsB;IACnC;;;;;;;OAOG;IACH,gBAAgB,CAAC,EAAE,MAAM,EAAE,CAAC;IAC5B;;;OAGG;IACH,gBAAgB,CAAC,EAAE,MAAM,EAAE,CAAC;IAC5B,kEAAkE;IAClE,mBAAmB,CAAC,EAAE,OAAO,CAAC;IAC9B,sFAAsF;IACtF,SAAS,CAAC,EAAE,MAAM,CAAC;CACtB;AAED,gCAAgC;AAChC,MAAM,WAAW,qBAAqB,CAAC,QAAQ,SAAS,sBAAsB,GAAG,sBAAsB;IACnG,0CAA0C;IAC1C,OAAO,EAAE,MAAM,CAAC;IAChB,0EAA0E;IAC1E,QAAQ,CAAC,EAAE,MAAM,CAAC;IAClB,gCAAgC;IAChC,OAAO,CAAC,EAAE,QAAQ,CAAC;CACtB;AAED,2EAA2E;AAC3E,MAAM,WAAW,qBAAqB;IAClC,OAAO,EAAE,OAAO,CAAC;IACjB,gEAAgE;IAChE,OAAO,EAAE,MAAM,CAAC;IAChB,gDAAgD;IAChD,UAAU,EAAE,MAAM,CAAC;IACnB,YAAY,CAAC,EAAE,MAAM,CAAC;IACtB,QAAQ,EAAE,MAAM,EAAE,CAAC;IACnB,gFAAgF;IAChF,iBAAiB,EAAE,MAAM,CAAC;CAC7B;AAED;;;;;;;;;;;;;;;;;;;;;;GAsBG;AACH,8BAAsB,kBAAkB;IACpC,uEAAuE;IACvE,aAAoB,GAAG,IAAI,MAAM,CAAC;IAElC,2FAA2F;IAC3F,SAAS,CAAC,QAAQ,CAAC,SAAS,CAAC,MAAM,EAAE,qBAAqB,GAAG,MAAM;IAEnE;;;;;;OAMG;IACI,KAAK,CAAC,MAAM,EAAE,qBAAqB,GAAG,qBAAqB;IA6BlE;;;;;OAKG;WACW,OAAO,CAAC,GAAG,EAAE,MAAM,GAAG,kBAAkB,GAAG,IAAI;IAW7D,wEAAwE;IACxE,SAAS,CAAC,gBAAgB,CAAC,OAAO,EAAE,MAAM,EAAE,OAAO,CAAC,EAAE,sBAAsB,EAAE,QAAQ,CAAC,EAAE,MAAM,EAAE,GAAG,MAAM;IAY1G;;;OAGG;IACH,SAAS,CAAC,mBAAmB,CAAC,OAAO,EAAE,MAAM,GAAG,MAAM;IAQtD,iEAAiE;IACjE,OAAO,CAAC,WAAW;CAStB"}