@memberjunction/ai-segmentation 0.0.0 → 5.51.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (54) hide show
  1. package/README.md +177 -43
  2. package/dist/generic/AdaptiveBoundarySegmenter.d.ts +98 -0
  3. package/dist/generic/AdaptiveBoundarySegmenter.d.ts.map +1 -0
  4. package/dist/generic/AdaptiveBoundarySegmenter.js +177 -0
  5. package/dist/generic/AdaptiveBoundarySegmenter.js.map +1 -0
  6. package/dist/generic/BaseContentCleaner.d.ts +101 -0
  7. package/dist/generic/BaseContentCleaner.d.ts.map +1 -0
  8. package/dist/generic/BaseContentCleaner.js +114 -0
  9. package/dist/generic/BaseContentCleaner.js.map +1 -0
  10. package/dist/generic/BaseSegmenter.d.ts +106 -0
  11. package/dist/generic/BaseSegmenter.d.ts.map +1 -0
  12. package/dist/generic/BaseSegmenter.js +260 -0
  13. package/dist/generic/BaseSegmenter.js.map +1 -0
  14. package/dist/generic/FixedWindowSegmenter.d.ts +49 -0
  15. package/dist/generic/FixedWindowSegmenter.d.ts.map +1 -0
  16. package/dist/generic/FixedWindowSegmenter.js +106 -0
  17. package/dist/generic/FixedWindowSegmenter.js.map +1 -0
  18. package/dist/generic/HtmlContentCleaner.d.ts +61 -0
  19. package/dist/generic/HtmlContentCleaner.d.ts.map +1 -0
  20. package/dist/generic/HtmlContentCleaner.js +130 -0
  21. package/dist/generic/HtmlContentCleaner.js.map +1 -0
  22. package/dist/generic/PagedContentSegmenter.d.ts +55 -0
  23. package/dist/generic/PagedContentSegmenter.d.ts.map +1 -0
  24. package/dist/generic/PagedContentSegmenter.js +99 -0
  25. package/dist/generic/PagedContentSegmenter.js.map +1 -0
  26. package/dist/generic/PlainTextContentCleaner.d.ts +22 -0
  27. package/dist/generic/PlainTextContentCleaner.d.ts.map +1 -0
  28. package/dist/generic/PlainTextContentCleaner.js +37 -0
  29. package/dist/generic/PlainTextContentCleaner.js.map +1 -0
  30. package/dist/generic/Segmentation.types.d.ts +198 -0
  31. package/dist/generic/Segmentation.types.d.ts.map +1 -0
  32. package/dist/generic/Segmentation.types.js +19 -0
  33. package/dist/generic/Segmentation.types.js.map +1 -0
  34. package/dist/generic/SegmentationResolver.d.ts +43 -0
  35. package/dist/generic/SegmentationResolver.d.ts.map +1 -0
  36. package/dist/generic/SegmentationResolver.js +83 -0
  37. package/dist/generic/SegmentationResolver.js.map +1 -0
  38. package/dist/generic/SemanticTextSegmenter.d.ts +80 -0
  39. package/dist/generic/SemanticTextSegmenter.d.ts.map +1 -0
  40. package/dist/generic/SemanticTextSegmenter.js +201 -0
  41. package/dist/generic/SemanticTextSegmenter.js.map +1 -0
  42. package/dist/generic/StructuralTextSegmenter.d.ts +63 -0
  43. package/dist/generic/StructuralTextSegmenter.d.ts.map +1 -0
  44. package/dist/generic/StructuralTextSegmenter.js +177 -0
  45. package/dist/generic/StructuralTextSegmenter.js.map +1 -0
  46. package/dist/generic/TranscriptSegmenter.d.ts +78 -0
  47. package/dist/generic/TranscriptSegmenter.d.ts.map +1 -0
  48. package/dist/generic/TranscriptSegmenter.js +194 -0
  49. package/dist/generic/TranscriptSegmenter.js.map +1 -0
  50. package/dist/index.d.ts +33 -0
  51. package/dist/index.d.ts.map +1 -0
  52. package/dist/index.js +36 -0
  53. package/dist/index.js.map +1 -0
  54. package/package.json +33 -7
@@ -0,0 +1,106 @@
1
+ /**
2
+ * @fileoverview Fixed-window segmenter — the universal fallback strategy.
3
+ *
4
+ * @module @memberjunction/ai-segmentation
5
+ */
6
+ var __decorate = (this && this.__decorate) || function (decorators, target, key, desc) {
7
+ var c = arguments.length, r = c < 3 ? target : desc === null ? desc = Object.getOwnPropertyDescriptor(target, key) : desc, d;
8
+ if (typeof Reflect === "object" && typeof Reflect.decorate === "function") r = Reflect.decorate(decorators, target, key, desc);
9
+ else for (var i = decorators.length - 1; i >= 0; i--) if (d = decorators[i]) r = (c < 3 ? d(r) : c > 3 ? d(target, key, r) : d(target, key)) || r;
10
+ return c > 3 && r && Object.defineProperty(target, key, r), r;
11
+ };
12
+ import { RegisterClass } from '@memberjunction/global';
13
+ import { BaseSegmenter } from './BaseSegmenter.js';
14
+ import { TextChunker } from '@memberjunction/ai-vectors';
15
+ /** Registration key for {@link FixedWindowSegmenter}. */
16
+ export const FIXED_WINDOW_SEGMENTER_KEY = 'FixedWindow';
17
+ /**
18
+ * Splits content into uniform windows: token-bounded chunks for text, and
19
+ * fixed-duration windows for audio/video that has no transcript.
20
+ *
21
+ * This is the safety net, not the recommended default. It requires no LLM call,
22
+ * no transcript, and no document structure, so it always produces *something* —
23
+ * which makes it the right choice for logs, machine-generated text, and media
24
+ * that arrives without cues. Where structure or a transcript exists, prefer
25
+ * `StructuralText` or `Transcript`, both of which cut on real boundaries.
26
+ */
27
+ let FixedWindowSegmenter = class FixedWindowSegmenter extends BaseSegmenter {
28
+ get Key() {
29
+ return FIXED_WINDOW_SEGMENTER_KEY;
30
+ }
31
+ get SupportedModalities() {
32
+ return ['text', 'image', 'audio', 'video', 'multimodal'];
33
+ }
34
+ SegmentCore(params) {
35
+ if (params.Text && params.Text.trim().length > 0) {
36
+ return this.segmentText(params);
37
+ }
38
+ if (params.Media) {
39
+ return this.segmentMedia(params);
40
+ }
41
+ return [];
42
+ }
43
+ /** Token-bounded text windows, offsets preserved. */
44
+ segmentText(params) {
45
+ const settings = this.resolveOptions(params.Options);
46
+ const chunks = TextChunker.ChunkText({
47
+ Text: params.Text ?? '',
48
+ MaxChunkTokens: settings.MaxSegmentTokens,
49
+ OverlapTokens: settings.OverlapTokens,
50
+ Strategy: params.Options?.TextStrategy ?? 'sentence',
51
+ });
52
+ return chunks.map((chunk) => ({
53
+ Modality: 'text',
54
+ Text: chunk.Text,
55
+ StartOffset: chunk.StartOffset,
56
+ EndOffset: chunk.EndOffset,
57
+ }));
58
+ }
59
+ /** Fixed-duration media windows, or a single segment for untimed media. */
60
+ segmentMedia(params) {
61
+ const modality = this.resolveMediaModality(params);
62
+ if (modality === 'image' || !params.DurationMs || params.DurationMs <= 0) {
63
+ return [{ Modality: modality, Media: params.Media }];
64
+ }
65
+ return this.buildTimeWindows(params, modality);
66
+ }
67
+ /** Walk the asset duration emitting one window per step. */
68
+ buildTimeWindows(params, modality) {
69
+ const windowMs = Math.max(params.Options?.WindowMs ?? 30_000, 1);
70
+ // Cap overlap at half the window, matching TextChunker's rule for text. Beyond 50% each
71
+ // window is mostly a copy of the previous one, and as overlap approaches the window size the
72
+ // segment count explodes — a 1s window with 5s of overlap would emit one segment per
73
+ // millisecond, and every one of those is a paid multimodal embedding call.
74
+ const overlapMs = Math.min(params.Options?.WindowOverlapMs ?? 0, Math.floor(windowMs / 2));
75
+ const step = Math.max(windowMs - overlapMs, 1);
76
+ const duration = params.DurationMs ?? 0;
77
+ const segments = [];
78
+ for (let start = 0; start < duration; start += step) {
79
+ const end = Math.min(start + windowMs, duration);
80
+ segments.push({ Modality: modality, Media: params.Media, StartMs: start, EndMs: end });
81
+ if (end >= duration) {
82
+ break;
83
+ }
84
+ }
85
+ return segments;
86
+ }
87
+ /** Infer the media modality from its mime type. */
88
+ resolveMediaModality(params) {
89
+ const mime = (params.Media?.MimeType ?? params.MimeType ?? '').toLowerCase();
90
+ if (mime.startsWith('video/')) {
91
+ return 'video';
92
+ }
93
+ if (mime.startsWith('audio/')) {
94
+ return 'audio';
95
+ }
96
+ if (mime.startsWith('image/')) {
97
+ return 'image';
98
+ }
99
+ return 'multimodal';
100
+ }
101
+ };
102
+ FixedWindowSegmenter = __decorate([
103
+ RegisterClass(BaseSegmenter, FIXED_WINDOW_SEGMENTER_KEY)
104
+ ], FixedWindowSegmenter);
105
+ export { FixedWindowSegmenter };
106
+ //# sourceMappingURL=FixedWindowSegmenter.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"FixedWindowSegmenter.js","sourceRoot":"","sources":["../../src/generic/FixedWindowSegmenter.ts"],"names":[],"mappings":"AAAA;;;;GAIG;;;;;;;AAEH,OAAO,EAAE,aAAa,EAAE,MAAM,wBAAwB,CAAC;AACvD,OAAO,EAAE,aAAa,EAAE,MAAM,iBAAiB,CAAC;AAChD,OAAO,EAAE,WAAW,EAAE,MAAM,4BAA4B,CAAC;AAGzD,yDAAyD;AACzD,MAAM,CAAC,MAAM,0BAA0B,GAAG,aAAa,CAAC;AAmBxD;;;;;;;;;GASG;AAEI,IAAM,oBAAoB,GAA1B,MAAM,oBAAqB,SAAQ,aAAa;IACnD,IAAW,GAAG;QACV,OAAO,0BAA0B,CAAC;IACtC,CAAC;IAED,IAAW,mBAAmB;QAC1B,OAAO,CAAC,MAAM,EAAE,OAAO,EAAE,OAAO,EAAE,OAAO,EAAE,YAAY,CAAC,CAAC;IAC7D,CAAC;IAES,WAAW,CAAC,MAA0D;QAC5E,IAAI,MAAM,CAAC,IAAI,IAAI,MAAM,CAAC,IAAI,CAAC,IAAI,EAAE,CAAC,MAAM,GAAG,CAAC,EAAE,CAAC;YAC/C,OAAO,IAAI,CAAC,WAAW,CAAC,MAAM,CAAC,CAAC;QACpC,CAAC;QACD,IAAI,MAAM,CAAC,KAAK,EAAE,CAAC;YACf,OAAO,IAAI,CAAC,YAAY,CAAC,MAAM,CAAC,CAAC;QACrC,CAAC;QACD,OAAO,EAAE,CAAC;IACd,CAAC;IAED,qDAAqD;IAC7C,WAAW,CAAC,MAA0D;QAC1E,MAAM,QAAQ,GAAG,IAAI,CAAC,cAAc,CAAC,MAAM,CAAC,OAAO,CAAC,CAAC;QACrD,MAAM,MAAM,GAAG,WAAW,CAAC,SAAS,CAAC;YACjC,IAAI,EAAE,MAAM,CAAC,IAAI,IAAI,EAAE;YACvB,cAAc,EAAE,QAAQ,CAAC,gBAAgB;YACzC,aAAa,EAAE,QAAQ,CAAC,aAAa;YACrC,QAAQ,EAAE,MAAM,CAAC,OAAO,EAAE,YAAY,IAAI,UAAU;SACvD,CAAC,CAAC;QACH,OAAO,MAAM,CAAC,GAAG,CAAC,CAAC,KAAK,EAAE,EAAE,CAAC,CAAC;YAC1B,QAAQ,EAAE,MAAM;YAChB,IAAI,EAAE,KAAK,CAAC,IAAI;YAChB,WAAW,EAAE,KAAK,CAAC,WAAW;YAC9B,SAAS,EAAE,KAAK,CAAC,SAAS;SAC7B,CAAC,CAAC,CAAC;IACR,CAAC;IAED,2EAA2E;IACnE,YAAY,CAAC,MAA0D;QAC3E,MAAM,QAAQ,GAAG,IAAI,CAAC,oBAAoB,CAAC,MAAM,CAAC,CAAC;QACnD,IAAI,QAAQ,KAAK,OAAO,IAAI,CAAC,MAAM,CAAC,UAAU,IAAI,MAAM,CAAC,UAAU,IAAI,CAAC,EAAE,CAAC;YACvE,OAAO,CAAC,EAAE,QAAQ,EAAE,QAAQ,EAAE,KAAK,EAAE,MAAM,CAAC,KAAK,EAAE,CAAC,CAAC;QACzD,CAAC;QACD,OAAO,IAAI,CAAC,gBAAgB,CAAC,MAAM,EAAE,QAAQ,CAAC,CAAC;IACnD,CAAC;IAED,4DAA4D;IACpD,gBAAgB,CACpB,MAA0D,EAC1D,QAAyB;QAEzB,MAAM,QAAQ,GAAG,IAAI,CAAC,GAAG,CAAC,MAAM,CAAC,OAAO,EAAE,QAAQ,IAAI,MAAM,EAAE,CAAC,CAAC,CAAC;QACjE,wFAAwF;QACxF,6FAA6F;QAC7F,qFAAqF;QACrF,2EAA2E;QAC3E,MAAM,SAAS,GAAG,IAAI,CAAC,GAAG,CAAC,MAAM,CAAC,OAAO,EAAE,eAAe,IAAI,CAAC,EAAE,IAAI,CAAC,KAAK,CAAC,QAAQ,GAAG,CAAC,CAAC,CAAC,CAAC;QAC3F,MAAM,IAAI,GAAG,IAAI,CAAC,GAAG,CAAC,QAAQ,GAAG,SAAS,EAAE,CAAC,CAAC,CAAC;QAC/C,MAAM,QAAQ,GAAG,MAAM,CAAC,UAAU,IAAI,CAAC,CAAC;QACxC,MAAM,QAAQ,GAAiB,EAAE,CAAC;QAElC,KAAK,IAAI,KAAK,GAAG,CAAC,EAAE,KAAK,GAAG,QAAQ,EAAE,KAAK,IAAI,IAAI,EAAE,CAAC;YAClD,MAAM,GAAG,GAAG,IAAI,CAAC,GAAG,CAAC,KAAK,GAAG,QAAQ,EAAE,QAAQ,CAAC,CAAC;YACjD,QAAQ,CAAC,IAAI,CAAC,EAAE,QAAQ,EAAE,QAAQ,EAAE,KAAK,EAAE,MAAM,CAAC,KAAK,EAAE,OAAO,EAAE,KAAK,EAAE,KAAK,EAAE,GAAG,EAAE,CAAC,CAAC;YACvF,IAAI,GAAG,IAAI,QAAQ,EAAE,CAAC;gBAClB,MAAM;YACV,CAAC;QACL,CAAC;QACD,OAAO,QAAQ,CAAC;IACpB,CAAC;IAED,mDAAmD;IAC3C,oBAAoB,CAAC,MAA0D;QACnF,MAAM,IAAI,GAAG,CAAC,MAAM,CAAC,KAAK,EAAE,QAAQ,IAAI,MAAM,CAAC,QAAQ,IAAI,EAAE,CAAC,CAAC,WAAW,EAAE,CAAC;QAC7E,IAAI,IAAI,CAAC,UAAU,CAAC,QAAQ,CAAC,EAAE,CAAC;YAC5B,OAAO,OAAO,CAAC;QACnB,CAAC;QACD,IAAI,IAAI,CAAC,UAAU,CAAC,QAAQ,CAAC,EAAE,CAAC;YAC5B,OAAO,OAAO,CAAC;QACnB,CAAC;QACD,IAAI,IAAI,CAAC,UAAU,CAAC,QAAQ,CAAC,EAAE,CAAC;YAC5B,OAAO,OAAO,CAAC;QACnB,CAAC;QACD,OAAO,YAAY,CAAC;IACxB,CAAC;CACJ,CAAA;AApFY,oBAAoB;IADhC,aAAa,CAAC,aAAa,EAAE,0BAA0B,CAAC;GAC5C,oBAAoB,CAoFhC"}
@@ -0,0 +1,61 @@
1
+ /**
2
+ * @fileoverview Selector-driven HTML cleaner.
3
+ *
4
+ * @module @memberjunction/ai-segmentation
5
+ */
6
+ import { BaseContentCleaner, ContentCleaningOptions, ContentCleaningParams } from './BaseContentCleaner.js';
7
+ /** Registration key for {@link HtmlContentCleaner}. */
8
+ export declare const HTML_CONTENT_CLEANER_KEY = "Html";
9
+ /**
10
+ * Elements removed before extraction unless the caller overrides `ExcludeSelectors`.
11
+ * These carry no document content on essentially any site.
12
+ */
13
+ export declare const DEFAULT_HTML_EXCLUDE_SELECTORS: string[];
14
+ /** Options specific to {@link HtmlContentCleaner}. */
15
+ export interface HtmlContentCleaningOptions extends ContentCleaningOptions {
16
+ /**
17
+ * Replace the built-in exclusion list rather than adding to it. Default: false
18
+ * (caller selectors are appended to {@link DEFAULT_HTML_EXCLUDE_SELECTORS}).
19
+ */
20
+ ReplaceDefaultExcludes?: boolean;
21
+ /** Keep `alt` text from images as content. Default: false. */
22
+ IncludeImageAltText?: boolean;
23
+ }
24
+ /**
25
+ * Extracts readable text from HTML using CSS selectors.
26
+ *
27
+ * Real-world source pages are mostly not content: navigation, sidebars, cookie banners,
28
+ * share widgets, related-article rails, and advertising typically outweigh the article
29
+ * itself. Stripping tags alone keeps all of that text, and because chrome repeats across
30
+ * every page of a site it produces many near-identical chunks that crowd out real answers
31
+ * at retrieval time.
32
+ *
33
+ * The high-leverage control is `IncludeSelectors` — naming the one element that holds the
34
+ * content (`.article-body`, `main`, `#post`) discards everything else without having to
35
+ * enumerate what to drop. `ExcludeSelectors` then handles whatever survives inside it.
36
+ *
37
+ * Both are per-source configuration because the right selector is a property of the site's
38
+ * template, not of MemberJunction. A sensible default exclusion list handles sources that
39
+ * haven't been tuned yet.
40
+ */
41
+ export declare class HtmlContentCleaner extends BaseContentCleaner {
42
+ get Key(): string;
43
+ protected CleanCore(params: ContentCleaningParams<HtmlContentCleaningOptions>): string;
44
+ /** Drop excluded elements from the document. */
45
+ private removeExcluded;
46
+ /**
47
+ * Narrow to the included selectors when supplied, else the body.
48
+ * Returns a selection whose text is the document's content.
49
+ */
50
+ private resolveScope;
51
+ /** Append image alt text so meaningful figures aren't lost, when requested. */
52
+ private promoteImageAltText;
53
+ /**
54
+ * Extract text, inserting newlines at block boundaries.
55
+ *
56
+ * cheerio's `.text()` concatenates without separators, so `<p>a</p><p>b</p>` becomes
57
+ * "ab" — which destroys the paragraph breaks segmenters depend on for boundaries.
58
+ */
59
+ private extractText;
60
+ }
61
+ //# sourceMappingURL=HtmlContentCleaner.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"HtmlContentCleaner.d.ts","sourceRoot":"","sources":["../../src/generic/HtmlContentCleaner.ts"],"names":[],"mappings":"AAAA;;;;GAIG;AAIH,OAAO,EAAE,kBAAkB,EAAE,sBAAsB,EAAE,qBAAqB,EAAE,MAAM,sBAAsB,CAAC;AAEzG,uDAAuD;AACvD,eAAO,MAAM,wBAAwB,SAAS,CAAC;AAE/C;;;GAGG;AACH,eAAO,MAAM,8BAA8B,UAa1C,CAAC;AAEF,sDAAsD;AACtD,MAAM,WAAW,0BAA2B,SAAQ,sBAAsB;IACtE;;;OAGG;IACH,sBAAsB,CAAC,EAAE,OAAO,CAAC;IACjC,8DAA8D;IAC9D,mBAAmB,CAAC,EAAE,OAAO,CAAC;CACjC;AAED;;;;;;;;;;;;;;;;GAgBG;AACH,qBACa,kBAAmB,SAAQ,kBAAkB;IACtD,IAAW,GAAG,IAAI,MAAM,CAEvB;IAED,SAAS,CAAC,SAAS,CAAC,MAAM,EAAE,qBAAqB,CAAC,0BAA0B,CAAC,GAAG,MAAM;IAWtF,gDAAgD;IAChD,OAAO,CAAC,cAAc;IAkBtB;;;OAGG;IACH,OAAO,CAAC,YAAY;IAoBpB,+EAA+E;IAC/E,OAAO,CAAC,mBAAmB;IAgB3B;;;;;OAKG;IACH,OAAO,CAAC,WAAW;CAQtB"}
@@ -0,0 +1,130 @@
1
+ /**
2
+ * @fileoverview Selector-driven HTML cleaner.
3
+ *
4
+ * @module @memberjunction/ai-segmentation
5
+ */
6
+ var __decorate = (this && this.__decorate) || function (decorators, target, key, desc) {
7
+ var c = arguments.length, r = c < 3 ? target : desc === null ? desc = Object.getOwnPropertyDescriptor(target, key) : desc, d;
8
+ if (typeof Reflect === "object" && typeof Reflect.decorate === "function") r = Reflect.decorate(decorators, target, key, desc);
9
+ else for (var i = decorators.length - 1; i >= 0; i--) if (d = decorators[i]) r = (c < 3 ? d(r) : c > 3 ? d(target, key, r) : d(target, key)) || r;
10
+ return c > 3 && r && Object.defineProperty(target, key, r), r;
11
+ };
12
+ import * as cheerio from 'cheerio';
13
+ import { RegisterClass } from '@memberjunction/global';
14
+ import { BaseContentCleaner } from './BaseContentCleaner.js';
15
+ /** Registration key for {@link HtmlContentCleaner}. */
16
+ export const HTML_CONTENT_CLEANER_KEY = 'Html';
17
+ /**
18
+ * Elements removed before extraction unless the caller overrides `ExcludeSelectors`.
19
+ * These carry no document content on essentially any site.
20
+ */
21
+ export const DEFAULT_HTML_EXCLUDE_SELECTORS = [
22
+ 'script',
23
+ 'style',
24
+ 'noscript',
25
+ 'iframe',
26
+ 'svg',
27
+ 'nav',
28
+ 'header',
29
+ 'footer',
30
+ 'aside',
31
+ 'form',
32
+ '[aria-hidden="true"]',
33
+ '.hidden',
34
+ ];
35
+ /**
36
+ * Extracts readable text from HTML using CSS selectors.
37
+ *
38
+ * Real-world source pages are mostly not content: navigation, sidebars, cookie banners,
39
+ * share widgets, related-article rails, and advertising typically outweigh the article
40
+ * itself. Stripping tags alone keeps all of that text, and because chrome repeats across
41
+ * every page of a site it produces many near-identical chunks that crowd out real answers
42
+ * at retrieval time.
43
+ *
44
+ * The high-leverage control is `IncludeSelectors` — naming the one element that holds the
45
+ * content (`.article-body`, `main`, `#post`) discards everything else without having to
46
+ * enumerate what to drop. `ExcludeSelectors` then handles whatever survives inside it.
47
+ *
48
+ * Both are per-source configuration because the right selector is a property of the site's
49
+ * template, not of MemberJunction. A sensible default exclusion list handles sources that
50
+ * haven't been tuned yet.
51
+ */
52
+ let HtmlContentCleaner = class HtmlContentCleaner extends BaseContentCleaner {
53
+ get Key() {
54
+ return HTML_CONTENT_CLEANER_KEY;
55
+ }
56
+ CleanCore(params) {
57
+ const options = params.Options;
58
+ const $ = cheerio.load(params.Content);
59
+ this.removeExcluded($, options);
60
+ const scope = this.resolveScope($, options);
61
+ this.promoteImageAltText($, scope, options);
62
+ return this.extractText($, scope);
63
+ }
64
+ /** Drop excluded elements from the document. */
65
+ removeExcluded($, options) {
66
+ const selectors = options?.ReplaceDefaultExcludes
67
+ ? options?.ExcludeSelectors ?? []
68
+ : [...DEFAULT_HTML_EXCLUDE_SELECTORS, ...(options?.ExcludeSelectors ?? [])];
69
+ for (const selector of selectors) {
70
+ try {
71
+ $(selector).remove();
72
+ }
73
+ catch {
74
+ // An invalid selector shouldn't fail the whole clean — skip it and continue,
75
+ // since the rest of the rules are still worth applying.
76
+ }
77
+ }
78
+ }
79
+ /**
80
+ * Narrow to the included selectors when supplied, else the body.
81
+ * Returns a selection whose text is the document's content.
82
+ */
83
+ resolveScope($, options) {
84
+ const includes = options?.IncludeSelectors ?? [];
85
+ for (const selector of includes) {
86
+ try {
87
+ const matched = $(selector);
88
+ if (matched.length > 0) {
89
+ return matched;
90
+ }
91
+ }
92
+ catch {
93
+ // Ignore an invalid include selector and try the next one.
94
+ }
95
+ }
96
+ const body = $('body');
97
+ const scope = body.length > 0 ? body : $.root();
98
+ return scope;
99
+ }
100
+ /** Append image alt text so meaningful figures aren't lost, when requested. */
101
+ promoteImageAltText($, scope, options) {
102
+ if (!options?.IncludeImageAltText) {
103
+ return;
104
+ }
105
+ $(scope).find('img').each((_i, el) => {
106
+ const alt = $(el).attr('alt');
107
+ if (alt && alt.trim().length > 0) {
108
+ $(el).replaceWith(`<p>${alt.trim()}</p>`);
109
+ }
110
+ });
111
+ }
112
+ /**
113
+ * Extract text, inserting newlines at block boundaries.
114
+ *
115
+ * cheerio's `.text()` concatenates without separators, so `<p>a</p><p>b</p>` becomes
116
+ * "ab" — which destroys the paragraph breaks segmenters depend on for boundaries.
117
+ */
118
+ extractText($, scope) {
119
+ $(scope).find('p, div, section, article, li, tr, h1, h2, h3, h4, h5, h6, blockquote, pre').each((_i, el) => {
120
+ $(el).append('\n\n');
121
+ });
122
+ $(scope).find('br').replaceWith('\n');
123
+ return $(scope).text();
124
+ }
125
+ };
126
+ HtmlContentCleaner = __decorate([
127
+ RegisterClass(BaseContentCleaner, HTML_CONTENT_CLEANER_KEY)
128
+ ], HtmlContentCleaner);
129
+ export { HtmlContentCleaner };
130
+ //# sourceMappingURL=HtmlContentCleaner.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"HtmlContentCleaner.js","sourceRoot":"","sources":["../../src/generic/HtmlContentCleaner.ts"],"names":[],"mappings":"AAAA;;;;GAIG;;;;;;;AAEH,OAAO,KAAK,OAAO,MAAM,SAAS,CAAC;AACnC,OAAO,EAAE,aAAa,EAAE,MAAM,wBAAwB,CAAC;AACvD,OAAO,EAAE,kBAAkB,EAAiD,MAAM,sBAAsB,CAAC;AAEzG,uDAAuD;AACvD,MAAM,CAAC,MAAM,wBAAwB,GAAG,MAAM,CAAC;AAE/C;;;GAGG;AACH,MAAM,CAAC,MAAM,8BAA8B,GAAG;IAC1C,QAAQ;IACR,OAAO;IACP,UAAU;IACV,QAAQ;IACR,KAAK;IACL,KAAK;IACL,QAAQ;IACR,QAAQ;IACR,OAAO;IACP,MAAM;IACN,sBAAsB;IACtB,SAAS;CACZ,CAAC;AAaF;;;;;;;;;;;;;;;;GAgBG;AAEI,IAAM,kBAAkB,GAAxB,MAAM,kBAAmB,SAAQ,kBAAkB;IACtD,IAAW,GAAG;QACV,OAAO,wBAAwB,CAAC;IACpC,CAAC;IAES,SAAS,CAAC,MAAyD;QACzE,MAAM,OAAO,GAAG,MAAM,CAAC,OAAO,CAAC;QAC/B,MAAM,CAAC,GAAG,OAAO,CAAC,IAAI,CAAC,MAAM,CAAC,OAAO,CAAC,CAAC;QAEvC,IAAI,CAAC,cAAc,CAAC,CAAC,EAAE,OAAO,CAAC,CAAC;QAChC,MAAM,KAAK,GAAG,IAAI,CAAC,YAAY,CAAC,CAAC,EAAE,OAAO,CAAC,CAAC;QAC5C,IAAI,CAAC,mBAAmB,CAAC,CAAC,EAAE,KAAK,EAAE,OAAO,CAAC,CAAC;QAE5C,OAAO,IAAI,CAAC,WAAW,CAAC,CAAC,EAAE,KAAK,CAAC,CAAC;IACtC,CAAC;IAED,gDAAgD;IACxC,cAAc,CAClB,CAAqB,EACrB,OAAoC;QAEpC,MAAM,SAAS,GAAG,OAAO,EAAE,sBAAsB;YAC7C,CAAC,CAAC,OAAO,EAAE,gBAAgB,IAAI,EAAE;YACjC,CAAC,CAAC,CAAC,GAAG,8BAA8B,EAAE,GAAG,CAAC,OAAO,EAAE,gBAAgB,IAAI,EAAE,CAAC,CAAC,CAAC;QAEhF,KAAK,MAAM,QAAQ,IAAI,SAAS,EAAE,CAAC;YAC/B,IAAI,CAAC;gBACD,CAAC,CAAC,QAAQ,CAAC,CAAC,MAAM,EAAE,CAAC;YACzB,CAAC;YAAC,MAAM,CAAC;gBACL,6EAA6E;gBAC7E,wDAAwD;YAC5D,CAAC;QACL,CAAC;IACL,CAAC;IAED;;;OAGG;IACK,YAAY,CAChB,CAAqB,EACrB,OAAoC;QAEpC,MAAM,QAAQ,GAAG,OAAO,EAAE,gBAAgB,IAAI,EAAE,CAAC;QACjD,KAAK,MAAM,QAAQ,IAAI,QAAQ,EAAE,CAAC;YAC9B,IAAI,CAAC;gBACD,MAAM,OAAO,GAAG,CAAC,CAAC,QAAQ,CAAC,CAAC;gBAC5B,IAAI,OAAO,CAAC,MAAM,GAAG,CAAC,EAAE,CAAC;oBACrB,OAAO,OAA4C,CAAC;gBACxD,CAAC;YACL,CAAC;YAAC,MAAM,CAAC;gBACL,2DAA2D;YAC/D,CAAC;QACL,CAAC;QACD,MAAM,IAAI,GAAG,CAAC,CAAC,MAAM,CAAC,CAAC;QACvB,MAAM,KAAK,GAAG,IAAI,CAAC,MAAM,GAAG,CAAC,CAAC,CAAC,CAAC,IAAI,CAAC,CAAC,CAAC,CAAC,CAAC,IAAI,EAAE,CAAC;QAChD,OAAO,KAA0C,CAAC;IACtD,CAAC;IAED,+EAA+E;IACvE,mBAAmB,CACvB,CAAqB,EACrB,KAA6B,EAC7B,OAAoC;QAEpC,IAAI,CAAC,OAAO,EAAE,mBAAmB,EAAE,CAAC;YAChC,OAAO;QACX,CAAC;QACD,CAAC,CAAC,KAAK,CAAC,CAAC,IAAI,CAAC,KAAK,CAAC,CAAC,IAAI,CAAC,CAAC,EAAE,EAAE,EAAE,EAAE,EAAE;YACjC,MAAM,GAAG,GAAG,CAAC,CAAC,EAAE,CAAC,CAAC,IAAI,CAAC,KAAK,CAAC,CAAC;YAC9B,IAAI,GAAG,IAAI,GAAG,CAAC,IAAI,EAAE,CAAC,MAAM,GAAG,CAAC,EAAE,CAAC;gBAC/B,CAAC,CAAC,EAAE,CAAC,CAAC,WAAW,CAAC,MAAM,GAAG,CAAC,IAAI,EAAE,MAAM,CAAC,CAAC;YAC9C,CAAC;QACL,CAAC,CAAC,CAAC;IACP,CAAC;IAED;;;;;OAKG;IACK,WAAW,CAAC,CAAqB,EAAE,KAA6B;QACpE,CAAC,CAAC,KAAK,CAAC,CAAC,IAAI,CAAC,2EAA2E,CAAC,CAAC,IAAI,CAAC,CAAC,EAAE,EAAE,EAAE,EAAE,EAAE;YACvG,CAAC,CAAC,EAAE,CAAC,CAAC,MAAM,CAAC,MAAM,CAAC,CAAC;QACzB,CAAC,CAAC,CAAC;QACH,CAAC,CAAC,KAAK,CAAC,CAAC,IAAI,CAAC,IAAI,CAAC,CAAC,WAAW,CAAC,IAAI,CAAC,CAAC;QAEtC,OAAO,CAAC,CAAC,KAAK,CAAC,CAAC,IAAI,EAAE,CAAC;IAC3B,CAAC;CACJ,CAAA;AA1FY,kBAAkB;IAD9B,aAAa,CAAC,kBAAkB,EAAE,wBAAwB,CAAC;GAC/C,kBAAkB,CA0F9B"}
@@ -0,0 +1,55 @@
1
+ /**
2
+ * @fileoverview Page-per-segment segmenter for paginated sources (PDF, slide decks).
3
+ *
4
+ * @module @memberjunction/ai-segmentation
5
+ */
6
+ import { BaseSegmenter } from './BaseSegmenter.js';
7
+ import { ContentModality, RawSegment, SegmentationOptions, SegmentationParams } from './Segmentation.types.js';
8
+ /** Registration key for {@link PagedContentSegmenter}. */
9
+ export declare const PAGED_CONTENT_SEGMENTER_KEY = "PagedContent";
10
+ /** Options specific to {@link PagedContentSegmenter}. */
11
+ export interface PagedContentSegmentationOptions extends SegmentationOptions {
12
+ /**
13
+ * Merge consecutive pages until the token target is reached, instead of emitting one
14
+ * segment per page. Useful for documents with very short pages (slide decks) where a
15
+ * single slide is too little context to retrieve on. Default: false.
16
+ */
17
+ MergeSmallPages?: boolean;
18
+ /** Prefix each segment's text with a "Page N" label. Default: false. */
19
+ IncludePageLabel?: boolean;
20
+ }
21
+ /**
22
+ * Emits one segment per page of a paginated source, preserving `PageNumber`.
23
+ *
24
+ * Page boundaries are authored boundaries — an author decided where the page broke — which
25
+ * makes them a better split point than any inferred one, and they give citation-grade
26
+ * provenance: a retrieved chunk resolves to "page 14 of this PDF" rather than a character
27
+ * offset nobody can act on.
28
+ *
29
+ * Each page may carry text, a media reference, or both. The media case is what enables
30
+ * embedding a PDF page *as an image* with a multimodal model — preserving tables, charts,
31
+ * and layout that text extraction flattens or loses entirely — while the extracted text
32
+ * rides along for lexical search and agent reasoning.
33
+ *
34
+ * Pages are supplied by the caller via `SegmentationParams.Pages`; this segmenter does no
35
+ * PDF parsing itself, keeping document-format dependencies out of the segmentation layer.
36
+ */
37
+ export declare class PagedContentSegmenter extends BaseSegmenter {
38
+ get Key(): string;
39
+ get SupportedModalities(): ContentModality[];
40
+ protected SegmentCore(params: SegmentationParams<PagedContentSegmentationOptions>): RawSegment[];
41
+ /** True when a page carries text or media. */
42
+ private hasPageContent;
43
+ /** One page becomes one segment. */
44
+ private pageToSegment;
45
+ /**
46
+ * Merge consecutive TEXT-ONLY pages up to the token target.
47
+ *
48
+ * A page carrying media is never merged: its media reference identifies one page, so
49
+ * folding another page's text into it would make the segment's provenance a lie.
50
+ */
51
+ private mergePages;
52
+ /** Text pages are text; a page with media is an image, or multimodal when it has both. */
53
+ private resolveModality;
54
+ }
55
+ //# sourceMappingURL=PagedContentSegmenter.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"PagedContentSegmenter.d.ts","sourceRoot":"","sources":["../../src/generic/PagedContentSegmenter.ts"],"names":[],"mappings":"AAAA;;;;GAIG;AAGH,OAAO,EAAE,aAAa,EAAE,MAAM,iBAAiB,CAAC;AAChD,OAAO,EAAE,eAAe,EAAe,UAAU,EAAE,mBAAmB,EAAE,kBAAkB,EAAE,MAAM,sBAAsB,CAAC;AAEzH,0DAA0D;AAC1D,eAAO,MAAM,2BAA2B,iBAAiB,CAAC;AAE1D,yDAAyD;AACzD,MAAM,WAAW,+BAAgC,SAAQ,mBAAmB;IACxE;;;;OAIG;IACH,eAAe,CAAC,EAAE,OAAO,CAAC;IAC1B,wEAAwE;IACxE,gBAAgB,CAAC,EAAE,OAAO,CAAC;CAC9B;AAED;;;;;;;;;;;;;;;GAeG;AACH,qBACa,qBAAsB,SAAQ,aAAa;IACpD,IAAW,GAAG,IAAI,MAAM,CAEvB;IAED,IAAW,mBAAmB,IAAI,eAAe,EAAE,CAElD;IAED,SAAS,CAAC,WAAW,CAAC,MAAM,EAAE,kBAAkB,CAAC,+BAA+B,CAAC,GAAG,UAAU,EAAE;IAWhG,8CAA8C;IAC9C,OAAO,CAAC,cAAc;IAItB,oCAAoC;IACpC,OAAO,CAAC,aAAa;IAWrB;;;;;OAKG;IACH,OAAO,CAAC,UAAU;IAoBlB,0FAA0F;IAC1F,OAAO,CAAC,eAAe;CAM1B"}
@@ -0,0 +1,99 @@
1
+ /**
2
+ * @fileoverview Page-per-segment segmenter for paginated sources (PDF, slide decks).
3
+ *
4
+ * @module @memberjunction/ai-segmentation
5
+ */
6
+ var __decorate = (this && this.__decorate) || function (decorators, target, key, desc) {
7
+ var c = arguments.length, r = c < 3 ? target : desc === null ? desc = Object.getOwnPropertyDescriptor(target, key) : desc, d;
8
+ if (typeof Reflect === "object" && typeof Reflect.decorate === "function") r = Reflect.decorate(decorators, target, key, desc);
9
+ else for (var i = decorators.length - 1; i >= 0; i--) if (d = decorators[i]) r = (c < 3 ? d(r) : c > 3 ? d(target, key, r) : d(target, key)) || r;
10
+ return c > 3 && r && Object.defineProperty(target, key, r), r;
11
+ };
12
+ import { RegisterClass } from '@memberjunction/global';
13
+ import { BaseSegmenter } from './BaseSegmenter.js';
14
+ /** Registration key for {@link PagedContentSegmenter}. */
15
+ export const PAGED_CONTENT_SEGMENTER_KEY = 'PagedContent';
16
+ /**
17
+ * Emits one segment per page of a paginated source, preserving `PageNumber`.
18
+ *
19
+ * Page boundaries are authored boundaries — an author decided where the page broke — which
20
+ * makes them a better split point than any inferred one, and they give citation-grade
21
+ * provenance: a retrieved chunk resolves to "page 14 of this PDF" rather than a character
22
+ * offset nobody can act on.
23
+ *
24
+ * Each page may carry text, a media reference, or both. The media case is what enables
25
+ * embedding a PDF page *as an image* with a multimodal model — preserving tables, charts,
26
+ * and layout that text extraction flattens or loses entirely — while the extracted text
27
+ * rides along for lexical search and agent reasoning.
28
+ *
29
+ * Pages are supplied by the caller via `SegmentationParams.Pages`; this segmenter does no
30
+ * PDF parsing itself, keeping document-format dependencies out of the segmentation layer.
31
+ */
32
+ let PagedContentSegmenter = class PagedContentSegmenter extends BaseSegmenter {
33
+ get Key() {
34
+ return PAGED_CONTENT_SEGMENTER_KEY;
35
+ }
36
+ get SupportedModalities() {
37
+ return ['text', 'image', 'multimodal'];
38
+ }
39
+ SegmentCore(params) {
40
+ const pages = (params.Pages ?? []).filter((p) => this.hasPageContent(p));
41
+ if (pages.length === 0) {
42
+ return [];
43
+ }
44
+ const ordered = [...pages].sort((a, b) => a.PageNumber - b.PageNumber);
45
+ return params.Options?.MergeSmallPages
46
+ ? this.mergePages(ordered, params)
47
+ : ordered.map((page) => this.pageToSegment(page, params.Options));
48
+ }
49
+ /** True when a page carries text or media. */
50
+ hasPageContent(page) {
51
+ return (!!page.Text && page.Text.trim().length > 0) || !!page.Media;
52
+ }
53
+ /** One page becomes one segment. */
54
+ pageToSegment(page, options) {
55
+ const label = options?.IncludePageLabel ? `Page ${page.PageNumber}\n` : '';
56
+ const text = page.Text?.trim();
57
+ return {
58
+ Modality: this.resolveModality(page),
59
+ Text: text ? `${label}${text}` : undefined,
60
+ Media: page.Media,
61
+ PageNumber: page.PageNumber,
62
+ };
63
+ }
64
+ /**
65
+ * Merge consecutive TEXT-ONLY pages up to the token target.
66
+ *
67
+ * A page carrying media is never merged: its media reference identifies one page, so
68
+ * folding another page's text into it would make the segment's provenance a lie.
69
+ */
70
+ mergePages(pages, params) {
71
+ const maxTokens = this.resolveOptions(params.Options).MaxSegmentTokens;
72
+ const segments = [];
73
+ for (const page of pages) {
74
+ const candidate = this.pageToSegment(page, params.Options);
75
+ const previous = segments[segments.length - 1];
76
+ const mergeable = previous && !previous.Media && !candidate.Media &&
77
+ this.tokensOf(previous) + this.tokensOf(candidate) <= maxTokens;
78
+ if (mergeable) {
79
+ previous.Text = `${previous.Text ?? ''}\n\n${candidate.Text ?? ''}`.trim();
80
+ }
81
+ else {
82
+ segments.push(candidate);
83
+ }
84
+ }
85
+ return segments;
86
+ }
87
+ /** Text pages are text; a page with media is an image, or multimodal when it has both. */
88
+ resolveModality(page) {
89
+ if (!page.Media) {
90
+ return 'text';
91
+ }
92
+ return page.Text && page.Text.trim().length > 0 ? 'multimodal' : 'image';
93
+ }
94
+ };
95
+ PagedContentSegmenter = __decorate([
96
+ RegisterClass(BaseSegmenter, PAGED_CONTENT_SEGMENTER_KEY)
97
+ ], PagedContentSegmenter);
98
+ export { PagedContentSegmenter };
99
+ //# sourceMappingURL=PagedContentSegmenter.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"PagedContentSegmenter.js","sourceRoot":"","sources":["../../src/generic/PagedContentSegmenter.ts"],"names":[],"mappings":"AAAA;;;;GAIG;;;;;;;AAEH,OAAO,EAAE,aAAa,EAAE,MAAM,wBAAwB,CAAC;AACvD,OAAO,EAAE,aAAa,EAAE,MAAM,iBAAiB,CAAC;AAGhD,0DAA0D;AAC1D,MAAM,CAAC,MAAM,2BAA2B,GAAG,cAAc,CAAC;AAc1D;;;;;;;;;;;;;;;GAeG;AAEI,IAAM,qBAAqB,GAA3B,MAAM,qBAAsB,SAAQ,aAAa;IACpD,IAAW,GAAG;QACV,OAAO,2BAA2B,CAAC;IACvC,CAAC;IAED,IAAW,mBAAmB;QAC1B,OAAO,CAAC,MAAM,EAAE,OAAO,EAAE,YAAY,CAAC,CAAC;IAC3C,CAAC;IAES,WAAW,CAAC,MAA2D;QAC7E,MAAM,KAAK,GAAG,CAAC,MAAM,CAAC,KAAK,IAAI,EAAE,CAAC,CAAC,MAAM,CAAC,CAAC,CAAC,EAAE,EAAE,CAAC,IAAI,CAAC,cAAc,CAAC,CAAC,CAAC,CAAC,CAAC;QACzE,IAAI,KAAK,CAAC,MAAM,KAAK,CAAC,EAAE,CAAC;YACrB,OAAO,EAAE,CAAC;QACd,CAAC;QACD,MAAM,OAAO,GAAG,CAAC,GAAG,KAAK,CAAC,CAAC,IAAI,CAAC,CAAC,CAAC,EAAE,CAAC,EAAE,EAAE,CAAC,CAAC,CAAC,UAAU,GAAG,CAAC,CAAC,UAAU,CAAC,CAAC;QACvE,OAAO,MAAM,CAAC,OAAO,EAAE,eAAe;YAClC,CAAC,CAAC,IAAI,CAAC,UAAU,CAAC,OAAO,EAAE,MAAM,CAAC;YAClC,CAAC,CAAC,OAAO,CAAC,GAAG,CAAC,CAAC,IAAI,EAAE,EAAE,CAAC,IAAI,CAAC,aAAa,CAAC,IAAI,EAAE,MAAM,CAAC,OAAO,CAAC,CAAC,CAAC;IAC1E,CAAC;IAED,8CAA8C;IACtC,cAAc,CAAC,IAAiB;QACpC,OAAO,CAAC,CAAC,CAAC,IAAI,CAAC,IAAI,IAAI,IAAI,CAAC,IAAI,CAAC,IAAI,EAAE,CAAC,MAAM,GAAG,CAAC,CAAC,IAAI,CAAC,CAAC,IAAI,CAAC,KAAK,CAAC;IACxE,CAAC;IAED,oCAAoC;IAC5B,aAAa,CAAC,IAAiB,EAAE,OAAyC;QAC9E,MAAM,KAAK,GAAG,OAAO,EAAE,gBAAgB,CAAC,CAAC,CAAC,QAAQ,IAAI,CAAC,UAAU,IAAI,CAAC,CAAC,CAAC,EAAE,CAAC;QAC3E,MAAM,IAAI,GAAG,IAAI,CAAC,IAAI,EAAE,IAAI,EAAE,CAAC;QAC/B,OAAO;YACH,QAAQ,EAAE,IAAI,CAAC,eAAe,CAAC,IAAI,CAAC;YACpC,IAAI,EAAE,IAAI,CAAC,CAAC,CAAC,GAAG,KAAK,GAAG,IAAI,EAAE,CAAC,CAAC,CAAC,SAAS;YAC1C,KAAK,EAAE,IAAI,CAAC,KAAK;YACjB,UAAU,EAAE,IAAI,CAAC,UAAU;SAC9B,CAAC;IACN,CAAC;IAED;;;;;OAKG;IACK,UAAU,CAAC,KAAoB,EAAE,MAA2D;QAChG,MAAM,SAAS,GAAG,IAAI,CAAC,cAAc,CAAC,MAAM,CAAC,OAAO,CAAC,CAAC,gBAAgB,CAAC;QACvE,MAAM,QAAQ,GAAiB,EAAE,CAAC;QAElC,KAAK,MAAM,IAAI,IAAI,KAAK,EAAE,CAAC;YACvB,MAAM,SAAS,GAAG,IAAI,CAAC,aAAa,CAAC,IAAI,EAAE,MAAM,CAAC,OAAO,CAAC,CAAC;YAC3D,MAAM,QAAQ,GAAG,QAAQ,CAAC,QAAQ,CAAC,MAAM,GAAG,CAAC,CAAC,CAAC;YAC/C,MAAM,SAAS,GACX,QAAQ,IAAI,CAAC,QAAQ,CAAC,KAAK,IAAI,CAAC,SAAS,CAAC,KAAK;gBAC/C,IAAI,CAAC,QAAQ,CAAC,QAAQ,CAAC,GAAG,IAAI,CAAC,QAAQ,CAAC,SAAS,CAAC,IAAI,SAAS,CAAC;YAEpE,IAAI,SAAS,EAAE,CAAC;gBACZ,QAAQ,CAAC,IAAI,GAAG,GAAG,QAAQ,CAAC,IAAI,IAAI,EAAE,OAAO,SAAS,CAAC,IAAI,IAAI,EAAE,EAAE,CAAC,IAAI,EAAE,CAAC;YAC/E,CAAC;iBAAM,CAAC;gBACJ,QAAQ,CAAC,IAAI,CAAC,SAAS,CAAC,CAAC;YAC7B,CAAC;QACL,CAAC;QACD,OAAO,QAAQ,CAAC;IACpB,CAAC;IAED,0FAA0F;IAClF,eAAe,CAAC,IAAiB;QACrC,IAAI,CAAC,IAAI,CAAC,KAAK,EAAE,CAAC;YACd,OAAO,MAAM,CAAC;QAClB,CAAC;QACD,OAAO,IAAI,CAAC,IAAI,IAAI,IAAI,CAAC,IAAI,CAAC,IAAI,EAAE,CAAC,MAAM,GAAG,CAAC,CAAC,CAAC,CAAC,YAAY,CAAC,CAAC,CAAC,OAAO,CAAC;IAC7E,CAAC;CACJ,CAAA;AAtEY,qBAAqB;IADjC,aAAa,CAAC,aAAa,EAAE,2BAA2B,CAAC;GAC7C,qBAAqB,CAsEjC"}
@@ -0,0 +1,22 @@
1
+ /**
2
+ * @fileoverview Pass-through cleaner for content that is already plain text.
3
+ *
4
+ * @module @memberjunction/ai-segmentation
5
+ */
6
+ import { BaseContentCleaner, ContentCleaningParams } from './BaseContentCleaner.js';
7
+ /** Registration key for {@link PlainTextContentCleaner}. */
8
+ export declare const PLAIN_TEXT_CONTENT_CLEANER_KEY = "PlainText";
9
+ /**
10
+ * Applies only the shared cleaning rules — whitespace normalization and optional
11
+ * truncation — leaving the text otherwise untouched.
12
+ *
13
+ * This is the safe default for sources that are already plain text (transcripts, extracted
14
+ * PDF/DOCX text, markdown). It still earns its place in the pipeline: extracted text is
15
+ * routinely full of ragged spacing and stray blank lines from the extractor, and those
16
+ * confuse the paragraph-boundary detection that segmenters rely on.
17
+ */
18
+ export declare class PlainTextContentCleaner extends BaseContentCleaner {
19
+ get Key(): string;
20
+ protected CleanCore(params: ContentCleaningParams): string;
21
+ }
22
+ //# sourceMappingURL=PlainTextContentCleaner.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"PlainTextContentCleaner.d.ts","sourceRoot":"","sources":["../../src/generic/PlainTextContentCleaner.ts"],"names":[],"mappings":"AAAA;;;;GAIG;AAGH,OAAO,EAAE,kBAAkB,EAAE,qBAAqB,EAAE,MAAM,sBAAsB,CAAC;AAEjF,4DAA4D;AAC5D,eAAO,MAAM,8BAA8B,cAAc,CAAC;AAE1D;;;;;;;;GAQG;AACH,qBACa,uBAAwB,SAAQ,kBAAkB;IAC3D,IAAW,GAAG,IAAI,MAAM,CAEvB;IAED,SAAS,CAAC,SAAS,CAAC,MAAM,EAAE,qBAAqB,GAAG,MAAM;CAG7D"}
@@ -0,0 +1,37 @@
1
+ /**
2
+ * @fileoverview Pass-through cleaner for content that is already plain text.
3
+ *
4
+ * @module @memberjunction/ai-segmentation
5
+ */
6
+ var __decorate = (this && this.__decorate) || function (decorators, target, key, desc) {
7
+ var c = arguments.length, r = c < 3 ? target : desc === null ? desc = Object.getOwnPropertyDescriptor(target, key) : desc, d;
8
+ if (typeof Reflect === "object" && typeof Reflect.decorate === "function") r = Reflect.decorate(decorators, target, key, desc);
9
+ else for (var i = decorators.length - 1; i >= 0; i--) if (d = decorators[i]) r = (c < 3 ? d(r) : c > 3 ? d(target, key, r) : d(target, key)) || r;
10
+ return c > 3 && r && Object.defineProperty(target, key, r), r;
11
+ };
12
+ import { RegisterClass } from '@memberjunction/global';
13
+ import { BaseContentCleaner } from './BaseContentCleaner.js';
14
+ /** Registration key for {@link PlainTextContentCleaner}. */
15
+ export const PLAIN_TEXT_CONTENT_CLEANER_KEY = 'PlainText';
16
+ /**
17
+ * Applies only the shared cleaning rules — whitespace normalization and optional
18
+ * truncation — leaving the text otherwise untouched.
19
+ *
20
+ * This is the safe default for sources that are already plain text (transcripts, extracted
21
+ * PDF/DOCX text, markdown). It still earns its place in the pipeline: extracted text is
22
+ * routinely full of ragged spacing and stray blank lines from the extractor, and those
23
+ * confuse the paragraph-boundary detection that segmenters rely on.
24
+ */
25
+ let PlainTextContentCleaner = class PlainTextContentCleaner extends BaseContentCleaner {
26
+ get Key() {
27
+ return PLAIN_TEXT_CONTENT_CLEANER_KEY;
28
+ }
29
+ CleanCore(params) {
30
+ return params.Content;
31
+ }
32
+ };
33
+ PlainTextContentCleaner = __decorate([
34
+ RegisterClass(BaseContentCleaner, PLAIN_TEXT_CONTENT_CLEANER_KEY)
35
+ ], PlainTextContentCleaner);
36
+ export { PlainTextContentCleaner };
37
+ //# sourceMappingURL=PlainTextContentCleaner.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"PlainTextContentCleaner.js","sourceRoot":"","sources":["../../src/generic/PlainTextContentCleaner.ts"],"names":[],"mappings":"AAAA;;;;GAIG;;;;;;;AAEH,OAAO,EAAE,aAAa,EAAE,MAAM,wBAAwB,CAAC;AACvD,OAAO,EAAE,kBAAkB,EAAyB,MAAM,sBAAsB,CAAC;AAEjF,4DAA4D;AAC5D,MAAM,CAAC,MAAM,8BAA8B,GAAG,WAAW,CAAC;AAE1D;;;;;;;;GAQG;AAEI,IAAM,uBAAuB,GAA7B,MAAM,uBAAwB,SAAQ,kBAAkB;IAC3D,IAAW,GAAG;QACV,OAAO,8BAA8B,CAAC;IAC1C,CAAC;IAES,SAAS,CAAC,MAA6B;QAC7C,OAAO,MAAM,CAAC,OAAO,CAAC;IAC1B,CAAC;CACJ,CAAA;AARY,uBAAuB;IADnC,aAAa,CAAC,kBAAkB,EAAE,8BAA8B,CAAC;GACrD,uBAAuB,CAQnC"}