@memberjunction/ai-segmentation 0.0.0 → 5.51.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +177 -43
- package/dist/generic/AdaptiveBoundarySegmenter.d.ts +98 -0
- package/dist/generic/AdaptiveBoundarySegmenter.d.ts.map +1 -0
- package/dist/generic/AdaptiveBoundarySegmenter.js +177 -0
- package/dist/generic/AdaptiveBoundarySegmenter.js.map +1 -0
- package/dist/generic/BaseContentCleaner.d.ts +101 -0
- package/dist/generic/BaseContentCleaner.d.ts.map +1 -0
- package/dist/generic/BaseContentCleaner.js +114 -0
- package/dist/generic/BaseContentCleaner.js.map +1 -0
- package/dist/generic/BaseSegmenter.d.ts +106 -0
- package/dist/generic/BaseSegmenter.d.ts.map +1 -0
- package/dist/generic/BaseSegmenter.js +260 -0
- package/dist/generic/BaseSegmenter.js.map +1 -0
- package/dist/generic/FixedWindowSegmenter.d.ts +49 -0
- package/dist/generic/FixedWindowSegmenter.d.ts.map +1 -0
- package/dist/generic/FixedWindowSegmenter.js +106 -0
- package/dist/generic/FixedWindowSegmenter.js.map +1 -0
- package/dist/generic/HtmlContentCleaner.d.ts +61 -0
- package/dist/generic/HtmlContentCleaner.d.ts.map +1 -0
- package/dist/generic/HtmlContentCleaner.js +130 -0
- package/dist/generic/HtmlContentCleaner.js.map +1 -0
- package/dist/generic/PagedContentSegmenter.d.ts +55 -0
- package/dist/generic/PagedContentSegmenter.d.ts.map +1 -0
- package/dist/generic/PagedContentSegmenter.js +99 -0
- package/dist/generic/PagedContentSegmenter.js.map +1 -0
- package/dist/generic/PlainTextContentCleaner.d.ts +22 -0
- package/dist/generic/PlainTextContentCleaner.d.ts.map +1 -0
- package/dist/generic/PlainTextContentCleaner.js +37 -0
- package/dist/generic/PlainTextContentCleaner.js.map +1 -0
- package/dist/generic/Segmentation.types.d.ts +198 -0
- package/dist/generic/Segmentation.types.d.ts.map +1 -0
- package/dist/generic/Segmentation.types.js +19 -0
- package/dist/generic/Segmentation.types.js.map +1 -0
- package/dist/generic/SegmentationResolver.d.ts +43 -0
- package/dist/generic/SegmentationResolver.d.ts.map +1 -0
- package/dist/generic/SegmentationResolver.js +83 -0
- package/dist/generic/SegmentationResolver.js.map +1 -0
- package/dist/generic/SemanticTextSegmenter.d.ts +80 -0
- package/dist/generic/SemanticTextSegmenter.d.ts.map +1 -0
- package/dist/generic/SemanticTextSegmenter.js +201 -0
- package/dist/generic/SemanticTextSegmenter.js.map +1 -0
- package/dist/generic/StructuralTextSegmenter.d.ts +63 -0
- package/dist/generic/StructuralTextSegmenter.d.ts.map +1 -0
- package/dist/generic/StructuralTextSegmenter.js +177 -0
- package/dist/generic/StructuralTextSegmenter.js.map +1 -0
- package/dist/generic/TranscriptSegmenter.d.ts +78 -0
- package/dist/generic/TranscriptSegmenter.d.ts.map +1 -0
- package/dist/generic/TranscriptSegmenter.js +194 -0
- package/dist/generic/TranscriptSegmenter.js.map +1 -0
- package/dist/index.d.ts +33 -0
- package/dist/index.d.ts.map +1 -0
- package/dist/index.js +36 -0
- package/dist/index.js.map +1 -0
- package/package.json +33 -7
|
@@ -0,0 +1,106 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview Fixed-window segmenter — the universal fallback strategy.
|
|
3
|
+
*
|
|
4
|
+
* @module @memberjunction/ai-segmentation
|
|
5
|
+
*/
|
|
6
|
+
var __decorate = (this && this.__decorate) || function (decorators, target, key, desc) {
|
|
7
|
+
var c = arguments.length, r = c < 3 ? target : desc === null ? desc = Object.getOwnPropertyDescriptor(target, key) : desc, d;
|
|
8
|
+
if (typeof Reflect === "object" && typeof Reflect.decorate === "function") r = Reflect.decorate(decorators, target, key, desc);
|
|
9
|
+
else for (var i = decorators.length - 1; i >= 0; i--) if (d = decorators[i]) r = (c < 3 ? d(r) : c > 3 ? d(target, key, r) : d(target, key)) || r;
|
|
10
|
+
return c > 3 && r && Object.defineProperty(target, key, r), r;
|
|
11
|
+
};
|
|
12
|
+
import { RegisterClass } from '@memberjunction/global';
|
|
13
|
+
import { BaseSegmenter } from './BaseSegmenter.js';
|
|
14
|
+
import { TextChunker } from '@memberjunction/ai-vectors';
|
|
15
|
+
/** Registration key for {@link FixedWindowSegmenter}. */
|
|
16
|
+
export const FIXED_WINDOW_SEGMENTER_KEY = 'FixedWindow';
|
|
17
|
+
/**
|
|
18
|
+
* Splits content into uniform windows: token-bounded chunks for text, and
|
|
19
|
+
* fixed-duration windows for audio/video that has no transcript.
|
|
20
|
+
*
|
|
21
|
+
* This is the safety net, not the recommended default. It requires no LLM call,
|
|
22
|
+
* no transcript, and no document structure, so it always produces *something* —
|
|
23
|
+
* which makes it the right choice for logs, machine-generated text, and media
|
|
24
|
+
* that arrives without cues. Where structure or a transcript exists, prefer
|
|
25
|
+
* `StructuralText` or `Transcript`, both of which cut on real boundaries.
|
|
26
|
+
*/
|
|
27
|
+
let FixedWindowSegmenter = class FixedWindowSegmenter extends BaseSegmenter {
|
|
28
|
+
get Key() {
|
|
29
|
+
return FIXED_WINDOW_SEGMENTER_KEY;
|
|
30
|
+
}
|
|
31
|
+
get SupportedModalities() {
|
|
32
|
+
return ['text', 'image', 'audio', 'video', 'multimodal'];
|
|
33
|
+
}
|
|
34
|
+
SegmentCore(params) {
|
|
35
|
+
if (params.Text && params.Text.trim().length > 0) {
|
|
36
|
+
return this.segmentText(params);
|
|
37
|
+
}
|
|
38
|
+
if (params.Media) {
|
|
39
|
+
return this.segmentMedia(params);
|
|
40
|
+
}
|
|
41
|
+
return [];
|
|
42
|
+
}
|
|
43
|
+
/** Token-bounded text windows, offsets preserved. */
|
|
44
|
+
segmentText(params) {
|
|
45
|
+
const settings = this.resolveOptions(params.Options);
|
|
46
|
+
const chunks = TextChunker.ChunkText({
|
|
47
|
+
Text: params.Text ?? '',
|
|
48
|
+
MaxChunkTokens: settings.MaxSegmentTokens,
|
|
49
|
+
OverlapTokens: settings.OverlapTokens,
|
|
50
|
+
Strategy: params.Options?.TextStrategy ?? 'sentence',
|
|
51
|
+
});
|
|
52
|
+
return chunks.map((chunk) => ({
|
|
53
|
+
Modality: 'text',
|
|
54
|
+
Text: chunk.Text,
|
|
55
|
+
StartOffset: chunk.StartOffset,
|
|
56
|
+
EndOffset: chunk.EndOffset,
|
|
57
|
+
}));
|
|
58
|
+
}
|
|
59
|
+
/** Fixed-duration media windows, or a single segment for untimed media. */
|
|
60
|
+
segmentMedia(params) {
|
|
61
|
+
const modality = this.resolveMediaModality(params);
|
|
62
|
+
if (modality === 'image' || !params.DurationMs || params.DurationMs <= 0) {
|
|
63
|
+
return [{ Modality: modality, Media: params.Media }];
|
|
64
|
+
}
|
|
65
|
+
return this.buildTimeWindows(params, modality);
|
|
66
|
+
}
|
|
67
|
+
/** Walk the asset duration emitting one window per step. */
|
|
68
|
+
buildTimeWindows(params, modality) {
|
|
69
|
+
const windowMs = Math.max(params.Options?.WindowMs ?? 30_000, 1);
|
|
70
|
+
// Cap overlap at half the window, matching TextChunker's rule for text. Beyond 50% each
|
|
71
|
+
// window is mostly a copy of the previous one, and as overlap approaches the window size the
|
|
72
|
+
// segment count explodes — a 1s window with 5s of overlap would emit one segment per
|
|
73
|
+
// millisecond, and every one of those is a paid multimodal embedding call.
|
|
74
|
+
const overlapMs = Math.min(params.Options?.WindowOverlapMs ?? 0, Math.floor(windowMs / 2));
|
|
75
|
+
const step = Math.max(windowMs - overlapMs, 1);
|
|
76
|
+
const duration = params.DurationMs ?? 0;
|
|
77
|
+
const segments = [];
|
|
78
|
+
for (let start = 0; start < duration; start += step) {
|
|
79
|
+
const end = Math.min(start + windowMs, duration);
|
|
80
|
+
segments.push({ Modality: modality, Media: params.Media, StartMs: start, EndMs: end });
|
|
81
|
+
if (end >= duration) {
|
|
82
|
+
break;
|
|
83
|
+
}
|
|
84
|
+
}
|
|
85
|
+
return segments;
|
|
86
|
+
}
|
|
87
|
+
/** Infer the media modality from its mime type. */
|
|
88
|
+
resolveMediaModality(params) {
|
|
89
|
+
const mime = (params.Media?.MimeType ?? params.MimeType ?? '').toLowerCase();
|
|
90
|
+
if (mime.startsWith('video/')) {
|
|
91
|
+
return 'video';
|
|
92
|
+
}
|
|
93
|
+
if (mime.startsWith('audio/')) {
|
|
94
|
+
return 'audio';
|
|
95
|
+
}
|
|
96
|
+
if (mime.startsWith('image/')) {
|
|
97
|
+
return 'image';
|
|
98
|
+
}
|
|
99
|
+
return 'multimodal';
|
|
100
|
+
}
|
|
101
|
+
};
|
|
102
|
+
FixedWindowSegmenter = __decorate([
|
|
103
|
+
RegisterClass(BaseSegmenter, FIXED_WINDOW_SEGMENTER_KEY)
|
|
104
|
+
], FixedWindowSegmenter);
|
|
105
|
+
export { FixedWindowSegmenter };
|
|
106
|
+
//# sourceMappingURL=FixedWindowSegmenter.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"FixedWindowSegmenter.js","sourceRoot":"","sources":["../../src/generic/FixedWindowSegmenter.ts"],"names":[],"mappings":"AAAA;;;;GAIG;;;;;;;AAEH,OAAO,EAAE,aAAa,EAAE,MAAM,wBAAwB,CAAC;AACvD,OAAO,EAAE,aAAa,EAAE,MAAM,iBAAiB,CAAC;AAChD,OAAO,EAAE,WAAW,EAAE,MAAM,4BAA4B,CAAC;AAGzD,yDAAyD;AACzD,MAAM,CAAC,MAAM,0BAA0B,GAAG,aAAa,CAAC;AAmBxD;;;;;;;;;GASG;AAEI,IAAM,oBAAoB,GAA1B,MAAM,oBAAqB,SAAQ,aAAa;IACnD,IAAW,GAAG;QACV,OAAO,0BAA0B,CAAC;IACtC,CAAC;IAED,IAAW,mBAAmB;QAC1B,OAAO,CAAC,MAAM,EAAE,OAAO,EAAE,OAAO,EAAE,OAAO,EAAE,YAAY,CAAC,CAAC;IAC7D,CAAC;IAES,WAAW,CAAC,MAA0D;QAC5E,IAAI,MAAM,CAAC,IAAI,IAAI,MAAM,CAAC,IAAI,CAAC,IAAI,EAAE,CAAC,MAAM,GAAG,CAAC,EAAE,CAAC;YAC/C,OAAO,IAAI,CAAC,WAAW,CAAC,MAAM,CAAC,CAAC;QACpC,CAAC;QACD,IAAI,MAAM,CAAC,KAAK,EAAE,CAAC;YACf,OAAO,IAAI,CAAC,YAAY,CAAC,MAAM,CAAC,CAAC;QACrC,CAAC;QACD,OAAO,EAAE,CAAC;IACd,CAAC;IAED,qDAAqD;IAC7C,WAAW,CAAC,MAA0D;QAC1E,MAAM,QAAQ,GAAG,IAAI,CAAC,cAAc,CAAC,MAAM,CAAC,OAAO,CAAC,CAAC;QACrD,MAAM,MAAM,GAAG,WAAW,CAAC,SAAS,CAAC;YACjC,IAAI,EAAE,MAAM,CAAC,IAAI,IAAI,EAAE;YACvB,cAAc,EAAE,QAAQ,CAAC,gBAAgB;YACzC,aAAa,EAAE,QAAQ,CAAC,aAAa;YACrC,QAAQ,EAAE,MAAM,CAAC,OAAO,EAAE,YAAY,IAAI,UAAU;SACvD,CAAC,CAAC;QACH,OAAO,MAAM,CAAC,GAAG,CAAC,CAAC,KAAK,EAAE,EAAE,CAAC,CAAC;YAC1B,QAAQ,EAAE,MAAM;YAChB,IAAI,EAAE,KAAK,CAAC,IAAI;YAChB,WAAW,EAAE,KAAK,CAAC,WAAW;YAC9B,SAAS,EAAE,KAAK,CAAC,SAAS;SAC7B,CAAC,CAAC,CAAC;IACR,CAAC;IAED,2EAA2E;IACnE,YAAY,CAAC,MAA0D;QAC3E,MAAM,QAAQ,GAAG,IAAI,CAAC,oBAAoB,CAAC,MAAM,CAAC,CAAC;QACnD,IAAI,QAAQ,KAAK,OAAO,IAAI,CAAC,MAAM,CAAC,UAAU,IAAI,MAAM,CAAC,UAAU,IAAI,CAAC,EAAE,CAAC;YACvE,OAAO,CAAC,EAAE,QAAQ,EAAE,QAAQ,EAAE,KAAK,EAAE,MAAM,CAAC,KAAK,EAAE,CAAC,CAAC;QACzD,CAAC;QACD,OAAO,IAAI,CAAC,gBAAgB,CAAC,MAAM,EAAE,QAAQ,CAAC,CAAC;IACnD,CAAC;IAED,4DAA4D;IACpD,gBAAgB,CACpB,MAA0D,EAC1D,QAAyB;QAEzB,MAAM,QAAQ,GAAG,IAAI,CAAC,GAAG,CAAC,MAAM,CAAC,OAAO,EAAE,QAAQ,IAAI,MAAM,EAAE,CAAC,CAAC,CAAC;QACjE,wFAAwF;QACxF,6FAA6F;QAC7F,qFAAqF;QACrF,2EAA2E;QAC3E,MAAM,SAAS,GAAG,IAAI,CAAC,GAAG,CAAC,MAAM,CAAC,OAAO,EAAE,eAAe,IAAI,CAAC,EAAE,IAAI,CAAC,KAAK,CAAC,QAAQ,GAAG,CAAC,CAAC,CAAC,CAAC;QAC3F,MAAM,IAAI,GAAG,IAAI,CAAC,GAAG,CAAC,QAAQ,GAAG,SAAS,EAAE,CAAC,CAAC,CAAC;QAC/C,MAAM,QAAQ,GAAG,MAAM,CAAC,UAAU,IAAI,CAAC,CAAC;QACxC,MAAM,QAAQ,GAAiB,EAAE,CAAC;QAElC,KAAK,IAAI,KAAK,GAAG,CAAC,EAAE,KAAK,GAAG,QAAQ,EAAE,KAAK,IAAI,IAAI,EAAE,CAAC;YAClD,MAAM,GAAG,GAAG,IAAI,CAAC,GAAG,CAAC,KAAK,GAAG,QAAQ,EAAE,QAAQ,CAAC,CAAC;YACjD,QAAQ,CAAC,IAAI,CAAC,EAAE,QAAQ,EAAE,QAAQ,EAAE,KAAK,EAAE,MAAM,CAAC,KAAK,EAAE,OAAO,EAAE,KAAK,EAAE,KAAK,EAAE,GAAG,EAAE,CAAC,CAAC;YACvF,IAAI,GAAG,IAAI,QAAQ,EAAE,CAAC;gBAClB,MAAM;YACV,CAAC;QACL,CAAC;QACD,OAAO,QAAQ,CAAC;IACpB,CAAC;IAED,mDAAmD;IAC3C,oBAAoB,CAAC,MAA0D;QACnF,MAAM,IAAI,GAAG,CAAC,MAAM,CAAC,KAAK,EAAE,QAAQ,IAAI,MAAM,CAAC,QAAQ,IAAI,EAAE,CAAC,CAAC,WAAW,EAAE,CAAC;QAC7E,IAAI,IAAI,CAAC,UAAU,CAAC,QAAQ,CAAC,EAAE,CAAC;YAC5B,OAAO,OAAO,CAAC;QACnB,CAAC;QACD,IAAI,IAAI,CAAC,UAAU,CAAC,QAAQ,CAAC,EAAE,CAAC;YAC5B,OAAO,OAAO,CAAC;QACnB,CAAC;QACD,IAAI,IAAI,CAAC,UAAU,CAAC,QAAQ,CAAC,EAAE,CAAC;YAC5B,OAAO,OAAO,CAAC;QACnB,CAAC;QACD,OAAO,YAAY,CAAC;IACxB,CAAC;CACJ,CAAA;AApFY,oBAAoB;IADhC,aAAa,CAAC,aAAa,EAAE,0BAA0B,CAAC;GAC5C,oBAAoB,CAoFhC"}
|
|
@@ -0,0 +1,61 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview Selector-driven HTML cleaner.
|
|
3
|
+
*
|
|
4
|
+
* @module @memberjunction/ai-segmentation
|
|
5
|
+
*/
|
|
6
|
+
import { BaseContentCleaner, ContentCleaningOptions, ContentCleaningParams } from './BaseContentCleaner.js';
|
|
7
|
+
/** Registration key for {@link HtmlContentCleaner}. */
|
|
8
|
+
export declare const HTML_CONTENT_CLEANER_KEY = "Html";
|
|
9
|
+
/**
|
|
10
|
+
* Elements removed before extraction unless the caller overrides `ExcludeSelectors`.
|
|
11
|
+
* These carry no document content on essentially any site.
|
|
12
|
+
*/
|
|
13
|
+
export declare const DEFAULT_HTML_EXCLUDE_SELECTORS: string[];
|
|
14
|
+
/** Options specific to {@link HtmlContentCleaner}. */
|
|
15
|
+
export interface HtmlContentCleaningOptions extends ContentCleaningOptions {
|
|
16
|
+
/**
|
|
17
|
+
* Replace the built-in exclusion list rather than adding to it. Default: false
|
|
18
|
+
* (caller selectors are appended to {@link DEFAULT_HTML_EXCLUDE_SELECTORS}).
|
|
19
|
+
*/
|
|
20
|
+
ReplaceDefaultExcludes?: boolean;
|
|
21
|
+
/** Keep `alt` text from images as content. Default: false. */
|
|
22
|
+
IncludeImageAltText?: boolean;
|
|
23
|
+
}
|
|
24
|
+
/**
|
|
25
|
+
* Extracts readable text from HTML using CSS selectors.
|
|
26
|
+
*
|
|
27
|
+
* Real-world source pages are mostly not content: navigation, sidebars, cookie banners,
|
|
28
|
+
* share widgets, related-article rails, and advertising typically outweigh the article
|
|
29
|
+
* itself. Stripping tags alone keeps all of that text, and because chrome repeats across
|
|
30
|
+
* every page of a site it produces many near-identical chunks that crowd out real answers
|
|
31
|
+
* at retrieval time.
|
|
32
|
+
*
|
|
33
|
+
* The high-leverage control is `IncludeSelectors` — naming the one element that holds the
|
|
34
|
+
* content (`.article-body`, `main`, `#post`) discards everything else without having to
|
|
35
|
+
* enumerate what to drop. `ExcludeSelectors` then handles whatever survives inside it.
|
|
36
|
+
*
|
|
37
|
+
* Both are per-source configuration because the right selector is a property of the site's
|
|
38
|
+
* template, not of MemberJunction. A sensible default exclusion list handles sources that
|
|
39
|
+
* haven't been tuned yet.
|
|
40
|
+
*/
|
|
41
|
+
export declare class HtmlContentCleaner extends BaseContentCleaner {
|
|
42
|
+
get Key(): string;
|
|
43
|
+
protected CleanCore(params: ContentCleaningParams<HtmlContentCleaningOptions>): string;
|
|
44
|
+
/** Drop excluded elements from the document. */
|
|
45
|
+
private removeExcluded;
|
|
46
|
+
/**
|
|
47
|
+
* Narrow to the included selectors when supplied, else the body.
|
|
48
|
+
* Returns a selection whose text is the document's content.
|
|
49
|
+
*/
|
|
50
|
+
private resolveScope;
|
|
51
|
+
/** Append image alt text so meaningful figures aren't lost, when requested. */
|
|
52
|
+
private promoteImageAltText;
|
|
53
|
+
/**
|
|
54
|
+
* Extract text, inserting newlines at block boundaries.
|
|
55
|
+
*
|
|
56
|
+
* cheerio's `.text()` concatenates without separators, so `<p>a</p><p>b</p>` becomes
|
|
57
|
+
* "ab" — which destroys the paragraph breaks segmenters depend on for boundaries.
|
|
58
|
+
*/
|
|
59
|
+
private extractText;
|
|
60
|
+
}
|
|
61
|
+
//# sourceMappingURL=HtmlContentCleaner.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"HtmlContentCleaner.d.ts","sourceRoot":"","sources":["../../src/generic/HtmlContentCleaner.ts"],"names":[],"mappings":"AAAA;;;;GAIG;AAIH,OAAO,EAAE,kBAAkB,EAAE,sBAAsB,EAAE,qBAAqB,EAAE,MAAM,sBAAsB,CAAC;AAEzG,uDAAuD;AACvD,eAAO,MAAM,wBAAwB,SAAS,CAAC;AAE/C;;;GAGG;AACH,eAAO,MAAM,8BAA8B,UAa1C,CAAC;AAEF,sDAAsD;AACtD,MAAM,WAAW,0BAA2B,SAAQ,sBAAsB;IACtE;;;OAGG;IACH,sBAAsB,CAAC,EAAE,OAAO,CAAC;IACjC,8DAA8D;IAC9D,mBAAmB,CAAC,EAAE,OAAO,CAAC;CACjC;AAED;;;;;;;;;;;;;;;;GAgBG;AACH,qBACa,kBAAmB,SAAQ,kBAAkB;IACtD,IAAW,GAAG,IAAI,MAAM,CAEvB;IAED,SAAS,CAAC,SAAS,CAAC,MAAM,EAAE,qBAAqB,CAAC,0BAA0B,CAAC,GAAG,MAAM;IAWtF,gDAAgD;IAChD,OAAO,CAAC,cAAc;IAkBtB;;;OAGG;IACH,OAAO,CAAC,YAAY;IAoBpB,+EAA+E;IAC/E,OAAO,CAAC,mBAAmB;IAgB3B;;;;;OAKG;IACH,OAAO,CAAC,WAAW;CAQtB"}
|
|
@@ -0,0 +1,130 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview Selector-driven HTML cleaner.
|
|
3
|
+
*
|
|
4
|
+
* @module @memberjunction/ai-segmentation
|
|
5
|
+
*/
|
|
6
|
+
var __decorate = (this && this.__decorate) || function (decorators, target, key, desc) {
|
|
7
|
+
var c = arguments.length, r = c < 3 ? target : desc === null ? desc = Object.getOwnPropertyDescriptor(target, key) : desc, d;
|
|
8
|
+
if (typeof Reflect === "object" && typeof Reflect.decorate === "function") r = Reflect.decorate(decorators, target, key, desc);
|
|
9
|
+
else for (var i = decorators.length - 1; i >= 0; i--) if (d = decorators[i]) r = (c < 3 ? d(r) : c > 3 ? d(target, key, r) : d(target, key)) || r;
|
|
10
|
+
return c > 3 && r && Object.defineProperty(target, key, r), r;
|
|
11
|
+
};
|
|
12
|
+
import * as cheerio from 'cheerio';
|
|
13
|
+
import { RegisterClass } from '@memberjunction/global';
|
|
14
|
+
import { BaseContentCleaner } from './BaseContentCleaner.js';
|
|
15
|
+
/** Registration key for {@link HtmlContentCleaner}. */
|
|
16
|
+
export const HTML_CONTENT_CLEANER_KEY = 'Html';
|
|
17
|
+
/**
|
|
18
|
+
* Elements removed before extraction unless the caller overrides `ExcludeSelectors`.
|
|
19
|
+
* These carry no document content on essentially any site.
|
|
20
|
+
*/
|
|
21
|
+
export const DEFAULT_HTML_EXCLUDE_SELECTORS = [
|
|
22
|
+
'script',
|
|
23
|
+
'style',
|
|
24
|
+
'noscript',
|
|
25
|
+
'iframe',
|
|
26
|
+
'svg',
|
|
27
|
+
'nav',
|
|
28
|
+
'header',
|
|
29
|
+
'footer',
|
|
30
|
+
'aside',
|
|
31
|
+
'form',
|
|
32
|
+
'[aria-hidden="true"]',
|
|
33
|
+
'.hidden',
|
|
34
|
+
];
|
|
35
|
+
/**
|
|
36
|
+
* Extracts readable text from HTML using CSS selectors.
|
|
37
|
+
*
|
|
38
|
+
* Real-world source pages are mostly not content: navigation, sidebars, cookie banners,
|
|
39
|
+
* share widgets, related-article rails, and advertising typically outweigh the article
|
|
40
|
+
* itself. Stripping tags alone keeps all of that text, and because chrome repeats across
|
|
41
|
+
* every page of a site it produces many near-identical chunks that crowd out real answers
|
|
42
|
+
* at retrieval time.
|
|
43
|
+
*
|
|
44
|
+
* The high-leverage control is `IncludeSelectors` — naming the one element that holds the
|
|
45
|
+
* content (`.article-body`, `main`, `#post`) discards everything else without having to
|
|
46
|
+
* enumerate what to drop. `ExcludeSelectors` then handles whatever survives inside it.
|
|
47
|
+
*
|
|
48
|
+
* Both are per-source configuration because the right selector is a property of the site's
|
|
49
|
+
* template, not of MemberJunction. A sensible default exclusion list handles sources that
|
|
50
|
+
* haven't been tuned yet.
|
|
51
|
+
*/
|
|
52
|
+
let HtmlContentCleaner = class HtmlContentCleaner extends BaseContentCleaner {
|
|
53
|
+
get Key() {
|
|
54
|
+
return HTML_CONTENT_CLEANER_KEY;
|
|
55
|
+
}
|
|
56
|
+
CleanCore(params) {
|
|
57
|
+
const options = params.Options;
|
|
58
|
+
const $ = cheerio.load(params.Content);
|
|
59
|
+
this.removeExcluded($, options);
|
|
60
|
+
const scope = this.resolveScope($, options);
|
|
61
|
+
this.promoteImageAltText($, scope, options);
|
|
62
|
+
return this.extractText($, scope);
|
|
63
|
+
}
|
|
64
|
+
/** Drop excluded elements from the document. */
|
|
65
|
+
removeExcluded($, options) {
|
|
66
|
+
const selectors = options?.ReplaceDefaultExcludes
|
|
67
|
+
? options?.ExcludeSelectors ?? []
|
|
68
|
+
: [...DEFAULT_HTML_EXCLUDE_SELECTORS, ...(options?.ExcludeSelectors ?? [])];
|
|
69
|
+
for (const selector of selectors) {
|
|
70
|
+
try {
|
|
71
|
+
$(selector).remove();
|
|
72
|
+
}
|
|
73
|
+
catch {
|
|
74
|
+
// An invalid selector shouldn't fail the whole clean — skip it and continue,
|
|
75
|
+
// since the rest of the rules are still worth applying.
|
|
76
|
+
}
|
|
77
|
+
}
|
|
78
|
+
}
|
|
79
|
+
/**
|
|
80
|
+
* Narrow to the included selectors when supplied, else the body.
|
|
81
|
+
* Returns a selection whose text is the document's content.
|
|
82
|
+
*/
|
|
83
|
+
resolveScope($, options) {
|
|
84
|
+
const includes = options?.IncludeSelectors ?? [];
|
|
85
|
+
for (const selector of includes) {
|
|
86
|
+
try {
|
|
87
|
+
const matched = $(selector);
|
|
88
|
+
if (matched.length > 0) {
|
|
89
|
+
return matched;
|
|
90
|
+
}
|
|
91
|
+
}
|
|
92
|
+
catch {
|
|
93
|
+
// Ignore an invalid include selector and try the next one.
|
|
94
|
+
}
|
|
95
|
+
}
|
|
96
|
+
const body = $('body');
|
|
97
|
+
const scope = body.length > 0 ? body : $.root();
|
|
98
|
+
return scope;
|
|
99
|
+
}
|
|
100
|
+
/** Append image alt text so meaningful figures aren't lost, when requested. */
|
|
101
|
+
promoteImageAltText($, scope, options) {
|
|
102
|
+
if (!options?.IncludeImageAltText) {
|
|
103
|
+
return;
|
|
104
|
+
}
|
|
105
|
+
$(scope).find('img').each((_i, el) => {
|
|
106
|
+
const alt = $(el).attr('alt');
|
|
107
|
+
if (alt && alt.trim().length > 0) {
|
|
108
|
+
$(el).replaceWith(`<p>${alt.trim()}</p>`);
|
|
109
|
+
}
|
|
110
|
+
});
|
|
111
|
+
}
|
|
112
|
+
/**
|
|
113
|
+
* Extract text, inserting newlines at block boundaries.
|
|
114
|
+
*
|
|
115
|
+
* cheerio's `.text()` concatenates without separators, so `<p>a</p><p>b</p>` becomes
|
|
116
|
+
* "ab" — which destroys the paragraph breaks segmenters depend on for boundaries.
|
|
117
|
+
*/
|
|
118
|
+
extractText($, scope) {
|
|
119
|
+
$(scope).find('p, div, section, article, li, tr, h1, h2, h3, h4, h5, h6, blockquote, pre').each((_i, el) => {
|
|
120
|
+
$(el).append('\n\n');
|
|
121
|
+
});
|
|
122
|
+
$(scope).find('br').replaceWith('\n');
|
|
123
|
+
return $(scope).text();
|
|
124
|
+
}
|
|
125
|
+
};
|
|
126
|
+
HtmlContentCleaner = __decorate([
|
|
127
|
+
RegisterClass(BaseContentCleaner, HTML_CONTENT_CLEANER_KEY)
|
|
128
|
+
], HtmlContentCleaner);
|
|
129
|
+
export { HtmlContentCleaner };
|
|
130
|
+
//# sourceMappingURL=HtmlContentCleaner.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"HtmlContentCleaner.js","sourceRoot":"","sources":["../../src/generic/HtmlContentCleaner.ts"],"names":[],"mappings":"AAAA;;;;GAIG;;;;;;;AAEH,OAAO,KAAK,OAAO,MAAM,SAAS,CAAC;AACnC,OAAO,EAAE,aAAa,EAAE,MAAM,wBAAwB,CAAC;AACvD,OAAO,EAAE,kBAAkB,EAAiD,MAAM,sBAAsB,CAAC;AAEzG,uDAAuD;AACvD,MAAM,CAAC,MAAM,wBAAwB,GAAG,MAAM,CAAC;AAE/C;;;GAGG;AACH,MAAM,CAAC,MAAM,8BAA8B,GAAG;IAC1C,QAAQ;IACR,OAAO;IACP,UAAU;IACV,QAAQ;IACR,KAAK;IACL,KAAK;IACL,QAAQ;IACR,QAAQ;IACR,OAAO;IACP,MAAM;IACN,sBAAsB;IACtB,SAAS;CACZ,CAAC;AAaF;;;;;;;;;;;;;;;;GAgBG;AAEI,IAAM,kBAAkB,GAAxB,MAAM,kBAAmB,SAAQ,kBAAkB;IACtD,IAAW,GAAG;QACV,OAAO,wBAAwB,CAAC;IACpC,CAAC;IAES,SAAS,CAAC,MAAyD;QACzE,MAAM,OAAO,GAAG,MAAM,CAAC,OAAO,CAAC;QAC/B,MAAM,CAAC,GAAG,OAAO,CAAC,IAAI,CAAC,MAAM,CAAC,OAAO,CAAC,CAAC;QAEvC,IAAI,CAAC,cAAc,CAAC,CAAC,EAAE,OAAO,CAAC,CAAC;QAChC,MAAM,KAAK,GAAG,IAAI,CAAC,YAAY,CAAC,CAAC,EAAE,OAAO,CAAC,CAAC;QAC5C,IAAI,CAAC,mBAAmB,CAAC,CAAC,EAAE,KAAK,EAAE,OAAO,CAAC,CAAC;QAE5C,OAAO,IAAI,CAAC,WAAW,CAAC,CAAC,EAAE,KAAK,CAAC,CAAC;IACtC,CAAC;IAED,gDAAgD;IACxC,cAAc,CAClB,CAAqB,EACrB,OAAoC;QAEpC,MAAM,SAAS,GAAG,OAAO,EAAE,sBAAsB;YAC7C,CAAC,CAAC,OAAO,EAAE,gBAAgB,IAAI,EAAE;YACjC,CAAC,CAAC,CAAC,GAAG,8BAA8B,EAAE,GAAG,CAAC,OAAO,EAAE,gBAAgB,IAAI,EAAE,CAAC,CAAC,CAAC;QAEhF,KAAK,MAAM,QAAQ,IAAI,SAAS,EAAE,CAAC;YAC/B,IAAI,CAAC;gBACD,CAAC,CAAC,QAAQ,CAAC,CAAC,MAAM,EAAE,CAAC;YACzB,CAAC;YAAC,MAAM,CAAC;gBACL,6EAA6E;gBAC7E,wDAAwD;YAC5D,CAAC;QACL,CAAC;IACL,CAAC;IAED;;;OAGG;IACK,YAAY,CAChB,CAAqB,EACrB,OAAoC;QAEpC,MAAM,QAAQ,GAAG,OAAO,EAAE,gBAAgB,IAAI,EAAE,CAAC;QACjD,KAAK,MAAM,QAAQ,IAAI,QAAQ,EAAE,CAAC;YAC9B,IAAI,CAAC;gBACD,MAAM,OAAO,GAAG,CAAC,CAAC,QAAQ,CAAC,CAAC;gBAC5B,IAAI,OAAO,CAAC,MAAM,GAAG,CAAC,EAAE,CAAC;oBACrB,OAAO,OAA4C,CAAC;gBACxD,CAAC;YACL,CAAC;YAAC,MAAM,CAAC;gBACL,2DAA2D;YAC/D,CAAC;QACL,CAAC;QACD,MAAM,IAAI,GAAG,CAAC,CAAC,MAAM,CAAC,CAAC;QACvB,MAAM,KAAK,GAAG,IAAI,CAAC,MAAM,GAAG,CAAC,CAAC,CAAC,CAAC,IAAI,CAAC,CAAC,CAAC,CAAC,CAAC,IAAI,EAAE,CAAC;QAChD,OAAO,KAA0C,CAAC;IACtD,CAAC;IAED,+EAA+E;IACvE,mBAAmB,CACvB,CAAqB,EACrB,KAA6B,EAC7B,OAAoC;QAEpC,IAAI,CAAC,OAAO,EAAE,mBAAmB,EAAE,CAAC;YAChC,OAAO;QACX,CAAC;QACD,CAAC,CAAC,KAAK,CAAC,CAAC,IAAI,CAAC,KAAK,CAAC,CAAC,IAAI,CAAC,CAAC,EAAE,EAAE,EAAE,EAAE,EAAE;YACjC,MAAM,GAAG,GAAG,CAAC,CAAC,EAAE,CAAC,CAAC,IAAI,CAAC,KAAK,CAAC,CAAC;YAC9B,IAAI,GAAG,IAAI,GAAG,CAAC,IAAI,EAAE,CAAC,MAAM,GAAG,CAAC,EAAE,CAAC;gBAC/B,CAAC,CAAC,EAAE,CAAC,CAAC,WAAW,CAAC,MAAM,GAAG,CAAC,IAAI,EAAE,MAAM,CAAC,CAAC;YAC9C,CAAC;QACL,CAAC,CAAC,CAAC;IACP,CAAC;IAED;;;;;OAKG;IACK,WAAW,CAAC,CAAqB,EAAE,KAA6B;QACpE,CAAC,CAAC,KAAK,CAAC,CAAC,IAAI,CAAC,2EAA2E,CAAC,CAAC,IAAI,CAAC,CAAC,EAAE,EAAE,EAAE,EAAE,EAAE;YACvG,CAAC,CAAC,EAAE,CAAC,CAAC,MAAM,CAAC,MAAM,CAAC,CAAC;QACzB,CAAC,CAAC,CAAC;QACH,CAAC,CAAC,KAAK,CAAC,CAAC,IAAI,CAAC,IAAI,CAAC,CAAC,WAAW,CAAC,IAAI,CAAC,CAAC;QAEtC,OAAO,CAAC,CAAC,KAAK,CAAC,CAAC,IAAI,EAAE,CAAC;IAC3B,CAAC;CACJ,CAAA;AA1FY,kBAAkB;IAD9B,aAAa,CAAC,kBAAkB,EAAE,wBAAwB,CAAC;GAC/C,kBAAkB,CA0F9B"}
|
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview Page-per-segment segmenter for paginated sources (PDF, slide decks).
|
|
3
|
+
*
|
|
4
|
+
* @module @memberjunction/ai-segmentation
|
|
5
|
+
*/
|
|
6
|
+
import { BaseSegmenter } from './BaseSegmenter.js';
|
|
7
|
+
import { ContentModality, RawSegment, SegmentationOptions, SegmentationParams } from './Segmentation.types.js';
|
|
8
|
+
/** Registration key for {@link PagedContentSegmenter}. */
|
|
9
|
+
export declare const PAGED_CONTENT_SEGMENTER_KEY = "PagedContent";
|
|
10
|
+
/** Options specific to {@link PagedContentSegmenter}. */
|
|
11
|
+
export interface PagedContentSegmentationOptions extends SegmentationOptions {
|
|
12
|
+
/**
|
|
13
|
+
* Merge consecutive pages until the token target is reached, instead of emitting one
|
|
14
|
+
* segment per page. Useful for documents with very short pages (slide decks) where a
|
|
15
|
+
* single slide is too little context to retrieve on. Default: false.
|
|
16
|
+
*/
|
|
17
|
+
MergeSmallPages?: boolean;
|
|
18
|
+
/** Prefix each segment's text with a "Page N" label. Default: false. */
|
|
19
|
+
IncludePageLabel?: boolean;
|
|
20
|
+
}
|
|
21
|
+
/**
|
|
22
|
+
* Emits one segment per page of a paginated source, preserving `PageNumber`.
|
|
23
|
+
*
|
|
24
|
+
* Page boundaries are authored boundaries — an author decided where the page broke — which
|
|
25
|
+
* makes them a better split point than any inferred one, and they give citation-grade
|
|
26
|
+
* provenance: a retrieved chunk resolves to "page 14 of this PDF" rather than a character
|
|
27
|
+
* offset nobody can act on.
|
|
28
|
+
*
|
|
29
|
+
* Each page may carry text, a media reference, or both. The media case is what enables
|
|
30
|
+
* embedding a PDF page *as an image* with a multimodal model — preserving tables, charts,
|
|
31
|
+
* and layout that text extraction flattens or loses entirely — while the extracted text
|
|
32
|
+
* rides along for lexical search and agent reasoning.
|
|
33
|
+
*
|
|
34
|
+
* Pages are supplied by the caller via `SegmentationParams.Pages`; this segmenter does no
|
|
35
|
+
* PDF parsing itself, keeping document-format dependencies out of the segmentation layer.
|
|
36
|
+
*/
|
|
37
|
+
export declare class PagedContentSegmenter extends BaseSegmenter {
|
|
38
|
+
get Key(): string;
|
|
39
|
+
get SupportedModalities(): ContentModality[];
|
|
40
|
+
protected SegmentCore(params: SegmentationParams<PagedContentSegmentationOptions>): RawSegment[];
|
|
41
|
+
/** True when a page carries text or media. */
|
|
42
|
+
private hasPageContent;
|
|
43
|
+
/** One page becomes one segment. */
|
|
44
|
+
private pageToSegment;
|
|
45
|
+
/**
|
|
46
|
+
* Merge consecutive TEXT-ONLY pages up to the token target.
|
|
47
|
+
*
|
|
48
|
+
* A page carrying media is never merged: its media reference identifies one page, so
|
|
49
|
+
* folding another page's text into it would make the segment's provenance a lie.
|
|
50
|
+
*/
|
|
51
|
+
private mergePages;
|
|
52
|
+
/** Text pages are text; a page with media is an image, or multimodal when it has both. */
|
|
53
|
+
private resolveModality;
|
|
54
|
+
}
|
|
55
|
+
//# sourceMappingURL=PagedContentSegmenter.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"PagedContentSegmenter.d.ts","sourceRoot":"","sources":["../../src/generic/PagedContentSegmenter.ts"],"names":[],"mappings":"AAAA;;;;GAIG;AAGH,OAAO,EAAE,aAAa,EAAE,MAAM,iBAAiB,CAAC;AAChD,OAAO,EAAE,eAAe,EAAe,UAAU,EAAE,mBAAmB,EAAE,kBAAkB,EAAE,MAAM,sBAAsB,CAAC;AAEzH,0DAA0D;AAC1D,eAAO,MAAM,2BAA2B,iBAAiB,CAAC;AAE1D,yDAAyD;AACzD,MAAM,WAAW,+BAAgC,SAAQ,mBAAmB;IACxE;;;;OAIG;IACH,eAAe,CAAC,EAAE,OAAO,CAAC;IAC1B,wEAAwE;IACxE,gBAAgB,CAAC,EAAE,OAAO,CAAC;CAC9B;AAED;;;;;;;;;;;;;;;GAeG;AACH,qBACa,qBAAsB,SAAQ,aAAa;IACpD,IAAW,GAAG,IAAI,MAAM,CAEvB;IAED,IAAW,mBAAmB,IAAI,eAAe,EAAE,CAElD;IAED,SAAS,CAAC,WAAW,CAAC,MAAM,EAAE,kBAAkB,CAAC,+BAA+B,CAAC,GAAG,UAAU,EAAE;IAWhG,8CAA8C;IAC9C,OAAO,CAAC,cAAc;IAItB,oCAAoC;IACpC,OAAO,CAAC,aAAa;IAWrB;;;;;OAKG;IACH,OAAO,CAAC,UAAU;IAoBlB,0FAA0F;IAC1F,OAAO,CAAC,eAAe;CAM1B"}
|
|
@@ -0,0 +1,99 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview Page-per-segment segmenter for paginated sources (PDF, slide decks).
|
|
3
|
+
*
|
|
4
|
+
* @module @memberjunction/ai-segmentation
|
|
5
|
+
*/
|
|
6
|
+
var __decorate = (this && this.__decorate) || function (decorators, target, key, desc) {
|
|
7
|
+
var c = arguments.length, r = c < 3 ? target : desc === null ? desc = Object.getOwnPropertyDescriptor(target, key) : desc, d;
|
|
8
|
+
if (typeof Reflect === "object" && typeof Reflect.decorate === "function") r = Reflect.decorate(decorators, target, key, desc);
|
|
9
|
+
else for (var i = decorators.length - 1; i >= 0; i--) if (d = decorators[i]) r = (c < 3 ? d(r) : c > 3 ? d(target, key, r) : d(target, key)) || r;
|
|
10
|
+
return c > 3 && r && Object.defineProperty(target, key, r), r;
|
|
11
|
+
};
|
|
12
|
+
import { RegisterClass } from '@memberjunction/global';
|
|
13
|
+
import { BaseSegmenter } from './BaseSegmenter.js';
|
|
14
|
+
/** Registration key for {@link PagedContentSegmenter}. */
|
|
15
|
+
export const PAGED_CONTENT_SEGMENTER_KEY = 'PagedContent';
|
|
16
|
+
/**
|
|
17
|
+
* Emits one segment per page of a paginated source, preserving `PageNumber`.
|
|
18
|
+
*
|
|
19
|
+
* Page boundaries are authored boundaries — an author decided where the page broke — which
|
|
20
|
+
* makes them a better split point than any inferred one, and they give citation-grade
|
|
21
|
+
* provenance: a retrieved chunk resolves to "page 14 of this PDF" rather than a character
|
|
22
|
+
* offset nobody can act on.
|
|
23
|
+
*
|
|
24
|
+
* Each page may carry text, a media reference, or both. The media case is what enables
|
|
25
|
+
* embedding a PDF page *as an image* with a multimodal model — preserving tables, charts,
|
|
26
|
+
* and layout that text extraction flattens or loses entirely — while the extracted text
|
|
27
|
+
* rides along for lexical search and agent reasoning.
|
|
28
|
+
*
|
|
29
|
+
* Pages are supplied by the caller via `SegmentationParams.Pages`; this segmenter does no
|
|
30
|
+
* PDF parsing itself, keeping document-format dependencies out of the segmentation layer.
|
|
31
|
+
*/
|
|
32
|
+
let PagedContentSegmenter = class PagedContentSegmenter extends BaseSegmenter {
|
|
33
|
+
get Key() {
|
|
34
|
+
return PAGED_CONTENT_SEGMENTER_KEY;
|
|
35
|
+
}
|
|
36
|
+
get SupportedModalities() {
|
|
37
|
+
return ['text', 'image', 'multimodal'];
|
|
38
|
+
}
|
|
39
|
+
SegmentCore(params) {
|
|
40
|
+
const pages = (params.Pages ?? []).filter((p) => this.hasPageContent(p));
|
|
41
|
+
if (pages.length === 0) {
|
|
42
|
+
return [];
|
|
43
|
+
}
|
|
44
|
+
const ordered = [...pages].sort((a, b) => a.PageNumber - b.PageNumber);
|
|
45
|
+
return params.Options?.MergeSmallPages
|
|
46
|
+
? this.mergePages(ordered, params)
|
|
47
|
+
: ordered.map((page) => this.pageToSegment(page, params.Options));
|
|
48
|
+
}
|
|
49
|
+
/** True when a page carries text or media. */
|
|
50
|
+
hasPageContent(page) {
|
|
51
|
+
return (!!page.Text && page.Text.trim().length > 0) || !!page.Media;
|
|
52
|
+
}
|
|
53
|
+
/** One page becomes one segment. */
|
|
54
|
+
pageToSegment(page, options) {
|
|
55
|
+
const label = options?.IncludePageLabel ? `Page ${page.PageNumber}\n` : '';
|
|
56
|
+
const text = page.Text?.trim();
|
|
57
|
+
return {
|
|
58
|
+
Modality: this.resolveModality(page),
|
|
59
|
+
Text: text ? `${label}${text}` : undefined,
|
|
60
|
+
Media: page.Media,
|
|
61
|
+
PageNumber: page.PageNumber,
|
|
62
|
+
};
|
|
63
|
+
}
|
|
64
|
+
/**
|
|
65
|
+
* Merge consecutive TEXT-ONLY pages up to the token target.
|
|
66
|
+
*
|
|
67
|
+
* A page carrying media is never merged: its media reference identifies one page, so
|
|
68
|
+
* folding another page's text into it would make the segment's provenance a lie.
|
|
69
|
+
*/
|
|
70
|
+
mergePages(pages, params) {
|
|
71
|
+
const maxTokens = this.resolveOptions(params.Options).MaxSegmentTokens;
|
|
72
|
+
const segments = [];
|
|
73
|
+
for (const page of pages) {
|
|
74
|
+
const candidate = this.pageToSegment(page, params.Options);
|
|
75
|
+
const previous = segments[segments.length - 1];
|
|
76
|
+
const mergeable = previous && !previous.Media && !candidate.Media &&
|
|
77
|
+
this.tokensOf(previous) + this.tokensOf(candidate) <= maxTokens;
|
|
78
|
+
if (mergeable) {
|
|
79
|
+
previous.Text = `${previous.Text ?? ''}\n\n${candidate.Text ?? ''}`.trim();
|
|
80
|
+
}
|
|
81
|
+
else {
|
|
82
|
+
segments.push(candidate);
|
|
83
|
+
}
|
|
84
|
+
}
|
|
85
|
+
return segments;
|
|
86
|
+
}
|
|
87
|
+
/** Text pages are text; a page with media is an image, or multimodal when it has both. */
|
|
88
|
+
resolveModality(page) {
|
|
89
|
+
if (!page.Media) {
|
|
90
|
+
return 'text';
|
|
91
|
+
}
|
|
92
|
+
return page.Text && page.Text.trim().length > 0 ? 'multimodal' : 'image';
|
|
93
|
+
}
|
|
94
|
+
};
|
|
95
|
+
PagedContentSegmenter = __decorate([
|
|
96
|
+
RegisterClass(BaseSegmenter, PAGED_CONTENT_SEGMENTER_KEY)
|
|
97
|
+
], PagedContentSegmenter);
|
|
98
|
+
export { PagedContentSegmenter };
|
|
99
|
+
//# sourceMappingURL=PagedContentSegmenter.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"PagedContentSegmenter.js","sourceRoot":"","sources":["../../src/generic/PagedContentSegmenter.ts"],"names":[],"mappings":"AAAA;;;;GAIG;;;;;;;AAEH,OAAO,EAAE,aAAa,EAAE,MAAM,wBAAwB,CAAC;AACvD,OAAO,EAAE,aAAa,EAAE,MAAM,iBAAiB,CAAC;AAGhD,0DAA0D;AAC1D,MAAM,CAAC,MAAM,2BAA2B,GAAG,cAAc,CAAC;AAc1D;;;;;;;;;;;;;;;GAeG;AAEI,IAAM,qBAAqB,GAA3B,MAAM,qBAAsB,SAAQ,aAAa;IACpD,IAAW,GAAG;QACV,OAAO,2BAA2B,CAAC;IACvC,CAAC;IAED,IAAW,mBAAmB;QAC1B,OAAO,CAAC,MAAM,EAAE,OAAO,EAAE,YAAY,CAAC,CAAC;IAC3C,CAAC;IAES,WAAW,CAAC,MAA2D;QAC7E,MAAM,KAAK,GAAG,CAAC,MAAM,CAAC,KAAK,IAAI,EAAE,CAAC,CAAC,MAAM,CAAC,CAAC,CAAC,EAAE,EAAE,CAAC,IAAI,CAAC,cAAc,CAAC,CAAC,CAAC,CAAC,CAAC;QACzE,IAAI,KAAK,CAAC,MAAM,KAAK,CAAC,EAAE,CAAC;YACrB,OAAO,EAAE,CAAC;QACd,CAAC;QACD,MAAM,OAAO,GAAG,CAAC,GAAG,KAAK,CAAC,CAAC,IAAI,CAAC,CAAC,CAAC,EAAE,CAAC,EAAE,EAAE,CAAC,CAAC,CAAC,UAAU,GAAG,CAAC,CAAC,UAAU,CAAC,CAAC;QACvE,OAAO,MAAM,CAAC,OAAO,EAAE,eAAe;YAClC,CAAC,CAAC,IAAI,CAAC,UAAU,CAAC,OAAO,EAAE,MAAM,CAAC;YAClC,CAAC,CAAC,OAAO,CAAC,GAAG,CAAC,CAAC,IAAI,EAAE,EAAE,CAAC,IAAI,CAAC,aAAa,CAAC,IAAI,EAAE,MAAM,CAAC,OAAO,CAAC,CAAC,CAAC;IAC1E,CAAC;IAED,8CAA8C;IACtC,cAAc,CAAC,IAAiB;QACpC,OAAO,CAAC,CAAC,CAAC,IAAI,CAAC,IAAI,IAAI,IAAI,CAAC,IAAI,CAAC,IAAI,EAAE,CAAC,MAAM,GAAG,CAAC,CAAC,IAAI,CAAC,CAAC,IAAI,CAAC,KAAK,CAAC;IACxE,CAAC;IAED,oCAAoC;IAC5B,aAAa,CAAC,IAAiB,EAAE,OAAyC;QAC9E,MAAM,KAAK,GAAG,OAAO,EAAE,gBAAgB,CAAC,CAAC,CAAC,QAAQ,IAAI,CAAC,UAAU,IAAI,CAAC,CAAC,CAAC,EAAE,CAAC;QAC3E,MAAM,IAAI,GAAG,IAAI,CAAC,IAAI,EAAE,IAAI,EAAE,CAAC;QAC/B,OAAO;YACH,QAAQ,EAAE,IAAI,CAAC,eAAe,CAAC,IAAI,CAAC;YACpC,IAAI,EAAE,IAAI,CAAC,CAAC,CAAC,GAAG,KAAK,GAAG,IAAI,EAAE,CAAC,CAAC,CAAC,SAAS;YAC1C,KAAK,EAAE,IAAI,CAAC,KAAK;YACjB,UAAU,EAAE,IAAI,CAAC,UAAU;SAC9B,CAAC;IACN,CAAC;IAED;;;;;OAKG;IACK,UAAU,CAAC,KAAoB,EAAE,MAA2D;QAChG,MAAM,SAAS,GAAG,IAAI,CAAC,cAAc,CAAC,MAAM,CAAC,OAAO,CAAC,CAAC,gBAAgB,CAAC;QACvE,MAAM,QAAQ,GAAiB,EAAE,CAAC;QAElC,KAAK,MAAM,IAAI,IAAI,KAAK,EAAE,CAAC;YACvB,MAAM,SAAS,GAAG,IAAI,CAAC,aAAa,CAAC,IAAI,EAAE,MAAM,CAAC,OAAO,CAAC,CAAC;YAC3D,MAAM,QAAQ,GAAG,QAAQ,CAAC,QAAQ,CAAC,MAAM,GAAG,CAAC,CAAC,CAAC;YAC/C,MAAM,SAAS,GACX,QAAQ,IAAI,CAAC,QAAQ,CAAC,KAAK,IAAI,CAAC,SAAS,CAAC,KAAK;gBAC/C,IAAI,CAAC,QAAQ,CAAC,QAAQ,CAAC,GAAG,IAAI,CAAC,QAAQ,CAAC,SAAS,CAAC,IAAI,SAAS,CAAC;YAEpE,IAAI,SAAS,EAAE,CAAC;gBACZ,QAAQ,CAAC,IAAI,GAAG,GAAG,QAAQ,CAAC,IAAI,IAAI,EAAE,OAAO,SAAS,CAAC,IAAI,IAAI,EAAE,EAAE,CAAC,IAAI,EAAE,CAAC;YAC/E,CAAC;iBAAM,CAAC;gBACJ,QAAQ,CAAC,IAAI,CAAC,SAAS,CAAC,CAAC;YAC7B,CAAC;QACL,CAAC;QACD,OAAO,QAAQ,CAAC;IACpB,CAAC;IAED,0FAA0F;IAClF,eAAe,CAAC,IAAiB;QACrC,IAAI,CAAC,IAAI,CAAC,KAAK,EAAE,CAAC;YACd,OAAO,MAAM,CAAC;QAClB,CAAC;QACD,OAAO,IAAI,CAAC,IAAI,IAAI,IAAI,CAAC,IAAI,CAAC,IAAI,EAAE,CAAC,MAAM,GAAG,CAAC,CAAC,CAAC,CAAC,YAAY,CAAC,CAAC,CAAC,OAAO,CAAC;IAC7E,CAAC;CACJ,CAAA;AAtEY,qBAAqB;IADjC,aAAa,CAAC,aAAa,EAAE,2BAA2B,CAAC;GAC7C,qBAAqB,CAsEjC"}
|
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview Pass-through cleaner for content that is already plain text.
|
|
3
|
+
*
|
|
4
|
+
* @module @memberjunction/ai-segmentation
|
|
5
|
+
*/
|
|
6
|
+
import { BaseContentCleaner, ContentCleaningParams } from './BaseContentCleaner.js';
|
|
7
|
+
/** Registration key for {@link PlainTextContentCleaner}. */
|
|
8
|
+
export declare const PLAIN_TEXT_CONTENT_CLEANER_KEY = "PlainText";
|
|
9
|
+
/**
|
|
10
|
+
* Applies only the shared cleaning rules — whitespace normalization and optional
|
|
11
|
+
* truncation — leaving the text otherwise untouched.
|
|
12
|
+
*
|
|
13
|
+
* This is the safe default for sources that are already plain text (transcripts, extracted
|
|
14
|
+
* PDF/DOCX text, markdown). It still earns its place in the pipeline: extracted text is
|
|
15
|
+
* routinely full of ragged spacing and stray blank lines from the extractor, and those
|
|
16
|
+
* confuse the paragraph-boundary detection that segmenters rely on.
|
|
17
|
+
*/
|
|
18
|
+
export declare class PlainTextContentCleaner extends BaseContentCleaner {
|
|
19
|
+
get Key(): string;
|
|
20
|
+
protected CleanCore(params: ContentCleaningParams): string;
|
|
21
|
+
}
|
|
22
|
+
//# sourceMappingURL=PlainTextContentCleaner.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"PlainTextContentCleaner.d.ts","sourceRoot":"","sources":["../../src/generic/PlainTextContentCleaner.ts"],"names":[],"mappings":"AAAA;;;;GAIG;AAGH,OAAO,EAAE,kBAAkB,EAAE,qBAAqB,EAAE,MAAM,sBAAsB,CAAC;AAEjF,4DAA4D;AAC5D,eAAO,MAAM,8BAA8B,cAAc,CAAC;AAE1D;;;;;;;;GAQG;AACH,qBACa,uBAAwB,SAAQ,kBAAkB;IAC3D,IAAW,GAAG,IAAI,MAAM,CAEvB;IAED,SAAS,CAAC,SAAS,CAAC,MAAM,EAAE,qBAAqB,GAAG,MAAM;CAG7D"}
|
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview Pass-through cleaner for content that is already plain text.
|
|
3
|
+
*
|
|
4
|
+
* @module @memberjunction/ai-segmentation
|
|
5
|
+
*/
|
|
6
|
+
var __decorate = (this && this.__decorate) || function (decorators, target, key, desc) {
|
|
7
|
+
var c = arguments.length, r = c < 3 ? target : desc === null ? desc = Object.getOwnPropertyDescriptor(target, key) : desc, d;
|
|
8
|
+
if (typeof Reflect === "object" && typeof Reflect.decorate === "function") r = Reflect.decorate(decorators, target, key, desc);
|
|
9
|
+
else for (var i = decorators.length - 1; i >= 0; i--) if (d = decorators[i]) r = (c < 3 ? d(r) : c > 3 ? d(target, key, r) : d(target, key)) || r;
|
|
10
|
+
return c > 3 && r && Object.defineProperty(target, key, r), r;
|
|
11
|
+
};
|
|
12
|
+
import { RegisterClass } from '@memberjunction/global';
|
|
13
|
+
import { BaseContentCleaner } from './BaseContentCleaner.js';
|
|
14
|
+
/** Registration key for {@link PlainTextContentCleaner}. */
|
|
15
|
+
export const PLAIN_TEXT_CONTENT_CLEANER_KEY = 'PlainText';
|
|
16
|
+
/**
|
|
17
|
+
* Applies only the shared cleaning rules — whitespace normalization and optional
|
|
18
|
+
* truncation — leaving the text otherwise untouched.
|
|
19
|
+
*
|
|
20
|
+
* This is the safe default for sources that are already plain text (transcripts, extracted
|
|
21
|
+
* PDF/DOCX text, markdown). It still earns its place in the pipeline: extracted text is
|
|
22
|
+
* routinely full of ragged spacing and stray blank lines from the extractor, and those
|
|
23
|
+
* confuse the paragraph-boundary detection that segmenters rely on.
|
|
24
|
+
*/
|
|
25
|
+
let PlainTextContentCleaner = class PlainTextContentCleaner extends BaseContentCleaner {
|
|
26
|
+
get Key() {
|
|
27
|
+
return PLAIN_TEXT_CONTENT_CLEANER_KEY;
|
|
28
|
+
}
|
|
29
|
+
CleanCore(params) {
|
|
30
|
+
return params.Content;
|
|
31
|
+
}
|
|
32
|
+
};
|
|
33
|
+
PlainTextContentCleaner = __decorate([
|
|
34
|
+
RegisterClass(BaseContentCleaner, PLAIN_TEXT_CONTENT_CLEANER_KEY)
|
|
35
|
+
], PlainTextContentCleaner);
|
|
36
|
+
export { PlainTextContentCleaner };
|
|
37
|
+
//# sourceMappingURL=PlainTextContentCleaner.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"PlainTextContentCleaner.js","sourceRoot":"","sources":["../../src/generic/PlainTextContentCleaner.ts"],"names":[],"mappings":"AAAA;;;;GAIG;;;;;;;AAEH,OAAO,EAAE,aAAa,EAAE,MAAM,wBAAwB,CAAC;AACvD,OAAO,EAAE,kBAAkB,EAAyB,MAAM,sBAAsB,CAAC;AAEjF,4DAA4D;AAC5D,MAAM,CAAC,MAAM,8BAA8B,GAAG,WAAW,CAAC;AAE1D;;;;;;;;GAQG;AAEI,IAAM,uBAAuB,GAA7B,MAAM,uBAAwB,SAAQ,kBAAkB;IAC3D,IAAW,GAAG;QACV,OAAO,8BAA8B,CAAC;IAC1C,CAAC;IAES,SAAS,CAAC,MAA6B;QAC7C,OAAO,MAAM,CAAC,OAAO,CAAC;IAC1B,CAAC;CACJ,CAAA;AARY,uBAAuB;IADnC,aAAa,CAAC,kBAAkB,EAAE,8BAA8B,CAAC;GACrD,uBAAuB,CAQnC"}
|