@memberjunction/ai-segmentation 0.0.0 → 5.51.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +177 -43
- package/dist/generic/AdaptiveBoundarySegmenter.d.ts +98 -0
- package/dist/generic/AdaptiveBoundarySegmenter.d.ts.map +1 -0
- package/dist/generic/AdaptiveBoundarySegmenter.js +177 -0
- package/dist/generic/AdaptiveBoundarySegmenter.js.map +1 -0
- package/dist/generic/BaseContentCleaner.d.ts +101 -0
- package/dist/generic/BaseContentCleaner.d.ts.map +1 -0
- package/dist/generic/BaseContentCleaner.js +114 -0
- package/dist/generic/BaseContentCleaner.js.map +1 -0
- package/dist/generic/BaseSegmenter.d.ts +106 -0
- package/dist/generic/BaseSegmenter.d.ts.map +1 -0
- package/dist/generic/BaseSegmenter.js +260 -0
- package/dist/generic/BaseSegmenter.js.map +1 -0
- package/dist/generic/FixedWindowSegmenter.d.ts +49 -0
- package/dist/generic/FixedWindowSegmenter.d.ts.map +1 -0
- package/dist/generic/FixedWindowSegmenter.js +106 -0
- package/dist/generic/FixedWindowSegmenter.js.map +1 -0
- package/dist/generic/HtmlContentCleaner.d.ts +61 -0
- package/dist/generic/HtmlContentCleaner.d.ts.map +1 -0
- package/dist/generic/HtmlContentCleaner.js +130 -0
- package/dist/generic/HtmlContentCleaner.js.map +1 -0
- package/dist/generic/PagedContentSegmenter.d.ts +55 -0
- package/dist/generic/PagedContentSegmenter.d.ts.map +1 -0
- package/dist/generic/PagedContentSegmenter.js +99 -0
- package/dist/generic/PagedContentSegmenter.js.map +1 -0
- package/dist/generic/PlainTextContentCleaner.d.ts +22 -0
- package/dist/generic/PlainTextContentCleaner.d.ts.map +1 -0
- package/dist/generic/PlainTextContentCleaner.js +37 -0
- package/dist/generic/PlainTextContentCleaner.js.map +1 -0
- package/dist/generic/Segmentation.types.d.ts +198 -0
- package/dist/generic/Segmentation.types.d.ts.map +1 -0
- package/dist/generic/Segmentation.types.js +19 -0
- package/dist/generic/Segmentation.types.js.map +1 -0
- package/dist/generic/SegmentationResolver.d.ts +43 -0
- package/dist/generic/SegmentationResolver.d.ts.map +1 -0
- package/dist/generic/SegmentationResolver.js +83 -0
- package/dist/generic/SegmentationResolver.js.map +1 -0
- package/dist/generic/SemanticTextSegmenter.d.ts +80 -0
- package/dist/generic/SemanticTextSegmenter.d.ts.map +1 -0
- package/dist/generic/SemanticTextSegmenter.js +201 -0
- package/dist/generic/SemanticTextSegmenter.js.map +1 -0
- package/dist/generic/StructuralTextSegmenter.d.ts +63 -0
- package/dist/generic/StructuralTextSegmenter.d.ts.map +1 -0
- package/dist/generic/StructuralTextSegmenter.js +177 -0
- package/dist/generic/StructuralTextSegmenter.js.map +1 -0
- package/dist/generic/TranscriptSegmenter.d.ts +78 -0
- package/dist/generic/TranscriptSegmenter.d.ts.map +1 -0
- package/dist/generic/TranscriptSegmenter.js +194 -0
- package/dist/generic/TranscriptSegmenter.js.map +1 -0
- package/dist/index.d.ts +33 -0
- package/dist/index.d.ts.map +1 -0
- package/dist/index.js +36 -0
- package/dist/index.js.map +1 -0
- package/package.json +33 -7
|
@@ -0,0 +1,198 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview Type contract for content segmentation.
|
|
3
|
+
*
|
|
4
|
+
* Segmentation is the step that turns a piece of source content (a document, a
|
|
5
|
+
* recording, an image) into an ordered list of embeddable {@link ContentSegment}s.
|
|
6
|
+
* It sits *upstream* of embedding: a segmenter decides **what** gets embedded,
|
|
7
|
+
* the embedding model decides **how**.
|
|
8
|
+
*
|
|
9
|
+
* This is deliberately separate from `TextChunker` (in `@memberjunction/ai-vectors`). `TextChunker` answers
|
|
10
|
+
* "how do I split this string so it fits a token budget"; a segmenter answers
|
|
11
|
+
* "what are the meaningful units of this content" — which may be sections of a
|
|
12
|
+
* document, chapters of a recording, or a single image. Segmenters typically
|
|
13
|
+
* *use* `TextChunker` to enforce the token budget within a unit they identified.
|
|
14
|
+
*
|
|
15
|
+
* @module @memberjunction/ai-segmentation
|
|
16
|
+
*/
|
|
17
|
+
import { UserInfo } from '@memberjunction/core';
|
|
18
|
+
/**
|
|
19
|
+
* The modality of a segment's payload. Drives downstream index routing
|
|
20
|
+
* (text vs. multimodal vector index) and retrieval-time fusion.
|
|
21
|
+
*/
|
|
22
|
+
export type ContentModality = 'text' | 'image' | 'audio' | 'video' | 'multimodal';
|
|
23
|
+
/**
|
|
24
|
+
* A pointer to non-text content. Segmenters emit these for media segments
|
|
25
|
+
* instead of (or alongside) `Text`.
|
|
26
|
+
*
|
|
27
|
+
* Exactly one of `URL`, `Base64Data`, or (`StorageProviderKey` + `ObjectKey`)
|
|
28
|
+
* is expected to be populated; consumers resolve whichever is present.
|
|
29
|
+
*/
|
|
30
|
+
export interface MediaReference {
|
|
31
|
+
/** Directly fetchable URL (http(s) or data URL). */
|
|
32
|
+
URL?: string;
|
|
33
|
+
/** `@memberjunction/storage` provider key, when the media lives in MJ file storage. */
|
|
34
|
+
StorageProviderKey?: string;
|
|
35
|
+
/** Object/blob key within the storage provider. */
|
|
36
|
+
ObjectKey?: string;
|
|
37
|
+
/** IANA mime type, e.g. `video/mp4`. Used to gate provider capability checks. */
|
|
38
|
+
MimeType?: string;
|
|
39
|
+
/** Inline base64 payload (no data-URL prefix). Prefer URL/storage refs for large media. */
|
|
40
|
+
Base64Data?: string;
|
|
41
|
+
}
|
|
42
|
+
/**
|
|
43
|
+
* A single timed transcript cue — the unit produced by ASR and by MJ's
|
|
44
|
+
* realtime session capture (which records speaker + timings per turn).
|
|
45
|
+
*/
|
|
46
|
+
export interface TranscriptCue {
|
|
47
|
+
/** Cue start, milliseconds from the beginning of the asset. */
|
|
48
|
+
StartMs: number;
|
|
49
|
+
/** Cue end, milliseconds from the beginning of the asset. */
|
|
50
|
+
EndMs: number;
|
|
51
|
+
/** Spoken text for this cue. */
|
|
52
|
+
Text: string;
|
|
53
|
+
/** Optional speaker label/id — a speaker change is a strong boundary signal. */
|
|
54
|
+
Speaker?: string;
|
|
55
|
+
}
|
|
56
|
+
/**
|
|
57
|
+
* One embeddable unit of content produced by a segmenter.
|
|
58
|
+
*
|
|
59
|
+
* A segment carries `Text`, `Media`, or **both** (the "dual representation" case:
|
|
60
|
+
* a video chapter with a native media reference *and* its transcript, so it can be
|
|
61
|
+
* embedded natively for retrieval while remaining readable for an agent).
|
|
62
|
+
*/
|
|
63
|
+
export interface ContentSegment {
|
|
64
|
+
/** 0-based position within the full segment list. Assigned by {@link BaseSegmenter}. */
|
|
65
|
+
Sequence: number;
|
|
66
|
+
/** Payload modality. */
|
|
67
|
+
Modality: ContentModality;
|
|
68
|
+
/** Textual payload — the extracted/transcribed text for this segment. */
|
|
69
|
+
Text?: string;
|
|
70
|
+
/** Media payload pointer, for non-text segments. */
|
|
71
|
+
Media?: MediaReference;
|
|
72
|
+
/** Human-readable label, e.g. a heading or a generated chapter title. */
|
|
73
|
+
Title?: string;
|
|
74
|
+
/** Inclusive start character offset within the source text. */
|
|
75
|
+
StartOffset?: number;
|
|
76
|
+
/** Exclusive end character offset within the source text. */
|
|
77
|
+
EndOffset?: number;
|
|
78
|
+
/** Segment start in milliseconds, for audio/video. */
|
|
79
|
+
StartMs?: number;
|
|
80
|
+
/** Segment end in milliseconds, for audio/video. */
|
|
81
|
+
EndMs?: number;
|
|
82
|
+
/** 1-based page number, for paginated sources (PDF, slides). */
|
|
83
|
+
PageNumber?: number;
|
|
84
|
+
/** `Sequence` of this segment's parent, for chapter -> sub-chapter hierarchies. */
|
|
85
|
+
ParentSequence?: number;
|
|
86
|
+
/** Nesting depth; 0 for top-level segments. */
|
|
87
|
+
Depth: number;
|
|
88
|
+
/** Estimated token count of `Text` (0 for pure-media segments). */
|
|
89
|
+
TokenEstimate: number;
|
|
90
|
+
/** Registration key of the segmenter that produced this segment — provenance. */
|
|
91
|
+
SegmenterKey: string;
|
|
92
|
+
/** Speaker label carried through from transcript cues, when known. */
|
|
93
|
+
Speaker?: string;
|
|
94
|
+
}
|
|
95
|
+
/**
|
|
96
|
+
* A segment as emitted by a concrete segmenter's `SegmentCore`, before the base
|
|
97
|
+
* class normalizes it (assigns `Sequence`/`Depth`/`TokenEstimate`, enforces the
|
|
98
|
+
* token ceiling, and resolves parent links).
|
|
99
|
+
*
|
|
100
|
+
* `ParentIndex` refers to the **index within the raw array** returned by
|
|
101
|
+
* `SegmentCore`; the base class remaps it to a real `Sequence` afterwards, which
|
|
102
|
+
* keeps subclasses from having to reason about post-split numbering.
|
|
103
|
+
*/
|
|
104
|
+
export interface RawSegment {
|
|
105
|
+
Modality: ContentModality;
|
|
106
|
+
Text?: string;
|
|
107
|
+
Media?: MediaReference;
|
|
108
|
+
Title?: string;
|
|
109
|
+
StartOffset?: number;
|
|
110
|
+
EndOffset?: number;
|
|
111
|
+
StartMs?: number;
|
|
112
|
+
EndMs?: number;
|
|
113
|
+
PageNumber?: number;
|
|
114
|
+
Speaker?: string;
|
|
115
|
+
/** Index into the raw segment array identifying this segment's parent. */
|
|
116
|
+
ParentIndex?: number;
|
|
117
|
+
}
|
|
118
|
+
/**
|
|
119
|
+
* One page of a paginated source (PDF, slide deck).
|
|
120
|
+
*
|
|
121
|
+
* A page may carry extracted text, a media reference to the rendered page, or both —
|
|
122
|
+
* the both case is what lets a page be embedded natively by a multimodal model (preserving
|
|
123
|
+
* tables and charts that text extraction flattens) while its text remains available for
|
|
124
|
+
* lexical search.
|
|
125
|
+
*/
|
|
126
|
+
export interface ContentPage {
|
|
127
|
+
/** One-based page number. */
|
|
128
|
+
PageNumber: number;
|
|
129
|
+
/** Extracted text for this page. */
|
|
130
|
+
Text?: string;
|
|
131
|
+
/** Reference to the rendered page image, when available. */
|
|
132
|
+
Media?: MediaReference;
|
|
133
|
+
}
|
|
134
|
+
/**
|
|
135
|
+
* Common knobs understood by every segmenter. Concrete segmenters extend this
|
|
136
|
+
* with their own strongly-typed options rather than accepting a loose bag.
|
|
137
|
+
*/
|
|
138
|
+
export interface SegmentationOptions {
|
|
139
|
+
/**
|
|
140
|
+
* Hard ceiling on tokens per text segment. The base class splits any
|
|
141
|
+
* oversized segment via `TextChunker` so no segmenter can exceed it.
|
|
142
|
+
* Default: 512.
|
|
143
|
+
*/
|
|
144
|
+
MaxSegmentTokens?: number;
|
|
145
|
+
/** Overlap tokens applied when an oversized segment must be split. Default: 10% of max. */
|
|
146
|
+
OverlapTokens?: number;
|
|
147
|
+
/**
|
|
148
|
+
* Segments whose text estimates below this many tokens are merged forward into
|
|
149
|
+
* the next segment, preventing a spray of near-empty vectors. Default: 0 (off).
|
|
150
|
+
*/
|
|
151
|
+
MinSegmentTokens?: number;
|
|
152
|
+
}
|
|
153
|
+
/**
|
|
154
|
+
* Input to {@link BaseSegmenter.Segment}.
|
|
155
|
+
*
|
|
156
|
+
* @typeParam TOptions - the concrete segmenter's options type.
|
|
157
|
+
*/
|
|
158
|
+
export interface SegmentationParams<TOptions extends SegmentationOptions = SegmentationOptions> {
|
|
159
|
+
/** Extracted text of the source content, when it has any. */
|
|
160
|
+
Text?: string;
|
|
161
|
+
/** Media pointer for the source asset, for image/audio/video content. */
|
|
162
|
+
Media?: MediaReference;
|
|
163
|
+
/** Timed transcript cues, when available (ASR output or MJ realtime capture). */
|
|
164
|
+
Cues?: TranscriptCue[];
|
|
165
|
+
/** Pages of a paginated source, for page-aware segmentation. */
|
|
166
|
+
Pages?: ContentPage[];
|
|
167
|
+
/** Total duration of the source asset in milliseconds, for AV content. */
|
|
168
|
+
DurationMs?: number;
|
|
169
|
+
/** Mime type of the source asset — lets a segmenter pick a structure parser. */
|
|
170
|
+
MimeType?: string;
|
|
171
|
+
/**
|
|
172
|
+
* Context user, required by segmenters that call MJ services (e.g. the LLM
|
|
173
|
+
* boundary pass in `SemanticTextSegmenter`). Always pass it in server-side code.
|
|
174
|
+
*/
|
|
175
|
+
ContextUser?: UserInfo;
|
|
176
|
+
/** Strategy-specific options. */
|
|
177
|
+
Options?: TOptions;
|
|
178
|
+
}
|
|
179
|
+
/**
|
|
180
|
+
* Result of a segmentation pass. Segmenters never throw for content-shaped
|
|
181
|
+
* problems — they return `Success: false` with an `ErrorMessage`, matching the
|
|
182
|
+
* convention used by `RunView` and `BaseEntity.Save`.
|
|
183
|
+
*/
|
|
184
|
+
export interface SegmentationResult {
|
|
185
|
+
/** False when segmentation could not be performed. */
|
|
186
|
+
Success: boolean;
|
|
187
|
+
/** The produced segments, in order. Empty when `Success` is false. */
|
|
188
|
+
Segments: ContentSegment[];
|
|
189
|
+
/** Registration key of the segmenter that ran. */
|
|
190
|
+
SegmenterKey: string;
|
|
191
|
+
/** Populated when `Success` is false. */
|
|
192
|
+
ErrorMessage?: string;
|
|
193
|
+
/** Non-fatal notes — e.g. "no cues supplied, fell back to fixed windows". */
|
|
194
|
+
Warnings: string[];
|
|
195
|
+
}
|
|
196
|
+
/** Default token ceiling applied when a caller does not specify one. */
|
|
197
|
+
export declare const DEFAULT_MAX_SEGMENT_TOKENS = 512;
|
|
198
|
+
//# sourceMappingURL=Segmentation.types.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"Segmentation.types.d.ts","sourceRoot":"","sources":["../../src/generic/Segmentation.types.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;GAeG;AAEH,OAAO,EAAE,QAAQ,EAAE,MAAM,sBAAsB,CAAC;AAEhD;;;GAGG;AACH,MAAM,MAAM,eAAe,GAAG,MAAM,GAAG,OAAO,GAAG,OAAO,GAAG,OAAO,GAAG,YAAY,CAAC;AAElF;;;;;;GAMG;AACH,MAAM,WAAW,cAAc;IAC3B,oDAAoD;IACpD,GAAG,CAAC,EAAE,MAAM,CAAC;IACb,uFAAuF;IACvF,kBAAkB,CAAC,EAAE,MAAM,CAAC;IAC5B,mDAAmD;IACnD,SAAS,CAAC,EAAE,MAAM,CAAC;IACnB,iFAAiF;IACjF,QAAQ,CAAC,EAAE,MAAM,CAAC;IAClB,2FAA2F;IAC3F,UAAU,CAAC,EAAE,MAAM,CAAC;CACvB;AAED;;;GAGG;AACH,MAAM,WAAW,aAAa;IAC1B,+DAA+D;IAC/D,OAAO,EAAE,MAAM,CAAC;IAChB,6DAA6D;IAC7D,KAAK,EAAE,MAAM,CAAC;IACd,gCAAgC;IAChC,IAAI,EAAE,MAAM,CAAC;IACb,gFAAgF;IAChF,OAAO,CAAC,EAAE,MAAM,CAAC;CACpB;AAED;;;;;;GAMG;AACH,MAAM,WAAW,cAAc;IAC3B,wFAAwF;IACxF,QAAQ,EAAE,MAAM,CAAC;IACjB,wBAAwB;IACxB,QAAQ,EAAE,eAAe,CAAC;IAC1B,yEAAyE;IACzE,IAAI,CAAC,EAAE,MAAM,CAAC;IACd,oDAAoD;IACpD,KAAK,CAAC,EAAE,cAAc,CAAC;IACvB,yEAAyE;IACzE,KAAK,CAAC,EAAE,MAAM,CAAC;IACf,+DAA+D;IAC/D,WAAW,CAAC,EAAE,MAAM,CAAC;IACrB,6DAA6D;IAC7D,SAAS,CAAC,EAAE,MAAM,CAAC;IACnB,sDAAsD;IACtD,OAAO,CAAC,EAAE,MAAM,CAAC;IACjB,oDAAoD;IACpD,KAAK,CAAC,EAAE,MAAM,CAAC;IACf,gEAAgE;IAChE,UAAU,CAAC,EAAE,MAAM,CAAC;IACpB,mFAAmF;IACnF,cAAc,CAAC,EAAE,MAAM,CAAC;IACxB,+CAA+C;IAC/C,KAAK,EAAE,MAAM,CAAC;IACd,mEAAmE;IACnE,aAAa,EAAE,MAAM,CAAC;IACtB,iFAAiF;IACjF,YAAY,EAAE,MAAM,CAAC;IACrB,sEAAsE;IACtE,OAAO,CAAC,EAAE,MAAM,CAAC;CACpB;AAED;;;;;;;;GAQG;AACH,MAAM,WAAW,UAAU;IACvB,QAAQ,EAAE,eAAe,CAAC;IAC1B,IAAI,CAAC,EAAE,MAAM,CAAC;IACd,KAAK,CAAC,EAAE,cAAc,CAAC;IACvB,KAAK,CAAC,EAAE,MAAM,CAAC;IACf,WAAW,CAAC,EAAE,MAAM,CAAC;IACrB,SAAS,CAAC,EAAE,MAAM,CAAC;IACnB,OAAO,CAAC,EAAE,MAAM,CAAC;IACjB,KAAK,CAAC,EAAE,MAAM,CAAC;IACf,UAAU,CAAC,EAAE,MAAM,CAAC;IACpB,OAAO,CAAC,EAAE,MAAM,CAAC;IACjB,0EAA0E;IAC1E,WAAW,CAAC,EAAE,MAAM,CAAC;CACxB;AAED;;;;;;;GAOG;AACH,MAAM,WAAW,WAAW;IACxB,6BAA6B;IAC7B,UAAU,EAAE,MAAM,CAAC;IACnB,oCAAoC;IACpC,IAAI,CAAC,EAAE,MAAM,CAAC;IACd,4DAA4D;IAC5D,KAAK,CAAC,EAAE,cAAc,CAAC;CAC1B;AAED;;;GAGG;AACH,MAAM,WAAW,mBAAmB;IAChC;;;;OAIG;IACH,gBAAgB,CAAC,EAAE,MAAM,CAAC;IAC1B,2FAA2F;IAC3F,aAAa,CAAC,EAAE,MAAM,CAAC;IACvB;;;OAGG;IACH,gBAAgB,CAAC,EAAE,MAAM,CAAC;CAC7B;AAED;;;;GAIG;AACH,MAAM,WAAW,kBAAkB,CAAC,QAAQ,SAAS,mBAAmB,GAAG,mBAAmB;IAC1F,6DAA6D;IAC7D,IAAI,CAAC,EAAE,MAAM,CAAC;IACd,yEAAyE;IACzE,KAAK,CAAC,EAAE,cAAc,CAAC;IACvB,iFAAiF;IACjF,IAAI,CAAC,EAAE,aAAa,EAAE,CAAC;IACvB,gEAAgE;IAChE,KAAK,CAAC,EAAE,WAAW,EAAE,CAAC;IACtB,0EAA0E;IAC1E,UAAU,CAAC,EAAE,MAAM,CAAC;IACpB,gFAAgF;IAChF,QAAQ,CAAC,EAAE,MAAM,CAAC;IAClB;;;OAGG;IACH,WAAW,CAAC,EAAE,QAAQ,CAAC;IACvB,iCAAiC;IACjC,OAAO,CAAC,EAAE,QAAQ,CAAC;CACtB;AAED;;;;GAIG;AACH,MAAM,WAAW,kBAAkB;IAC/B,sDAAsD;IACtD,OAAO,EAAE,OAAO,CAAC;IACjB,sEAAsE;IACtE,QAAQ,EAAE,cAAc,EAAE,CAAC;IAC3B,kDAAkD;IAClD,YAAY,EAAE,MAAM,CAAC;IACrB,yCAAyC;IACzC,YAAY,CAAC,EAAE,MAAM,CAAC;IACtB,6EAA6E;IAC7E,QAAQ,EAAE,MAAM,EAAE,CAAC;CACtB;AAED,wEAAwE;AACxE,eAAO,MAAM,0BAA0B,MAAM,CAAC"}
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview Type contract for content segmentation.
|
|
3
|
+
*
|
|
4
|
+
* Segmentation is the step that turns a piece of source content (a document, a
|
|
5
|
+
* recording, an image) into an ordered list of embeddable {@link ContentSegment}s.
|
|
6
|
+
* It sits *upstream* of embedding: a segmenter decides **what** gets embedded,
|
|
7
|
+
* the embedding model decides **how**.
|
|
8
|
+
*
|
|
9
|
+
* This is deliberately separate from `TextChunker` (in `@memberjunction/ai-vectors`). `TextChunker` answers
|
|
10
|
+
* "how do I split this string so it fits a token budget"; a segmenter answers
|
|
11
|
+
* "what are the meaningful units of this content" — which may be sections of a
|
|
12
|
+
* document, chapters of a recording, or a single image. Segmenters typically
|
|
13
|
+
* *use* `TextChunker` to enforce the token budget within a unit they identified.
|
|
14
|
+
*
|
|
15
|
+
* @module @memberjunction/ai-segmentation
|
|
16
|
+
*/
|
|
17
|
+
/** Default token ceiling applied when a caller does not specify one. */
|
|
18
|
+
export const DEFAULT_MAX_SEGMENT_TOKENS = 512;
|
|
19
|
+
//# sourceMappingURL=Segmentation.types.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"Segmentation.types.js","sourceRoot":"","sources":["../../src/generic/Segmentation.types.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;GAeG;AA+LH,wEAAwE;AACxE,MAAM,CAAC,MAAM,0BAA0B,GAAG,GAAG,CAAC"}
|
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview Helpers for selecting a segmenter from configuration or content shape.
|
|
3
|
+
*
|
|
4
|
+
* @module @memberjunction/ai-segmentation
|
|
5
|
+
*/
|
|
6
|
+
import { BaseSegmenter } from './BaseSegmenter.js';
|
|
7
|
+
import { BaseContentCleaner } from './BaseContentCleaner.js';
|
|
8
|
+
import { SegmentationParams } from './Segmentation.types.js';
|
|
9
|
+
/**
|
|
10
|
+
* Resolve a segmenter by registration key, falling back safely.
|
|
11
|
+
*
|
|
12
|
+
* Configuration is data, and data drifts — a `Content Type` may name a segmenter
|
|
13
|
+
* that has been renamed or lives in a package this process didn't load. Rather
|
|
14
|
+
* than throw mid-ingestion, an unresolvable key logs and degrades to the
|
|
15
|
+
* fixed-window segmenter, which can segment anything.
|
|
16
|
+
*
|
|
17
|
+
* @param key - registration key from metadata; when omitted the fallback is used.
|
|
18
|
+
* @param fallbackKey - key to try before the built-in last resort.
|
|
19
|
+
*/
|
|
20
|
+
export declare function ResolveSegmenter(key?: string, fallbackKey?: string): BaseSegmenter;
|
|
21
|
+
/**
|
|
22
|
+
* Suggest the best-fit segmenter key for a piece of content.
|
|
23
|
+
*
|
|
24
|
+
* The ordering encodes the quality hierarchy: a real transcript beats document
|
|
25
|
+
* structure, which beats uniform windows. Callers should treat this as a default
|
|
26
|
+
* that explicit configuration may override.
|
|
27
|
+
*/
|
|
28
|
+
export declare function SuggestSegmenterKey(params: SegmentationParams): string;
|
|
29
|
+
/**
|
|
30
|
+
* Resolve a content cleaner by registration key, falling back safely.
|
|
31
|
+
*
|
|
32
|
+
* Mirrors {@link ResolveSegmenter}: an unresolvable key logs and degrades to the
|
|
33
|
+
* plain-text cleaner (whitespace normalization only) rather than throwing mid-ingestion.
|
|
34
|
+
*/
|
|
35
|
+
export declare function ResolveContentCleaner(key?: string, fallbackKey?: string): BaseContentCleaner;
|
|
36
|
+
/**
|
|
37
|
+
* Suggest a cleaner for a piece of content based on its mime type.
|
|
38
|
+
*
|
|
39
|
+
* HTML is the only format that genuinely needs structural cleaning; everything else is
|
|
40
|
+
* already text and only wants whitespace normalization.
|
|
41
|
+
*/
|
|
42
|
+
export declare function SuggestCleanerKey(mimeType?: string): string;
|
|
43
|
+
//# sourceMappingURL=SegmentationResolver.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"SegmentationResolver.d.ts","sourceRoot":"","sources":["../../src/generic/SegmentationResolver.ts"],"names":[],"mappings":"AAAA;;;;GAIG;AAGH,OAAO,EAAE,aAAa,EAAE,MAAM,iBAAiB,CAAC;AAKhD,OAAO,EAAE,kBAAkB,EAAE,MAAM,sBAAsB,CAAC;AAG1D,OAAO,EAAE,kBAAkB,EAAE,MAAM,sBAAsB,CAAC;AAE1D;;;;;;;;;;GAUG;AACH,wBAAgB,gBAAgB,CAAC,GAAG,CAAC,EAAE,MAAM,EAAE,WAAW,CAAC,EAAE,MAAM,GAAG,aAAa,CAUlF;AAED;;;;;;GAMG;AACH,wBAAgB,mBAAmB,CAAC,MAAM,EAAE,kBAAkB,GAAG,MAAM,CAWtE;AAED;;;;;GAKG;AACH,wBAAgB,qBAAqB,CAAC,GAAG,CAAC,EAAE,MAAM,EAAE,WAAW,CAAC,EAAE,MAAM,GAAG,kBAAkB,CAU5F;AAED;;;;;GAKG;AACH,wBAAgB,iBAAiB,CAAC,QAAQ,CAAC,EAAE,MAAM,GAAG,MAAM,CAG3D"}
|
|
@@ -0,0 +1,83 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview Helpers for selecting a segmenter from configuration or content shape.
|
|
3
|
+
*
|
|
4
|
+
* @module @memberjunction/ai-segmentation
|
|
5
|
+
*/
|
|
6
|
+
import { LogStatus } from '@memberjunction/core';
|
|
7
|
+
import { BaseSegmenter } from './BaseSegmenter.js';
|
|
8
|
+
import { FIXED_WINDOW_SEGMENTER_KEY, FixedWindowSegmenter } from './FixedWindowSegmenter.js';
|
|
9
|
+
import { STRUCTURAL_TEXT_SEGMENTER_KEY } from './StructuralTextSegmenter.js';
|
|
10
|
+
import { TRANSCRIPT_SEGMENTER_KEY } from './TranscriptSegmenter.js';
|
|
11
|
+
import { PAGED_CONTENT_SEGMENTER_KEY } from './PagedContentSegmenter.js';
|
|
12
|
+
import { BaseContentCleaner } from './BaseContentCleaner.js';
|
|
13
|
+
import { HTML_CONTENT_CLEANER_KEY } from './HtmlContentCleaner.js';
|
|
14
|
+
import { PLAIN_TEXT_CONTENT_CLEANER_KEY, PlainTextContentCleaner } from './PlainTextContentCleaner.js';
|
|
15
|
+
/**
|
|
16
|
+
* Resolve a segmenter by registration key, falling back safely.
|
|
17
|
+
*
|
|
18
|
+
* Configuration is data, and data drifts — a `Content Type` may name a segmenter
|
|
19
|
+
* that has been renamed or lives in a package this process didn't load. Rather
|
|
20
|
+
* than throw mid-ingestion, an unresolvable key logs and degrades to the
|
|
21
|
+
* fixed-window segmenter, which can segment anything.
|
|
22
|
+
*
|
|
23
|
+
* @param key - registration key from metadata; when omitted the fallback is used.
|
|
24
|
+
* @param fallbackKey - key to try before the built-in last resort.
|
|
25
|
+
*/
|
|
26
|
+
export function ResolveSegmenter(key, fallbackKey) {
|
|
27
|
+
const requested = key ? BaseSegmenter.Resolve(key) : null;
|
|
28
|
+
if (requested) {
|
|
29
|
+
return requested;
|
|
30
|
+
}
|
|
31
|
+
if (key) {
|
|
32
|
+
LogStatus(`[Segmentation] Segmenter '${key}' is not registered — falling back.`);
|
|
33
|
+
}
|
|
34
|
+
const fallback = fallbackKey ? BaseSegmenter.Resolve(fallbackKey) : null;
|
|
35
|
+
return fallback ?? BaseSegmenter.Resolve(FIXED_WINDOW_SEGMENTER_KEY) ?? new FixedWindowSegmenter();
|
|
36
|
+
}
|
|
37
|
+
/**
|
|
38
|
+
* Suggest the best-fit segmenter key for a piece of content.
|
|
39
|
+
*
|
|
40
|
+
* The ordering encodes the quality hierarchy: a real transcript beats document
|
|
41
|
+
* structure, which beats uniform windows. Callers should treat this as a default
|
|
42
|
+
* that explicit configuration may override.
|
|
43
|
+
*/
|
|
44
|
+
export function SuggestSegmenterKey(params) {
|
|
45
|
+
if (params.Cues && params.Cues.length > 0) {
|
|
46
|
+
return TRANSCRIPT_SEGMENTER_KEY;
|
|
47
|
+
}
|
|
48
|
+
if (params.Pages && params.Pages.length > 0) {
|
|
49
|
+
return PAGED_CONTENT_SEGMENTER_KEY;
|
|
50
|
+
}
|
|
51
|
+
if (params.Text && params.Text.trim().length > 0) {
|
|
52
|
+
return STRUCTURAL_TEXT_SEGMENTER_KEY;
|
|
53
|
+
}
|
|
54
|
+
return FIXED_WINDOW_SEGMENTER_KEY;
|
|
55
|
+
}
|
|
56
|
+
/**
|
|
57
|
+
* Resolve a content cleaner by registration key, falling back safely.
|
|
58
|
+
*
|
|
59
|
+
* Mirrors {@link ResolveSegmenter}: an unresolvable key logs and degrades to the
|
|
60
|
+
* plain-text cleaner (whitespace normalization only) rather than throwing mid-ingestion.
|
|
61
|
+
*/
|
|
62
|
+
export function ResolveContentCleaner(key, fallbackKey) {
|
|
63
|
+
const requested = key ? BaseContentCleaner.Resolve(key) : null;
|
|
64
|
+
if (requested) {
|
|
65
|
+
return requested;
|
|
66
|
+
}
|
|
67
|
+
if (key) {
|
|
68
|
+
LogStatus(`[Segmentation] Content cleaner '${key}' is not registered — falling back.`);
|
|
69
|
+
}
|
|
70
|
+
const fallback = fallbackKey ? BaseContentCleaner.Resolve(fallbackKey) : null;
|
|
71
|
+
return fallback ?? BaseContentCleaner.Resolve(PLAIN_TEXT_CONTENT_CLEANER_KEY) ?? new PlainTextContentCleaner();
|
|
72
|
+
}
|
|
73
|
+
/**
|
|
74
|
+
* Suggest a cleaner for a piece of content based on its mime type.
|
|
75
|
+
*
|
|
76
|
+
* HTML is the only format that genuinely needs structural cleaning; everything else is
|
|
77
|
+
* already text and only wants whitespace normalization.
|
|
78
|
+
*/
|
|
79
|
+
export function SuggestCleanerKey(mimeType) {
|
|
80
|
+
const mime = (mimeType ?? '').toLowerCase();
|
|
81
|
+
return mime.includes('html') || mime.includes('xml') ? HTML_CONTENT_CLEANER_KEY : PLAIN_TEXT_CONTENT_CLEANER_KEY;
|
|
82
|
+
}
|
|
83
|
+
//# sourceMappingURL=SegmentationResolver.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"SegmentationResolver.js","sourceRoot":"","sources":["../../src/generic/SegmentationResolver.ts"],"names":[],"mappings":"AAAA;;;;GAIG;AAEH,OAAO,EAAE,SAAS,EAAE,MAAM,sBAAsB,CAAC;AACjD,OAAO,EAAE,aAAa,EAAE,MAAM,iBAAiB,CAAC;AAChD,OAAO,EAAE,0BAA0B,EAAE,oBAAoB,EAAE,MAAM,wBAAwB,CAAC;AAC1F,OAAO,EAAE,6BAA6B,EAAE,MAAM,2BAA2B,CAAC;AAC1E,OAAO,EAAE,wBAAwB,EAAE,MAAM,uBAAuB,CAAC;AACjE,OAAO,EAAE,2BAA2B,EAAE,MAAM,yBAAyB,CAAC;AACtE,OAAO,EAAE,kBAAkB,EAAE,MAAM,sBAAsB,CAAC;AAC1D,OAAO,EAAE,wBAAwB,EAAE,MAAM,sBAAsB,CAAC;AAChE,OAAO,EAAE,8BAA8B,EAAE,uBAAuB,EAAE,MAAM,2BAA2B,CAAC;AAGpG;;;;;;;;;;GAUG;AACH,MAAM,UAAU,gBAAgB,CAAC,GAAY,EAAE,WAAoB;IAC/D,MAAM,SAAS,GAAG,GAAG,CAAC,CAAC,CAAC,aAAa,CAAC,OAAO,CAAC,GAAG,CAAC,CAAC,CAAC,CAAC,IAAI,CAAC;IAC1D,IAAI,SAAS,EAAE,CAAC;QACZ,OAAO,SAAS,CAAC;IACrB,CAAC;IACD,IAAI,GAAG,EAAE,CAAC;QACN,SAAS,CAAC,6BAA6B,GAAG,qCAAqC,CAAC,CAAC;IACrF,CAAC;IACD,MAAM,QAAQ,GAAG,WAAW,CAAC,CAAC,CAAC,aAAa,CAAC,OAAO,CAAC,WAAW,CAAC,CAAC,CAAC,CAAC,IAAI,CAAC;IACzE,OAAO,QAAQ,IAAI,aAAa,CAAC,OAAO,CAAC,0BAA0B,CAAC,IAAI,IAAI,oBAAoB,EAAE,CAAC;AACvG,CAAC;AAED;;;;;;GAMG;AACH,MAAM,UAAU,mBAAmB,CAAC,MAA0B;IAC1D,IAAI,MAAM,CAAC,IAAI,IAAI,MAAM,CAAC,IAAI,CAAC,MAAM,GAAG,CAAC,EAAE,CAAC;QACxC,OAAO,wBAAwB,CAAC;IACpC,CAAC;IACD,IAAI,MAAM,CAAC,KAAK,IAAI,MAAM,CAAC,KAAK,CAAC,MAAM,GAAG,CAAC,EAAE,CAAC;QAC1C,OAAO,2BAA2B,CAAC;IACvC,CAAC;IACD,IAAI,MAAM,CAAC,IAAI,IAAI,MAAM,CAAC,IAAI,CAAC,IAAI,EAAE,CAAC,MAAM,GAAG,CAAC,EAAE,CAAC;QAC/C,OAAO,6BAA6B,CAAC;IACzC,CAAC;IACD,OAAO,0BAA0B,CAAC;AACtC,CAAC;AAED;;;;;GAKG;AACH,MAAM,UAAU,qBAAqB,CAAC,GAAY,EAAE,WAAoB;IACpE,MAAM,SAAS,GAAG,GAAG,CAAC,CAAC,CAAC,kBAAkB,CAAC,OAAO,CAAC,GAAG,CAAC,CAAC,CAAC,CAAC,IAAI,CAAC;IAC/D,IAAI,SAAS,EAAE,CAAC;QACZ,OAAO,SAAS,CAAC;IACrB,CAAC;IACD,IAAI,GAAG,EAAE,CAAC;QACN,SAAS,CAAC,mCAAmC,GAAG,qCAAqC,CAAC,CAAC;IAC3F,CAAC;IACD,MAAM,QAAQ,GAAG,WAAW,CAAC,CAAC,CAAC,kBAAkB,CAAC,OAAO,CAAC,WAAW,CAAC,CAAC,CAAC,CAAC,IAAI,CAAC;IAC9E,OAAO,QAAQ,IAAI,kBAAkB,CAAC,OAAO,CAAC,8BAA8B,CAAC,IAAI,IAAI,uBAAuB,EAAE,CAAC;AACnH,CAAC;AAED;;;;;GAKG;AACH,MAAM,UAAU,iBAAiB,CAAC,QAAiB;IAC/C,MAAM,IAAI,GAAG,CAAC,QAAQ,IAAI,EAAE,CAAC,CAAC,WAAW,EAAE,CAAC;IAC5C,OAAO,IAAI,CAAC,QAAQ,CAAC,MAAM,CAAC,IAAI,IAAI,CAAC,QAAQ,CAAC,KAAK,CAAC,CAAC,CAAC,CAAC,wBAAwB,CAAC,CAAC,CAAC,8BAA8B,CAAC;AACrH,CAAC"}
|
|
@@ -0,0 +1,80 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview LLM-driven semantic segmenter — finds topic boundaries in prose.
|
|
3
|
+
*
|
|
4
|
+
* @module @memberjunction/ai-segmentation
|
|
5
|
+
*/
|
|
6
|
+
import { BaseSegmenter } from './BaseSegmenter.js';
|
|
7
|
+
import { ContentModality, RawSegment, SegmentationOptions, SegmentationParams } from './Segmentation.types.js';
|
|
8
|
+
/** Registration key for {@link SemanticTextSegmenter}. */
|
|
9
|
+
export declare const SEMANTIC_TEXT_SEGMENTER_KEY = "SemanticText";
|
|
10
|
+
/** Default name of the `MJ: AI Prompts` record driving boundary detection. */
|
|
11
|
+
export declare const SEMANTIC_SEGMENTATION_PROMPT_NAME = "Content Semantic Segmentation";
|
|
12
|
+
/** Options specific to {@link SemanticTextSegmenter}. */
|
|
13
|
+
export interface SemanticTextSegmentationOptions extends SegmentationOptions {
|
|
14
|
+
/** Name of the `MJ: AI Prompts` record to run. Default: {@link SEMANTIC_SEGMENTATION_PROMPT_NAME}. */
|
|
15
|
+
PromptName?: string;
|
|
16
|
+
/** Optional model override (an `MJ: AI Models` ID) for the boundary pass. */
|
|
17
|
+
ModelID?: string;
|
|
18
|
+
/**
|
|
19
|
+
* Skip the LLM call entirely when the document estimates below this many
|
|
20
|
+
* tokens — short documents rarely contain multiple topics and the call would
|
|
21
|
+
* not repay its cost. Default: 750.
|
|
22
|
+
*/
|
|
23
|
+
MinTokensForLLM?: number;
|
|
24
|
+
/**
|
|
25
|
+
* Maximum characters of each block shown to the model. Boundary detection only
|
|
26
|
+
* needs the opening of a block, so truncating keeps the prompt cheap on long
|
|
27
|
+
* documents. Default: 240.
|
|
28
|
+
*/
|
|
29
|
+
BlockPreviewChars?: number;
|
|
30
|
+
/** Maximum blocks sent in one pass. Default: 300. */
|
|
31
|
+
MaxBlocks?: number;
|
|
32
|
+
}
|
|
33
|
+
/**
|
|
34
|
+
* Segments prose by asking an LLM where the topics change.
|
|
35
|
+
*
|
|
36
|
+
* Structure-aware segmentation only works when the author left structure behind.
|
|
37
|
+
* Transcripts, scanned reports, and long-form articles frequently have none — the
|
|
38
|
+
* topic shifts, but no heading marks it. This segmenter finds those latent
|
|
39
|
+
* boundaries and names them, producing titled sections that behave like headings
|
|
40
|
+
* the document never had.
|
|
41
|
+
*
|
|
42
|
+
* ## Cost posture
|
|
43
|
+
*
|
|
44
|
+
* The LLM pass is the expensive part of ingestion, so this class is written to
|
|
45
|
+
* avoid it whenever it wouldn't pay off: short documents short-circuit to
|
|
46
|
+
* structural segmentation, blocks are truncated to a preview before being shown to
|
|
47
|
+
* the model, and any failure degrades to `StructuralText` rather than failing the
|
|
48
|
+
* ingestion run. The model is asked to classify *block indices*, never character
|
|
49
|
+
* offsets — models are unreliable at arithmetic over long strings, and a wrong
|
|
50
|
+
* offset would silently corrupt chunk provenance.
|
|
51
|
+
*
|
|
52
|
+
* Because it runs through `AIPromptRunner`, every pass is a tracked `MJ: AI Prompt
|
|
53
|
+
* Run` with full token and cost attribution, and the prompt itself is versioned
|
|
54
|
+
* metadata rather than a string literal in code.
|
|
55
|
+
*/
|
|
56
|
+
export declare class SemanticTextSegmenter extends BaseSegmenter {
|
|
57
|
+
get Key(): string;
|
|
58
|
+
get SupportedModalities(): ContentModality[];
|
|
59
|
+
protected SegmentCore(params: SegmentationParams<SemanticTextSegmentationOptions>): Promise<RawSegment[]>;
|
|
60
|
+
/** Split the document into paragraph blocks with real offsets. */
|
|
61
|
+
private buildBlocks;
|
|
62
|
+
/** Append a block when the slice has content. */
|
|
63
|
+
private pushBlock;
|
|
64
|
+
/** Run the segmentation prompt and return validated boundary block indices. */
|
|
65
|
+
private proposeBoundaries;
|
|
66
|
+
/** Render numbered, truncated blocks for the prompt. */
|
|
67
|
+
private renderBlocks;
|
|
68
|
+
/** Accept either a parsed object or a JSON string from the runner. */
|
|
69
|
+
private parseResult;
|
|
70
|
+
/**
|
|
71
|
+
* Clamp, dedupe, and sort model output. A hallucinated or out-of-range block
|
|
72
|
+
* index must never become a chunk offset, so anything unusable is dropped here.
|
|
73
|
+
*/
|
|
74
|
+
private sanitizeBoundaries;
|
|
75
|
+
/** Join blocks between consecutive boundaries into one segment each. */
|
|
76
|
+
private buildSegments;
|
|
77
|
+
/** Degrade to structural segmentation — never fail an ingestion run over segmentation. */
|
|
78
|
+
private fallback;
|
|
79
|
+
}
|
|
80
|
+
//# sourceMappingURL=SemanticTextSegmenter.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"SemanticTextSegmenter.d.ts","sourceRoot":"","sources":["../../src/generic/SemanticTextSegmenter.ts"],"names":[],"mappings":"AAAA;;;;GAIG;AAOH,OAAO,EAAE,aAAa,EAAE,MAAM,iBAAiB,CAAC;AAEhD,OAAO,EAAE,eAAe,EAAE,UAAU,EAAE,mBAAmB,EAAE,kBAAkB,EAAE,MAAM,sBAAsB,CAAC;AAE5G,0DAA0D;AAC1D,eAAO,MAAM,2BAA2B,iBAAiB,CAAC;AAE1D,8EAA8E;AAC9E,eAAO,MAAM,iCAAiC,kCAAkC,CAAC;AAEjF,yDAAyD;AACzD,MAAM,WAAW,+BAAgC,SAAQ,mBAAmB;IACxE,sGAAsG;IACtG,UAAU,CAAC,EAAE,MAAM,CAAC;IACpB,6EAA6E;IAC7E,OAAO,CAAC,EAAE,MAAM,CAAC;IACjB;;;;OAIG;IACH,eAAe,CAAC,EAAE,MAAM,CAAC;IACzB;;;;OAIG;IACH,iBAAiB,CAAC,EAAE,MAAM,CAAC;IAC3B,qDAAqD;IACrD,SAAS,CAAC,EAAE,MAAM,CAAC;CACtB;AAeD;;;;;;;;;;;;;;;;;;;;;;GAsBG;AACH,qBACa,qBAAsB,SAAQ,aAAa;IACpD,IAAW,GAAG,IAAI,MAAM,CAEvB;IAED,IAAW,mBAAmB,IAAI,eAAe,EAAE,CAElD;cAEe,WAAW,CAAC,MAAM,EAAE,kBAAkB,CAAC,+BAA+B,CAAC,GAAG,OAAO,CAAC,UAAU,EAAE,CAAC;IA0B/G,kEAAkE;IAClE,OAAO,CAAC,WAAW;IAcnB,iDAAiD;IACjD,OAAO,CAAC,SAAS;IAWjB,+EAA+E;YACjE,iBAAiB;IA6B/B,wDAAwD;IACxD,OAAO,CAAC,YAAY;IAKpB,sEAAsE;IACtE,OAAO,CAAC,WAAW;IAenB;;;OAGG;IACH,OAAO,CAAC,kBAAkB;IA2B1B,wEAAwE;IACxE,OAAO,CAAC,aAAa;IAerB,0FAA0F;YAC5E,QAAQ;CAWzB"}
|
|
@@ -0,0 +1,201 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview LLM-driven semantic segmenter — finds topic boundaries in prose.
|
|
3
|
+
*
|
|
4
|
+
* @module @memberjunction/ai-segmentation
|
|
5
|
+
*/
|
|
6
|
+
var __decorate = (this && this.__decorate) || function (decorators, target, key, desc) {
|
|
7
|
+
var c = arguments.length, r = c < 3 ? target : desc === null ? desc = Object.getOwnPropertyDescriptor(target, key) : desc, d;
|
|
8
|
+
if (typeof Reflect === "object" && typeof Reflect.decorate === "function") r = Reflect.decorate(decorators, target, key, desc);
|
|
9
|
+
else for (var i = decorators.length - 1; i >= 0; i--) if (d = decorators[i]) r = (c < 3 ? d(r) : c > 3 ? d(target, key, r) : d(target, key)) || r;
|
|
10
|
+
return c > 3 && r && Object.defineProperty(target, key, r), r;
|
|
11
|
+
};
|
|
12
|
+
import { LogError, LogStatus } from '@memberjunction/core';
|
|
13
|
+
import { RegisterClass } from '@memberjunction/global';
|
|
14
|
+
import { AIEngine } from '@memberjunction/aiengine';
|
|
15
|
+
import { AIPromptRunner } from '@memberjunction/ai-prompts';
|
|
16
|
+
import { AIPromptParams } from '@memberjunction/ai-core-plus';
|
|
17
|
+
import { BaseSegmenter } from './BaseSegmenter.js';
|
|
18
|
+
import { StructuralTextSegmenter } from './StructuralTextSegmenter.js';
|
|
19
|
+
/** Registration key for {@link SemanticTextSegmenter}. */
|
|
20
|
+
export const SEMANTIC_TEXT_SEGMENTER_KEY = 'SemanticText';
|
|
21
|
+
/** Default name of the `MJ: AI Prompts` record driving boundary detection. */
|
|
22
|
+
export const SEMANTIC_SEGMENTATION_PROMPT_NAME = 'Content Semantic Segmentation';
|
|
23
|
+
/**
|
|
24
|
+
* Segments prose by asking an LLM where the topics change.
|
|
25
|
+
*
|
|
26
|
+
* Structure-aware segmentation only works when the author left structure behind.
|
|
27
|
+
* Transcripts, scanned reports, and long-form articles frequently have none — the
|
|
28
|
+
* topic shifts, but no heading marks it. This segmenter finds those latent
|
|
29
|
+
* boundaries and names them, producing titled sections that behave like headings
|
|
30
|
+
* the document never had.
|
|
31
|
+
*
|
|
32
|
+
* ## Cost posture
|
|
33
|
+
*
|
|
34
|
+
* The LLM pass is the expensive part of ingestion, so this class is written to
|
|
35
|
+
* avoid it whenever it wouldn't pay off: short documents short-circuit to
|
|
36
|
+
* structural segmentation, blocks are truncated to a preview before being shown to
|
|
37
|
+
* the model, and any failure degrades to `StructuralText` rather than failing the
|
|
38
|
+
* ingestion run. The model is asked to classify *block indices*, never character
|
|
39
|
+
* offsets — models are unreliable at arithmetic over long strings, and a wrong
|
|
40
|
+
* offset would silently corrupt chunk provenance.
|
|
41
|
+
*
|
|
42
|
+
* Because it runs through `AIPromptRunner`, every pass is a tracked `MJ: AI Prompt
|
|
43
|
+
* Run` with full token and cost attribution, and the prompt itself is versioned
|
|
44
|
+
* metadata rather than a string literal in code.
|
|
45
|
+
*/
|
|
46
|
+
let SemanticTextSegmenter = class SemanticTextSegmenter extends BaseSegmenter {
|
|
47
|
+
get Key() {
|
|
48
|
+
return SEMANTIC_TEXT_SEGMENTER_KEY;
|
|
49
|
+
}
|
|
50
|
+
get SupportedModalities() {
|
|
51
|
+
return ['text'];
|
|
52
|
+
}
|
|
53
|
+
async SegmentCore(params) {
|
|
54
|
+
const text = params.Text ?? '';
|
|
55
|
+
if (text.trim().length === 0) {
|
|
56
|
+
return [];
|
|
57
|
+
}
|
|
58
|
+
const minTokens = params.Options?.MinTokensForLLM ?? 750;
|
|
59
|
+
if (this.tokensOf({ Modality: 'text', Text: text }) < minTokens) {
|
|
60
|
+
return this.fallback(params);
|
|
61
|
+
}
|
|
62
|
+
const blocks = this.buildBlocks(text, params.Options);
|
|
63
|
+
if (blocks.length < 2) {
|
|
64
|
+
return this.fallback(params);
|
|
65
|
+
}
|
|
66
|
+
const boundaries = await this.proposeBoundaries(blocks, params);
|
|
67
|
+
if (!boundaries || boundaries.length === 0) {
|
|
68
|
+
return this.fallback(params);
|
|
69
|
+
}
|
|
70
|
+
return this.buildSegments(blocks, boundaries);
|
|
71
|
+
}
|
|
72
|
+
// ─────────────────────────────────────────────
|
|
73
|
+
// Block preparation
|
|
74
|
+
// ─────────────────────────────────────────────
|
|
75
|
+
/** Split the document into paragraph blocks with real offsets. */
|
|
76
|
+
buildBlocks(text, options) {
|
|
77
|
+
const blocks = [];
|
|
78
|
+
const regex = /\n\s*\n/g;
|
|
79
|
+
let cursor = 0;
|
|
80
|
+
let match = regex.exec(text);
|
|
81
|
+
while (match !== null) {
|
|
82
|
+
this.pushBlock(blocks, text, cursor, match.index);
|
|
83
|
+
cursor = match.index + match[0].length;
|
|
84
|
+
match = regex.exec(text);
|
|
85
|
+
}
|
|
86
|
+
this.pushBlock(blocks, text, cursor, text.length);
|
|
87
|
+
return blocks.slice(0, options?.MaxBlocks ?? 300);
|
|
88
|
+
}
|
|
89
|
+
/** Append a block when the slice has content. */
|
|
90
|
+
pushBlock(blocks, text, start, end) {
|
|
91
|
+
const body = text.slice(start, end).trim();
|
|
92
|
+
if (body.length > 0) {
|
|
93
|
+
blocks.push({ Index: blocks.length, Text: body, StartOffset: start, EndOffset: end });
|
|
94
|
+
}
|
|
95
|
+
}
|
|
96
|
+
// ─────────────────────────────────────────────
|
|
97
|
+
// LLM boundary detection
|
|
98
|
+
// ─────────────────────────────────────────────
|
|
99
|
+
/** Run the segmentation prompt and return validated boundary block indices. */
|
|
100
|
+
async proposeBoundaries(blocks, params) {
|
|
101
|
+
const promptName = params.Options?.PromptName ?? SEMANTIC_SEGMENTATION_PROMPT_NAME;
|
|
102
|
+
const prompt = AIEngine.Instance.Prompts.find((p) => p.Name === promptName && p.Status === 'Active');
|
|
103
|
+
if (!prompt) {
|
|
104
|
+
LogStatus(`[SemanticTextSegmenter] Prompt '${promptName}' not found — falling back to structural segmentation.`);
|
|
105
|
+
return null;
|
|
106
|
+
}
|
|
107
|
+
const promptParams = new AIPromptParams();
|
|
108
|
+
promptParams.prompt = prompt;
|
|
109
|
+
promptParams.contextUser = params.ContextUser;
|
|
110
|
+
promptParams.data = { document: this.renderBlocks(blocks, params.Options), blockCount: blocks.length };
|
|
111
|
+
promptParams.attemptJSONRepair = true;
|
|
112
|
+
promptParams.additionalParameters = { temperature: 0.0 };
|
|
113
|
+
if (params.Options?.ModelID) {
|
|
114
|
+
promptParams.override = { modelId: params.Options.ModelID };
|
|
115
|
+
}
|
|
116
|
+
const result = await new AIPromptRunner().ExecutePrompt(promptParams);
|
|
117
|
+
if (!result.success) {
|
|
118
|
+
LogError(`[SemanticTextSegmenter] Boundary prompt failed: ${result.errorMessage ?? 'unknown error'}`);
|
|
119
|
+
return null;
|
|
120
|
+
}
|
|
121
|
+
return this.sanitizeBoundaries(this.parseResult(result.result), blocks.length);
|
|
122
|
+
}
|
|
123
|
+
/** Render numbered, truncated blocks for the prompt. */
|
|
124
|
+
renderBlocks(blocks, options) {
|
|
125
|
+
const previewChars = options?.BlockPreviewChars ?? 240;
|
|
126
|
+
return blocks.map((b) => `[${b.Index}] ${b.Text.slice(0, previewChars)}`).join('\n\n');
|
|
127
|
+
}
|
|
128
|
+
/** Accept either a parsed object or a JSON string from the runner. */
|
|
129
|
+
parseResult(raw) {
|
|
130
|
+
if (!raw) {
|
|
131
|
+
return null;
|
|
132
|
+
}
|
|
133
|
+
if (typeof raw !== 'string') {
|
|
134
|
+
return raw;
|
|
135
|
+
}
|
|
136
|
+
try {
|
|
137
|
+
return JSON.parse(raw);
|
|
138
|
+
}
|
|
139
|
+
catch {
|
|
140
|
+
LogError(`[SemanticTextSegmenter] Could not parse boundary JSON: ${raw.substring(0, 200)}`);
|
|
141
|
+
return null;
|
|
142
|
+
}
|
|
143
|
+
}
|
|
144
|
+
/**
|
|
145
|
+
* Clamp, dedupe, and sort model output. A hallucinated or out-of-range block
|
|
146
|
+
* index must never become a chunk offset, so anything unusable is dropped here.
|
|
147
|
+
*/
|
|
148
|
+
sanitizeBoundaries(response, blockCount) {
|
|
149
|
+
const raw = response?.boundaries ?? [];
|
|
150
|
+
const seen = new Set();
|
|
151
|
+
const cleaned = [];
|
|
152
|
+
for (const entry of raw) {
|
|
153
|
+
const index = Number(entry?.startBlock);
|
|
154
|
+
if (!Number.isInteger(index) || index < 0 || index >= blockCount || seen.has(index)) {
|
|
155
|
+
continue;
|
|
156
|
+
}
|
|
157
|
+
seen.add(index);
|
|
158
|
+
cleaned.push({ startBlock: index, title: entry.title?.trim() || undefined });
|
|
159
|
+
}
|
|
160
|
+
cleaned.sort((a, b) => a.startBlock - b.startBlock);
|
|
161
|
+
if (cleaned.length > 0 && cleaned[0].startBlock !== 0) {
|
|
162
|
+
cleaned.unshift({ startBlock: 0 });
|
|
163
|
+
}
|
|
164
|
+
return cleaned;
|
|
165
|
+
}
|
|
166
|
+
// ─────────────────────────────────────────────
|
|
167
|
+
// Segment assembly
|
|
168
|
+
// ─────────────────────────────────────────────
|
|
169
|
+
/** Join blocks between consecutive boundaries into one segment each. */
|
|
170
|
+
buildSegments(blocks, boundaries) {
|
|
171
|
+
return boundaries.map((boundary, i) => {
|
|
172
|
+
const endBlock = i + 1 < boundaries.length ? boundaries[i + 1].startBlock : blocks.length;
|
|
173
|
+
const slice = blocks.slice(boundary.startBlock, endBlock);
|
|
174
|
+
const body = slice.map((b) => b.Text).join('\n\n');
|
|
175
|
+
return {
|
|
176
|
+
Modality: 'text',
|
|
177
|
+
Title: boundary.title,
|
|
178
|
+
Text: boundary.title ? `${boundary.title}\n${body}` : body,
|
|
179
|
+
StartOffset: slice[0]?.StartOffset,
|
|
180
|
+
EndOffset: slice[slice.length - 1]?.EndOffset,
|
|
181
|
+
};
|
|
182
|
+
});
|
|
183
|
+
}
|
|
184
|
+
/** Degrade to structural segmentation — never fail an ingestion run over segmentation. */
|
|
185
|
+
async fallback(params) {
|
|
186
|
+
const structural = new StructuralTextSegmenter();
|
|
187
|
+
const result = await structural.Segment({ Text: params.Text, MimeType: params.MimeType, Options: params.Options });
|
|
188
|
+
return result.Segments.map((s) => ({
|
|
189
|
+
Modality: s.Modality,
|
|
190
|
+
Text: s.Text,
|
|
191
|
+
Title: s.Title,
|
|
192
|
+
StartOffset: s.StartOffset,
|
|
193
|
+
EndOffset: s.EndOffset,
|
|
194
|
+
}));
|
|
195
|
+
}
|
|
196
|
+
};
|
|
197
|
+
SemanticTextSegmenter = __decorate([
|
|
198
|
+
RegisterClass(BaseSegmenter, SEMANTIC_TEXT_SEGMENTER_KEY)
|
|
199
|
+
], SemanticTextSegmenter);
|
|
200
|
+
export { SemanticTextSegmenter };
|
|
201
|
+
//# sourceMappingURL=SemanticTextSegmenter.js.map
|