@memberjunction/ai-segmentation 0.0.0 → 5.51.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (54) hide show
  1. package/README.md +177 -43
  2. package/dist/generic/AdaptiveBoundarySegmenter.d.ts +98 -0
  3. package/dist/generic/AdaptiveBoundarySegmenter.d.ts.map +1 -0
  4. package/dist/generic/AdaptiveBoundarySegmenter.js +177 -0
  5. package/dist/generic/AdaptiveBoundarySegmenter.js.map +1 -0
  6. package/dist/generic/BaseContentCleaner.d.ts +101 -0
  7. package/dist/generic/BaseContentCleaner.d.ts.map +1 -0
  8. package/dist/generic/BaseContentCleaner.js +114 -0
  9. package/dist/generic/BaseContentCleaner.js.map +1 -0
  10. package/dist/generic/BaseSegmenter.d.ts +106 -0
  11. package/dist/generic/BaseSegmenter.d.ts.map +1 -0
  12. package/dist/generic/BaseSegmenter.js +260 -0
  13. package/dist/generic/BaseSegmenter.js.map +1 -0
  14. package/dist/generic/FixedWindowSegmenter.d.ts +49 -0
  15. package/dist/generic/FixedWindowSegmenter.d.ts.map +1 -0
  16. package/dist/generic/FixedWindowSegmenter.js +106 -0
  17. package/dist/generic/FixedWindowSegmenter.js.map +1 -0
  18. package/dist/generic/HtmlContentCleaner.d.ts +61 -0
  19. package/dist/generic/HtmlContentCleaner.d.ts.map +1 -0
  20. package/dist/generic/HtmlContentCleaner.js +130 -0
  21. package/dist/generic/HtmlContentCleaner.js.map +1 -0
  22. package/dist/generic/PagedContentSegmenter.d.ts +55 -0
  23. package/dist/generic/PagedContentSegmenter.d.ts.map +1 -0
  24. package/dist/generic/PagedContentSegmenter.js +99 -0
  25. package/dist/generic/PagedContentSegmenter.js.map +1 -0
  26. package/dist/generic/PlainTextContentCleaner.d.ts +22 -0
  27. package/dist/generic/PlainTextContentCleaner.d.ts.map +1 -0
  28. package/dist/generic/PlainTextContentCleaner.js +37 -0
  29. package/dist/generic/PlainTextContentCleaner.js.map +1 -0
  30. package/dist/generic/Segmentation.types.d.ts +198 -0
  31. package/dist/generic/Segmentation.types.d.ts.map +1 -0
  32. package/dist/generic/Segmentation.types.js +19 -0
  33. package/dist/generic/Segmentation.types.js.map +1 -0
  34. package/dist/generic/SegmentationResolver.d.ts +43 -0
  35. package/dist/generic/SegmentationResolver.d.ts.map +1 -0
  36. package/dist/generic/SegmentationResolver.js +83 -0
  37. package/dist/generic/SegmentationResolver.js.map +1 -0
  38. package/dist/generic/SemanticTextSegmenter.d.ts +80 -0
  39. package/dist/generic/SemanticTextSegmenter.d.ts.map +1 -0
  40. package/dist/generic/SemanticTextSegmenter.js +201 -0
  41. package/dist/generic/SemanticTextSegmenter.js.map +1 -0
  42. package/dist/generic/StructuralTextSegmenter.d.ts +63 -0
  43. package/dist/generic/StructuralTextSegmenter.d.ts.map +1 -0
  44. package/dist/generic/StructuralTextSegmenter.js +177 -0
  45. package/dist/generic/StructuralTextSegmenter.js.map +1 -0
  46. package/dist/generic/TranscriptSegmenter.d.ts +78 -0
  47. package/dist/generic/TranscriptSegmenter.d.ts.map +1 -0
  48. package/dist/generic/TranscriptSegmenter.js +194 -0
  49. package/dist/generic/TranscriptSegmenter.js.map +1 -0
  50. package/dist/index.d.ts +33 -0
  51. package/dist/index.d.ts.map +1 -0
  52. package/dist/index.js +36 -0
  53. package/dist/index.js.map +1 -0
  54. package/package.json +33 -7
@@ -0,0 +1,198 @@
1
+ /**
2
+ * @fileoverview Type contract for content segmentation.
3
+ *
4
+ * Segmentation is the step that turns a piece of source content (a document, a
5
+ * recording, an image) into an ordered list of embeddable {@link ContentSegment}s.
6
+ * It sits *upstream* of embedding: a segmenter decides **what** gets embedded,
7
+ * the embedding model decides **how**.
8
+ *
9
+ * This is deliberately separate from `TextChunker` (in `@memberjunction/ai-vectors`). `TextChunker` answers
10
+ * "how do I split this string so it fits a token budget"; a segmenter answers
11
+ * "what are the meaningful units of this content" — which may be sections of a
12
+ * document, chapters of a recording, or a single image. Segmenters typically
13
+ * *use* `TextChunker` to enforce the token budget within a unit they identified.
14
+ *
15
+ * @module @memberjunction/ai-segmentation
16
+ */
17
+ import { UserInfo } from '@memberjunction/core';
18
+ /**
19
+ * The modality of a segment's payload. Drives downstream index routing
20
+ * (text vs. multimodal vector index) and retrieval-time fusion.
21
+ */
22
+ export type ContentModality = 'text' | 'image' | 'audio' | 'video' | 'multimodal';
23
+ /**
24
+ * A pointer to non-text content. Segmenters emit these for media segments
25
+ * instead of (or alongside) `Text`.
26
+ *
27
+ * Exactly one of `URL`, `Base64Data`, or (`StorageProviderKey` + `ObjectKey`)
28
+ * is expected to be populated; consumers resolve whichever is present.
29
+ */
30
+ export interface MediaReference {
31
+ /** Directly fetchable URL (http(s) or data URL). */
32
+ URL?: string;
33
+ /** `@memberjunction/storage` provider key, when the media lives in MJ file storage. */
34
+ StorageProviderKey?: string;
35
+ /** Object/blob key within the storage provider. */
36
+ ObjectKey?: string;
37
+ /** IANA mime type, e.g. `video/mp4`. Used to gate provider capability checks. */
38
+ MimeType?: string;
39
+ /** Inline base64 payload (no data-URL prefix). Prefer URL/storage refs for large media. */
40
+ Base64Data?: string;
41
+ }
42
+ /**
43
+ * A single timed transcript cue — the unit produced by ASR and by MJ's
44
+ * realtime session capture (which records speaker + timings per turn).
45
+ */
46
+ export interface TranscriptCue {
47
+ /** Cue start, milliseconds from the beginning of the asset. */
48
+ StartMs: number;
49
+ /** Cue end, milliseconds from the beginning of the asset. */
50
+ EndMs: number;
51
+ /** Spoken text for this cue. */
52
+ Text: string;
53
+ /** Optional speaker label/id — a speaker change is a strong boundary signal. */
54
+ Speaker?: string;
55
+ }
56
+ /**
57
+ * One embeddable unit of content produced by a segmenter.
58
+ *
59
+ * A segment carries `Text`, `Media`, or **both** (the "dual representation" case:
60
+ * a video chapter with a native media reference *and* its transcript, so it can be
61
+ * embedded natively for retrieval while remaining readable for an agent).
62
+ */
63
+ export interface ContentSegment {
64
+ /** 0-based position within the full segment list. Assigned by {@link BaseSegmenter}. */
65
+ Sequence: number;
66
+ /** Payload modality. */
67
+ Modality: ContentModality;
68
+ /** Textual payload — the extracted/transcribed text for this segment. */
69
+ Text?: string;
70
+ /** Media payload pointer, for non-text segments. */
71
+ Media?: MediaReference;
72
+ /** Human-readable label, e.g. a heading or a generated chapter title. */
73
+ Title?: string;
74
+ /** Inclusive start character offset within the source text. */
75
+ StartOffset?: number;
76
+ /** Exclusive end character offset within the source text. */
77
+ EndOffset?: number;
78
+ /** Segment start in milliseconds, for audio/video. */
79
+ StartMs?: number;
80
+ /** Segment end in milliseconds, for audio/video. */
81
+ EndMs?: number;
82
+ /** 1-based page number, for paginated sources (PDF, slides). */
83
+ PageNumber?: number;
84
+ /** `Sequence` of this segment's parent, for chapter -> sub-chapter hierarchies. */
85
+ ParentSequence?: number;
86
+ /** Nesting depth; 0 for top-level segments. */
87
+ Depth: number;
88
+ /** Estimated token count of `Text` (0 for pure-media segments). */
89
+ TokenEstimate: number;
90
+ /** Registration key of the segmenter that produced this segment — provenance. */
91
+ SegmenterKey: string;
92
+ /** Speaker label carried through from transcript cues, when known. */
93
+ Speaker?: string;
94
+ }
95
+ /**
96
+ * A segment as emitted by a concrete segmenter's `SegmentCore`, before the base
97
+ * class normalizes it (assigns `Sequence`/`Depth`/`TokenEstimate`, enforces the
98
+ * token ceiling, and resolves parent links).
99
+ *
100
+ * `ParentIndex` refers to the **index within the raw array** returned by
101
+ * `SegmentCore`; the base class remaps it to a real `Sequence` afterwards, which
102
+ * keeps subclasses from having to reason about post-split numbering.
103
+ */
104
+ export interface RawSegment {
105
+ Modality: ContentModality;
106
+ Text?: string;
107
+ Media?: MediaReference;
108
+ Title?: string;
109
+ StartOffset?: number;
110
+ EndOffset?: number;
111
+ StartMs?: number;
112
+ EndMs?: number;
113
+ PageNumber?: number;
114
+ Speaker?: string;
115
+ /** Index into the raw segment array identifying this segment's parent. */
116
+ ParentIndex?: number;
117
+ }
118
+ /**
119
+ * One page of a paginated source (PDF, slide deck).
120
+ *
121
+ * A page may carry extracted text, a media reference to the rendered page, or both —
122
+ * the both case is what lets a page be embedded natively by a multimodal model (preserving
123
+ * tables and charts that text extraction flattens) while its text remains available for
124
+ * lexical search.
125
+ */
126
+ export interface ContentPage {
127
+ /** One-based page number. */
128
+ PageNumber: number;
129
+ /** Extracted text for this page. */
130
+ Text?: string;
131
+ /** Reference to the rendered page image, when available. */
132
+ Media?: MediaReference;
133
+ }
134
+ /**
135
+ * Common knobs understood by every segmenter. Concrete segmenters extend this
136
+ * with their own strongly-typed options rather than accepting a loose bag.
137
+ */
138
+ export interface SegmentationOptions {
139
+ /**
140
+ * Hard ceiling on tokens per text segment. The base class splits any
141
+ * oversized segment via `TextChunker` so no segmenter can exceed it.
142
+ * Default: 512.
143
+ */
144
+ MaxSegmentTokens?: number;
145
+ /** Overlap tokens applied when an oversized segment must be split. Default: 10% of max. */
146
+ OverlapTokens?: number;
147
+ /**
148
+ * Segments whose text estimates below this many tokens are merged forward into
149
+ * the next segment, preventing a spray of near-empty vectors. Default: 0 (off).
150
+ */
151
+ MinSegmentTokens?: number;
152
+ }
153
+ /**
154
+ * Input to {@link BaseSegmenter.Segment}.
155
+ *
156
+ * @typeParam TOptions - the concrete segmenter's options type.
157
+ */
158
+ export interface SegmentationParams<TOptions extends SegmentationOptions = SegmentationOptions> {
159
+ /** Extracted text of the source content, when it has any. */
160
+ Text?: string;
161
+ /** Media pointer for the source asset, for image/audio/video content. */
162
+ Media?: MediaReference;
163
+ /** Timed transcript cues, when available (ASR output or MJ realtime capture). */
164
+ Cues?: TranscriptCue[];
165
+ /** Pages of a paginated source, for page-aware segmentation. */
166
+ Pages?: ContentPage[];
167
+ /** Total duration of the source asset in milliseconds, for AV content. */
168
+ DurationMs?: number;
169
+ /** Mime type of the source asset — lets a segmenter pick a structure parser. */
170
+ MimeType?: string;
171
+ /**
172
+ * Context user, required by segmenters that call MJ services (e.g. the LLM
173
+ * boundary pass in `SemanticTextSegmenter`). Always pass it in server-side code.
174
+ */
175
+ ContextUser?: UserInfo;
176
+ /** Strategy-specific options. */
177
+ Options?: TOptions;
178
+ }
179
+ /**
180
+ * Result of a segmentation pass. Segmenters never throw for content-shaped
181
+ * problems — they return `Success: false` with an `ErrorMessage`, matching the
182
+ * convention used by `RunView` and `BaseEntity.Save`.
183
+ */
184
+ export interface SegmentationResult {
185
+ /** False when segmentation could not be performed. */
186
+ Success: boolean;
187
+ /** The produced segments, in order. Empty when `Success` is false. */
188
+ Segments: ContentSegment[];
189
+ /** Registration key of the segmenter that ran. */
190
+ SegmenterKey: string;
191
+ /** Populated when `Success` is false. */
192
+ ErrorMessage?: string;
193
+ /** Non-fatal notes — e.g. "no cues supplied, fell back to fixed windows". */
194
+ Warnings: string[];
195
+ }
196
+ /** Default token ceiling applied when a caller does not specify one. */
197
+ export declare const DEFAULT_MAX_SEGMENT_TOKENS = 512;
198
+ //# sourceMappingURL=Segmentation.types.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"Segmentation.types.d.ts","sourceRoot":"","sources":["../../src/generic/Segmentation.types.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;GAeG;AAEH,OAAO,EAAE,QAAQ,EAAE,MAAM,sBAAsB,CAAC;AAEhD;;;GAGG;AACH,MAAM,MAAM,eAAe,GAAG,MAAM,GAAG,OAAO,GAAG,OAAO,GAAG,OAAO,GAAG,YAAY,CAAC;AAElF;;;;;;GAMG;AACH,MAAM,WAAW,cAAc;IAC3B,oDAAoD;IACpD,GAAG,CAAC,EAAE,MAAM,CAAC;IACb,uFAAuF;IACvF,kBAAkB,CAAC,EAAE,MAAM,CAAC;IAC5B,mDAAmD;IACnD,SAAS,CAAC,EAAE,MAAM,CAAC;IACnB,iFAAiF;IACjF,QAAQ,CAAC,EAAE,MAAM,CAAC;IAClB,2FAA2F;IAC3F,UAAU,CAAC,EAAE,MAAM,CAAC;CACvB;AAED;;;GAGG;AACH,MAAM,WAAW,aAAa;IAC1B,+DAA+D;IAC/D,OAAO,EAAE,MAAM,CAAC;IAChB,6DAA6D;IAC7D,KAAK,EAAE,MAAM,CAAC;IACd,gCAAgC;IAChC,IAAI,EAAE,MAAM,CAAC;IACb,gFAAgF;IAChF,OAAO,CAAC,EAAE,MAAM,CAAC;CACpB;AAED;;;;;;GAMG;AACH,MAAM,WAAW,cAAc;IAC3B,wFAAwF;IACxF,QAAQ,EAAE,MAAM,CAAC;IACjB,wBAAwB;IACxB,QAAQ,EAAE,eAAe,CAAC;IAC1B,yEAAyE;IACzE,IAAI,CAAC,EAAE,MAAM,CAAC;IACd,oDAAoD;IACpD,KAAK,CAAC,EAAE,cAAc,CAAC;IACvB,yEAAyE;IACzE,KAAK,CAAC,EAAE,MAAM,CAAC;IACf,+DAA+D;IAC/D,WAAW,CAAC,EAAE,MAAM,CAAC;IACrB,6DAA6D;IAC7D,SAAS,CAAC,EAAE,MAAM,CAAC;IACnB,sDAAsD;IACtD,OAAO,CAAC,EAAE,MAAM,CAAC;IACjB,oDAAoD;IACpD,KAAK,CAAC,EAAE,MAAM,CAAC;IACf,gEAAgE;IAChE,UAAU,CAAC,EAAE,MAAM,CAAC;IACpB,mFAAmF;IACnF,cAAc,CAAC,EAAE,MAAM,CAAC;IACxB,+CAA+C;IAC/C,KAAK,EAAE,MAAM,CAAC;IACd,mEAAmE;IACnE,aAAa,EAAE,MAAM,CAAC;IACtB,iFAAiF;IACjF,YAAY,EAAE,MAAM,CAAC;IACrB,sEAAsE;IACtE,OAAO,CAAC,EAAE,MAAM,CAAC;CACpB;AAED;;;;;;;;GAQG;AACH,MAAM,WAAW,UAAU;IACvB,QAAQ,EAAE,eAAe,CAAC;IAC1B,IAAI,CAAC,EAAE,MAAM,CAAC;IACd,KAAK,CAAC,EAAE,cAAc,CAAC;IACvB,KAAK,CAAC,EAAE,MAAM,CAAC;IACf,WAAW,CAAC,EAAE,MAAM,CAAC;IACrB,SAAS,CAAC,EAAE,MAAM,CAAC;IACnB,OAAO,CAAC,EAAE,MAAM,CAAC;IACjB,KAAK,CAAC,EAAE,MAAM,CAAC;IACf,UAAU,CAAC,EAAE,MAAM,CAAC;IACpB,OAAO,CAAC,EAAE,MAAM,CAAC;IACjB,0EAA0E;IAC1E,WAAW,CAAC,EAAE,MAAM,CAAC;CACxB;AAED;;;;;;;GAOG;AACH,MAAM,WAAW,WAAW;IACxB,6BAA6B;IAC7B,UAAU,EAAE,MAAM,CAAC;IACnB,oCAAoC;IACpC,IAAI,CAAC,EAAE,MAAM,CAAC;IACd,4DAA4D;IAC5D,KAAK,CAAC,EAAE,cAAc,CAAC;CAC1B;AAED;;;GAGG;AACH,MAAM,WAAW,mBAAmB;IAChC;;;;OAIG;IACH,gBAAgB,CAAC,EAAE,MAAM,CAAC;IAC1B,2FAA2F;IAC3F,aAAa,CAAC,EAAE,MAAM,CAAC;IACvB;;;OAGG;IACH,gBAAgB,CAAC,EAAE,MAAM,CAAC;CAC7B;AAED;;;;GAIG;AACH,MAAM,WAAW,kBAAkB,CAAC,QAAQ,SAAS,mBAAmB,GAAG,mBAAmB;IAC1F,6DAA6D;IAC7D,IAAI,CAAC,EAAE,MAAM,CAAC;IACd,yEAAyE;IACzE,KAAK,CAAC,EAAE,cAAc,CAAC;IACvB,iFAAiF;IACjF,IAAI,CAAC,EAAE,aAAa,EAAE,CAAC;IACvB,gEAAgE;IAChE,KAAK,CAAC,EAAE,WAAW,EAAE,CAAC;IACtB,0EAA0E;IAC1E,UAAU,CAAC,EAAE,MAAM,CAAC;IACpB,gFAAgF;IAChF,QAAQ,CAAC,EAAE,MAAM,CAAC;IAClB;;;OAGG;IACH,WAAW,CAAC,EAAE,QAAQ,CAAC;IACvB,iCAAiC;IACjC,OAAO,CAAC,EAAE,QAAQ,CAAC;CACtB;AAED;;;;GAIG;AACH,MAAM,WAAW,kBAAkB;IAC/B,sDAAsD;IACtD,OAAO,EAAE,OAAO,CAAC;IACjB,sEAAsE;IACtE,QAAQ,EAAE,cAAc,EAAE,CAAC;IAC3B,kDAAkD;IAClD,YAAY,EAAE,MAAM,CAAC;IACrB,yCAAyC;IACzC,YAAY,CAAC,EAAE,MAAM,CAAC;IACtB,6EAA6E;IAC7E,QAAQ,EAAE,MAAM,EAAE,CAAC;CACtB;AAED,wEAAwE;AACxE,eAAO,MAAM,0BAA0B,MAAM,CAAC"}
@@ -0,0 +1,19 @@
1
+ /**
2
+ * @fileoverview Type contract for content segmentation.
3
+ *
4
+ * Segmentation is the step that turns a piece of source content (a document, a
5
+ * recording, an image) into an ordered list of embeddable {@link ContentSegment}s.
6
+ * It sits *upstream* of embedding: a segmenter decides **what** gets embedded,
7
+ * the embedding model decides **how**.
8
+ *
9
+ * This is deliberately separate from `TextChunker` (in `@memberjunction/ai-vectors`). `TextChunker` answers
10
+ * "how do I split this string so it fits a token budget"; a segmenter answers
11
+ * "what are the meaningful units of this content" — which may be sections of a
12
+ * document, chapters of a recording, or a single image. Segmenters typically
13
+ * *use* `TextChunker` to enforce the token budget within a unit they identified.
14
+ *
15
+ * @module @memberjunction/ai-segmentation
16
+ */
17
+ /** Default token ceiling applied when a caller does not specify one. */
18
+ export const DEFAULT_MAX_SEGMENT_TOKENS = 512;
19
+ //# sourceMappingURL=Segmentation.types.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"Segmentation.types.js","sourceRoot":"","sources":["../../src/generic/Segmentation.types.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;GAeG;AA+LH,wEAAwE;AACxE,MAAM,CAAC,MAAM,0BAA0B,GAAG,GAAG,CAAC"}
@@ -0,0 +1,43 @@
1
+ /**
2
+ * @fileoverview Helpers for selecting a segmenter from configuration or content shape.
3
+ *
4
+ * @module @memberjunction/ai-segmentation
5
+ */
6
+ import { BaseSegmenter } from './BaseSegmenter.js';
7
+ import { BaseContentCleaner } from './BaseContentCleaner.js';
8
+ import { SegmentationParams } from './Segmentation.types.js';
9
+ /**
10
+ * Resolve a segmenter by registration key, falling back safely.
11
+ *
12
+ * Configuration is data, and data drifts — a `Content Type` may name a segmenter
13
+ * that has been renamed or lives in a package this process didn't load. Rather
14
+ * than throw mid-ingestion, an unresolvable key logs and degrades to the
15
+ * fixed-window segmenter, which can segment anything.
16
+ *
17
+ * @param key - registration key from metadata; when omitted the fallback is used.
18
+ * @param fallbackKey - key to try before the built-in last resort.
19
+ */
20
+ export declare function ResolveSegmenter(key?: string, fallbackKey?: string): BaseSegmenter;
21
+ /**
22
+ * Suggest the best-fit segmenter key for a piece of content.
23
+ *
24
+ * The ordering encodes the quality hierarchy: a real transcript beats document
25
+ * structure, which beats uniform windows. Callers should treat this as a default
26
+ * that explicit configuration may override.
27
+ */
28
+ export declare function SuggestSegmenterKey(params: SegmentationParams): string;
29
+ /**
30
+ * Resolve a content cleaner by registration key, falling back safely.
31
+ *
32
+ * Mirrors {@link ResolveSegmenter}: an unresolvable key logs and degrades to the
33
+ * plain-text cleaner (whitespace normalization only) rather than throwing mid-ingestion.
34
+ */
35
+ export declare function ResolveContentCleaner(key?: string, fallbackKey?: string): BaseContentCleaner;
36
+ /**
37
+ * Suggest a cleaner for a piece of content based on its mime type.
38
+ *
39
+ * HTML is the only format that genuinely needs structural cleaning; everything else is
40
+ * already text and only wants whitespace normalization.
41
+ */
42
+ export declare function SuggestCleanerKey(mimeType?: string): string;
43
+ //# sourceMappingURL=SegmentationResolver.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"SegmentationResolver.d.ts","sourceRoot":"","sources":["../../src/generic/SegmentationResolver.ts"],"names":[],"mappings":"AAAA;;;;GAIG;AAGH,OAAO,EAAE,aAAa,EAAE,MAAM,iBAAiB,CAAC;AAKhD,OAAO,EAAE,kBAAkB,EAAE,MAAM,sBAAsB,CAAC;AAG1D,OAAO,EAAE,kBAAkB,EAAE,MAAM,sBAAsB,CAAC;AAE1D;;;;;;;;;;GAUG;AACH,wBAAgB,gBAAgB,CAAC,GAAG,CAAC,EAAE,MAAM,EAAE,WAAW,CAAC,EAAE,MAAM,GAAG,aAAa,CAUlF;AAED;;;;;;GAMG;AACH,wBAAgB,mBAAmB,CAAC,MAAM,EAAE,kBAAkB,GAAG,MAAM,CAWtE;AAED;;;;;GAKG;AACH,wBAAgB,qBAAqB,CAAC,GAAG,CAAC,EAAE,MAAM,EAAE,WAAW,CAAC,EAAE,MAAM,GAAG,kBAAkB,CAU5F;AAED;;;;;GAKG;AACH,wBAAgB,iBAAiB,CAAC,QAAQ,CAAC,EAAE,MAAM,GAAG,MAAM,CAG3D"}
@@ -0,0 +1,83 @@
1
+ /**
2
+ * @fileoverview Helpers for selecting a segmenter from configuration or content shape.
3
+ *
4
+ * @module @memberjunction/ai-segmentation
5
+ */
6
+ import { LogStatus } from '@memberjunction/core';
7
+ import { BaseSegmenter } from './BaseSegmenter.js';
8
+ import { FIXED_WINDOW_SEGMENTER_KEY, FixedWindowSegmenter } from './FixedWindowSegmenter.js';
9
+ import { STRUCTURAL_TEXT_SEGMENTER_KEY } from './StructuralTextSegmenter.js';
10
+ import { TRANSCRIPT_SEGMENTER_KEY } from './TranscriptSegmenter.js';
11
+ import { PAGED_CONTENT_SEGMENTER_KEY } from './PagedContentSegmenter.js';
12
+ import { BaseContentCleaner } from './BaseContentCleaner.js';
13
+ import { HTML_CONTENT_CLEANER_KEY } from './HtmlContentCleaner.js';
14
+ import { PLAIN_TEXT_CONTENT_CLEANER_KEY, PlainTextContentCleaner } from './PlainTextContentCleaner.js';
15
+ /**
16
+ * Resolve a segmenter by registration key, falling back safely.
17
+ *
18
+ * Configuration is data, and data drifts — a `Content Type` may name a segmenter
19
+ * that has been renamed or lives in a package this process didn't load. Rather
20
+ * than throw mid-ingestion, an unresolvable key logs and degrades to the
21
+ * fixed-window segmenter, which can segment anything.
22
+ *
23
+ * @param key - registration key from metadata; when omitted the fallback is used.
24
+ * @param fallbackKey - key to try before the built-in last resort.
25
+ */
26
+ export function ResolveSegmenter(key, fallbackKey) {
27
+ const requested = key ? BaseSegmenter.Resolve(key) : null;
28
+ if (requested) {
29
+ return requested;
30
+ }
31
+ if (key) {
32
+ LogStatus(`[Segmentation] Segmenter '${key}' is not registered — falling back.`);
33
+ }
34
+ const fallback = fallbackKey ? BaseSegmenter.Resolve(fallbackKey) : null;
35
+ return fallback ?? BaseSegmenter.Resolve(FIXED_WINDOW_SEGMENTER_KEY) ?? new FixedWindowSegmenter();
36
+ }
37
+ /**
38
+ * Suggest the best-fit segmenter key for a piece of content.
39
+ *
40
+ * The ordering encodes the quality hierarchy: a real transcript beats document
41
+ * structure, which beats uniform windows. Callers should treat this as a default
42
+ * that explicit configuration may override.
43
+ */
44
+ export function SuggestSegmenterKey(params) {
45
+ if (params.Cues && params.Cues.length > 0) {
46
+ return TRANSCRIPT_SEGMENTER_KEY;
47
+ }
48
+ if (params.Pages && params.Pages.length > 0) {
49
+ return PAGED_CONTENT_SEGMENTER_KEY;
50
+ }
51
+ if (params.Text && params.Text.trim().length > 0) {
52
+ return STRUCTURAL_TEXT_SEGMENTER_KEY;
53
+ }
54
+ return FIXED_WINDOW_SEGMENTER_KEY;
55
+ }
56
+ /**
57
+ * Resolve a content cleaner by registration key, falling back safely.
58
+ *
59
+ * Mirrors {@link ResolveSegmenter}: an unresolvable key logs and degrades to the
60
+ * plain-text cleaner (whitespace normalization only) rather than throwing mid-ingestion.
61
+ */
62
+ export function ResolveContentCleaner(key, fallbackKey) {
63
+ const requested = key ? BaseContentCleaner.Resolve(key) : null;
64
+ if (requested) {
65
+ return requested;
66
+ }
67
+ if (key) {
68
+ LogStatus(`[Segmentation] Content cleaner '${key}' is not registered — falling back.`);
69
+ }
70
+ const fallback = fallbackKey ? BaseContentCleaner.Resolve(fallbackKey) : null;
71
+ return fallback ?? BaseContentCleaner.Resolve(PLAIN_TEXT_CONTENT_CLEANER_KEY) ?? new PlainTextContentCleaner();
72
+ }
73
+ /**
74
+ * Suggest a cleaner for a piece of content based on its mime type.
75
+ *
76
+ * HTML is the only format that genuinely needs structural cleaning; everything else is
77
+ * already text and only wants whitespace normalization.
78
+ */
79
+ export function SuggestCleanerKey(mimeType) {
80
+ const mime = (mimeType ?? '').toLowerCase();
81
+ return mime.includes('html') || mime.includes('xml') ? HTML_CONTENT_CLEANER_KEY : PLAIN_TEXT_CONTENT_CLEANER_KEY;
82
+ }
83
+ //# sourceMappingURL=SegmentationResolver.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"SegmentationResolver.js","sourceRoot":"","sources":["../../src/generic/SegmentationResolver.ts"],"names":[],"mappings":"AAAA;;;;GAIG;AAEH,OAAO,EAAE,SAAS,EAAE,MAAM,sBAAsB,CAAC;AACjD,OAAO,EAAE,aAAa,EAAE,MAAM,iBAAiB,CAAC;AAChD,OAAO,EAAE,0BAA0B,EAAE,oBAAoB,EAAE,MAAM,wBAAwB,CAAC;AAC1F,OAAO,EAAE,6BAA6B,EAAE,MAAM,2BAA2B,CAAC;AAC1E,OAAO,EAAE,wBAAwB,EAAE,MAAM,uBAAuB,CAAC;AACjE,OAAO,EAAE,2BAA2B,EAAE,MAAM,yBAAyB,CAAC;AACtE,OAAO,EAAE,kBAAkB,EAAE,MAAM,sBAAsB,CAAC;AAC1D,OAAO,EAAE,wBAAwB,EAAE,MAAM,sBAAsB,CAAC;AAChE,OAAO,EAAE,8BAA8B,EAAE,uBAAuB,EAAE,MAAM,2BAA2B,CAAC;AAGpG;;;;;;;;;;GAUG;AACH,MAAM,UAAU,gBAAgB,CAAC,GAAY,EAAE,WAAoB;IAC/D,MAAM,SAAS,GAAG,GAAG,CAAC,CAAC,CAAC,aAAa,CAAC,OAAO,CAAC,GAAG,CAAC,CAAC,CAAC,CAAC,IAAI,CAAC;IAC1D,IAAI,SAAS,EAAE,CAAC;QACZ,OAAO,SAAS,CAAC;IACrB,CAAC;IACD,IAAI,GAAG,EAAE,CAAC;QACN,SAAS,CAAC,6BAA6B,GAAG,qCAAqC,CAAC,CAAC;IACrF,CAAC;IACD,MAAM,QAAQ,GAAG,WAAW,CAAC,CAAC,CAAC,aAAa,CAAC,OAAO,CAAC,WAAW,CAAC,CAAC,CAAC,CAAC,IAAI,CAAC;IACzE,OAAO,QAAQ,IAAI,aAAa,CAAC,OAAO,CAAC,0BAA0B,CAAC,IAAI,IAAI,oBAAoB,EAAE,CAAC;AACvG,CAAC;AAED;;;;;;GAMG;AACH,MAAM,UAAU,mBAAmB,CAAC,MAA0B;IAC1D,IAAI,MAAM,CAAC,IAAI,IAAI,MAAM,CAAC,IAAI,CAAC,MAAM,GAAG,CAAC,EAAE,CAAC;QACxC,OAAO,wBAAwB,CAAC;IACpC,CAAC;IACD,IAAI,MAAM,CAAC,KAAK,IAAI,MAAM,CAAC,KAAK,CAAC,MAAM,GAAG,CAAC,EAAE,CAAC;QAC1C,OAAO,2BAA2B,CAAC;IACvC,CAAC;IACD,IAAI,MAAM,CAAC,IAAI,IAAI,MAAM,CAAC,IAAI,CAAC,IAAI,EAAE,CAAC,MAAM,GAAG,CAAC,EAAE,CAAC;QAC/C,OAAO,6BAA6B,CAAC;IACzC,CAAC;IACD,OAAO,0BAA0B,CAAC;AACtC,CAAC;AAED;;;;;GAKG;AACH,MAAM,UAAU,qBAAqB,CAAC,GAAY,EAAE,WAAoB;IACpE,MAAM,SAAS,GAAG,GAAG,CAAC,CAAC,CAAC,kBAAkB,CAAC,OAAO,CAAC,GAAG,CAAC,CAAC,CAAC,CAAC,IAAI,CAAC;IAC/D,IAAI,SAAS,EAAE,CAAC;QACZ,OAAO,SAAS,CAAC;IACrB,CAAC;IACD,IAAI,GAAG,EAAE,CAAC;QACN,SAAS,CAAC,mCAAmC,GAAG,qCAAqC,CAAC,CAAC;IAC3F,CAAC;IACD,MAAM,QAAQ,GAAG,WAAW,CAAC,CAAC,CAAC,kBAAkB,CAAC,OAAO,CAAC,WAAW,CAAC,CAAC,CAAC,CAAC,IAAI,CAAC;IAC9E,OAAO,QAAQ,IAAI,kBAAkB,CAAC,OAAO,CAAC,8BAA8B,CAAC,IAAI,IAAI,uBAAuB,EAAE,CAAC;AACnH,CAAC;AAED;;;;;GAKG;AACH,MAAM,UAAU,iBAAiB,CAAC,QAAiB;IAC/C,MAAM,IAAI,GAAG,CAAC,QAAQ,IAAI,EAAE,CAAC,CAAC,WAAW,EAAE,CAAC;IAC5C,OAAO,IAAI,CAAC,QAAQ,CAAC,MAAM,CAAC,IAAI,IAAI,CAAC,QAAQ,CAAC,KAAK,CAAC,CAAC,CAAC,CAAC,wBAAwB,CAAC,CAAC,CAAC,8BAA8B,CAAC;AACrH,CAAC"}
@@ -0,0 +1,80 @@
1
+ /**
2
+ * @fileoverview LLM-driven semantic segmenter — finds topic boundaries in prose.
3
+ *
4
+ * @module @memberjunction/ai-segmentation
5
+ */
6
+ import { BaseSegmenter } from './BaseSegmenter.js';
7
+ import { ContentModality, RawSegment, SegmentationOptions, SegmentationParams } from './Segmentation.types.js';
8
+ /** Registration key for {@link SemanticTextSegmenter}. */
9
+ export declare const SEMANTIC_TEXT_SEGMENTER_KEY = "SemanticText";
10
+ /** Default name of the `MJ: AI Prompts` record driving boundary detection. */
11
+ export declare const SEMANTIC_SEGMENTATION_PROMPT_NAME = "Content Semantic Segmentation";
12
+ /** Options specific to {@link SemanticTextSegmenter}. */
13
+ export interface SemanticTextSegmentationOptions extends SegmentationOptions {
14
+ /** Name of the `MJ: AI Prompts` record to run. Default: {@link SEMANTIC_SEGMENTATION_PROMPT_NAME}. */
15
+ PromptName?: string;
16
+ /** Optional model override (an `MJ: AI Models` ID) for the boundary pass. */
17
+ ModelID?: string;
18
+ /**
19
+ * Skip the LLM call entirely when the document estimates below this many
20
+ * tokens — short documents rarely contain multiple topics and the call would
21
+ * not repay its cost. Default: 750.
22
+ */
23
+ MinTokensForLLM?: number;
24
+ /**
25
+ * Maximum characters of each block shown to the model. Boundary detection only
26
+ * needs the opening of a block, so truncating keeps the prompt cheap on long
27
+ * documents. Default: 240.
28
+ */
29
+ BlockPreviewChars?: number;
30
+ /** Maximum blocks sent in one pass. Default: 300. */
31
+ MaxBlocks?: number;
32
+ }
33
+ /**
34
+ * Segments prose by asking an LLM where the topics change.
35
+ *
36
+ * Structure-aware segmentation only works when the author left structure behind.
37
+ * Transcripts, scanned reports, and long-form articles frequently have none — the
38
+ * topic shifts, but no heading marks it. This segmenter finds those latent
39
+ * boundaries and names them, producing titled sections that behave like headings
40
+ * the document never had.
41
+ *
42
+ * ## Cost posture
43
+ *
44
+ * The LLM pass is the expensive part of ingestion, so this class is written to
45
+ * avoid it whenever it wouldn't pay off: short documents short-circuit to
46
+ * structural segmentation, blocks are truncated to a preview before being shown to
47
+ * the model, and any failure degrades to `StructuralText` rather than failing the
48
+ * ingestion run. The model is asked to classify *block indices*, never character
49
+ * offsets — models are unreliable at arithmetic over long strings, and a wrong
50
+ * offset would silently corrupt chunk provenance.
51
+ *
52
+ * Because it runs through `AIPromptRunner`, every pass is a tracked `MJ: AI Prompt
53
+ * Run` with full token and cost attribution, and the prompt itself is versioned
54
+ * metadata rather than a string literal in code.
55
+ */
56
+ export declare class SemanticTextSegmenter extends BaseSegmenter {
57
+ get Key(): string;
58
+ get SupportedModalities(): ContentModality[];
59
+ protected SegmentCore(params: SegmentationParams<SemanticTextSegmentationOptions>): Promise<RawSegment[]>;
60
+ /** Split the document into paragraph blocks with real offsets. */
61
+ private buildBlocks;
62
+ /** Append a block when the slice has content. */
63
+ private pushBlock;
64
+ /** Run the segmentation prompt and return validated boundary block indices. */
65
+ private proposeBoundaries;
66
+ /** Render numbered, truncated blocks for the prompt. */
67
+ private renderBlocks;
68
+ /** Accept either a parsed object or a JSON string from the runner. */
69
+ private parseResult;
70
+ /**
71
+ * Clamp, dedupe, and sort model output. A hallucinated or out-of-range block
72
+ * index must never become a chunk offset, so anything unusable is dropped here.
73
+ */
74
+ private sanitizeBoundaries;
75
+ /** Join blocks between consecutive boundaries into one segment each. */
76
+ private buildSegments;
77
+ /** Degrade to structural segmentation — never fail an ingestion run over segmentation. */
78
+ private fallback;
79
+ }
80
+ //# sourceMappingURL=SemanticTextSegmenter.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"SemanticTextSegmenter.d.ts","sourceRoot":"","sources":["../../src/generic/SemanticTextSegmenter.ts"],"names":[],"mappings":"AAAA;;;;GAIG;AAOH,OAAO,EAAE,aAAa,EAAE,MAAM,iBAAiB,CAAC;AAEhD,OAAO,EAAE,eAAe,EAAE,UAAU,EAAE,mBAAmB,EAAE,kBAAkB,EAAE,MAAM,sBAAsB,CAAC;AAE5G,0DAA0D;AAC1D,eAAO,MAAM,2BAA2B,iBAAiB,CAAC;AAE1D,8EAA8E;AAC9E,eAAO,MAAM,iCAAiC,kCAAkC,CAAC;AAEjF,yDAAyD;AACzD,MAAM,WAAW,+BAAgC,SAAQ,mBAAmB;IACxE,sGAAsG;IACtG,UAAU,CAAC,EAAE,MAAM,CAAC;IACpB,6EAA6E;IAC7E,OAAO,CAAC,EAAE,MAAM,CAAC;IACjB;;;;OAIG;IACH,eAAe,CAAC,EAAE,MAAM,CAAC;IACzB;;;;OAIG;IACH,iBAAiB,CAAC,EAAE,MAAM,CAAC;IAC3B,qDAAqD;IACrD,SAAS,CAAC,EAAE,MAAM,CAAC;CACtB;AAeD;;;;;;;;;;;;;;;;;;;;;;GAsBG;AACH,qBACa,qBAAsB,SAAQ,aAAa;IACpD,IAAW,GAAG,IAAI,MAAM,CAEvB;IAED,IAAW,mBAAmB,IAAI,eAAe,EAAE,CAElD;cAEe,WAAW,CAAC,MAAM,EAAE,kBAAkB,CAAC,+BAA+B,CAAC,GAAG,OAAO,CAAC,UAAU,EAAE,CAAC;IA0B/G,kEAAkE;IAClE,OAAO,CAAC,WAAW;IAcnB,iDAAiD;IACjD,OAAO,CAAC,SAAS;IAWjB,+EAA+E;YACjE,iBAAiB;IA6B/B,wDAAwD;IACxD,OAAO,CAAC,YAAY;IAKpB,sEAAsE;IACtE,OAAO,CAAC,WAAW;IAenB;;;OAGG;IACH,OAAO,CAAC,kBAAkB;IA2B1B,wEAAwE;IACxE,OAAO,CAAC,aAAa;IAerB,0FAA0F;YAC5E,QAAQ;CAWzB"}
@@ -0,0 +1,201 @@
1
+ /**
2
+ * @fileoverview LLM-driven semantic segmenter — finds topic boundaries in prose.
3
+ *
4
+ * @module @memberjunction/ai-segmentation
5
+ */
6
+ var __decorate = (this && this.__decorate) || function (decorators, target, key, desc) {
7
+ var c = arguments.length, r = c < 3 ? target : desc === null ? desc = Object.getOwnPropertyDescriptor(target, key) : desc, d;
8
+ if (typeof Reflect === "object" && typeof Reflect.decorate === "function") r = Reflect.decorate(decorators, target, key, desc);
9
+ else for (var i = decorators.length - 1; i >= 0; i--) if (d = decorators[i]) r = (c < 3 ? d(r) : c > 3 ? d(target, key, r) : d(target, key)) || r;
10
+ return c > 3 && r && Object.defineProperty(target, key, r), r;
11
+ };
12
+ import { LogError, LogStatus } from '@memberjunction/core';
13
+ import { RegisterClass } from '@memberjunction/global';
14
+ import { AIEngine } from '@memberjunction/aiengine';
15
+ import { AIPromptRunner } from '@memberjunction/ai-prompts';
16
+ import { AIPromptParams } from '@memberjunction/ai-core-plus';
17
+ import { BaseSegmenter } from './BaseSegmenter.js';
18
+ import { StructuralTextSegmenter } from './StructuralTextSegmenter.js';
19
+ /** Registration key for {@link SemanticTextSegmenter}. */
20
+ export const SEMANTIC_TEXT_SEGMENTER_KEY = 'SemanticText';
21
+ /** Default name of the `MJ: AI Prompts` record driving boundary detection. */
22
+ export const SEMANTIC_SEGMENTATION_PROMPT_NAME = 'Content Semantic Segmentation';
23
+ /**
24
+ * Segments prose by asking an LLM where the topics change.
25
+ *
26
+ * Structure-aware segmentation only works when the author left structure behind.
27
+ * Transcripts, scanned reports, and long-form articles frequently have none — the
28
+ * topic shifts, but no heading marks it. This segmenter finds those latent
29
+ * boundaries and names them, producing titled sections that behave like headings
30
+ * the document never had.
31
+ *
32
+ * ## Cost posture
33
+ *
34
+ * The LLM pass is the expensive part of ingestion, so this class is written to
35
+ * avoid it whenever it wouldn't pay off: short documents short-circuit to
36
+ * structural segmentation, blocks are truncated to a preview before being shown to
37
+ * the model, and any failure degrades to `StructuralText` rather than failing the
38
+ * ingestion run. The model is asked to classify *block indices*, never character
39
+ * offsets — models are unreliable at arithmetic over long strings, and a wrong
40
+ * offset would silently corrupt chunk provenance.
41
+ *
42
+ * Because it runs through `AIPromptRunner`, every pass is a tracked `MJ: AI Prompt
43
+ * Run` with full token and cost attribution, and the prompt itself is versioned
44
+ * metadata rather than a string literal in code.
45
+ */
46
+ let SemanticTextSegmenter = class SemanticTextSegmenter extends BaseSegmenter {
47
+ get Key() {
48
+ return SEMANTIC_TEXT_SEGMENTER_KEY;
49
+ }
50
+ get SupportedModalities() {
51
+ return ['text'];
52
+ }
53
+ async SegmentCore(params) {
54
+ const text = params.Text ?? '';
55
+ if (text.trim().length === 0) {
56
+ return [];
57
+ }
58
+ const minTokens = params.Options?.MinTokensForLLM ?? 750;
59
+ if (this.tokensOf({ Modality: 'text', Text: text }) < minTokens) {
60
+ return this.fallback(params);
61
+ }
62
+ const blocks = this.buildBlocks(text, params.Options);
63
+ if (blocks.length < 2) {
64
+ return this.fallback(params);
65
+ }
66
+ const boundaries = await this.proposeBoundaries(blocks, params);
67
+ if (!boundaries || boundaries.length === 0) {
68
+ return this.fallback(params);
69
+ }
70
+ return this.buildSegments(blocks, boundaries);
71
+ }
72
+ // ─────────────────────────────────────────────
73
+ // Block preparation
74
+ // ─────────────────────────────────────────────
75
+ /** Split the document into paragraph blocks with real offsets. */
76
+ buildBlocks(text, options) {
77
+ const blocks = [];
78
+ const regex = /\n\s*\n/g;
79
+ let cursor = 0;
80
+ let match = regex.exec(text);
81
+ while (match !== null) {
82
+ this.pushBlock(blocks, text, cursor, match.index);
83
+ cursor = match.index + match[0].length;
84
+ match = regex.exec(text);
85
+ }
86
+ this.pushBlock(blocks, text, cursor, text.length);
87
+ return blocks.slice(0, options?.MaxBlocks ?? 300);
88
+ }
89
+ /** Append a block when the slice has content. */
90
+ pushBlock(blocks, text, start, end) {
91
+ const body = text.slice(start, end).trim();
92
+ if (body.length > 0) {
93
+ blocks.push({ Index: blocks.length, Text: body, StartOffset: start, EndOffset: end });
94
+ }
95
+ }
96
+ // ─────────────────────────────────────────────
97
+ // LLM boundary detection
98
+ // ─────────────────────────────────────────────
99
+ /** Run the segmentation prompt and return validated boundary block indices. */
100
+ async proposeBoundaries(blocks, params) {
101
+ const promptName = params.Options?.PromptName ?? SEMANTIC_SEGMENTATION_PROMPT_NAME;
102
+ const prompt = AIEngine.Instance.Prompts.find((p) => p.Name === promptName && p.Status === 'Active');
103
+ if (!prompt) {
104
+ LogStatus(`[SemanticTextSegmenter] Prompt '${promptName}' not found — falling back to structural segmentation.`);
105
+ return null;
106
+ }
107
+ const promptParams = new AIPromptParams();
108
+ promptParams.prompt = prompt;
109
+ promptParams.contextUser = params.ContextUser;
110
+ promptParams.data = { document: this.renderBlocks(blocks, params.Options), blockCount: blocks.length };
111
+ promptParams.attemptJSONRepair = true;
112
+ promptParams.additionalParameters = { temperature: 0.0 };
113
+ if (params.Options?.ModelID) {
114
+ promptParams.override = { modelId: params.Options.ModelID };
115
+ }
116
+ const result = await new AIPromptRunner().ExecutePrompt(promptParams);
117
+ if (!result.success) {
118
+ LogError(`[SemanticTextSegmenter] Boundary prompt failed: ${result.errorMessage ?? 'unknown error'}`);
119
+ return null;
120
+ }
121
+ return this.sanitizeBoundaries(this.parseResult(result.result), blocks.length);
122
+ }
123
+ /** Render numbered, truncated blocks for the prompt. */
124
+ renderBlocks(blocks, options) {
125
+ const previewChars = options?.BlockPreviewChars ?? 240;
126
+ return blocks.map((b) => `[${b.Index}] ${b.Text.slice(0, previewChars)}`).join('\n\n');
127
+ }
128
+ /** Accept either a parsed object or a JSON string from the runner. */
129
+ parseResult(raw) {
130
+ if (!raw) {
131
+ return null;
132
+ }
133
+ if (typeof raw !== 'string') {
134
+ return raw;
135
+ }
136
+ try {
137
+ return JSON.parse(raw);
138
+ }
139
+ catch {
140
+ LogError(`[SemanticTextSegmenter] Could not parse boundary JSON: ${raw.substring(0, 200)}`);
141
+ return null;
142
+ }
143
+ }
144
+ /**
145
+ * Clamp, dedupe, and sort model output. A hallucinated or out-of-range block
146
+ * index must never become a chunk offset, so anything unusable is dropped here.
147
+ */
148
+ sanitizeBoundaries(response, blockCount) {
149
+ const raw = response?.boundaries ?? [];
150
+ const seen = new Set();
151
+ const cleaned = [];
152
+ for (const entry of raw) {
153
+ const index = Number(entry?.startBlock);
154
+ if (!Number.isInteger(index) || index < 0 || index >= blockCount || seen.has(index)) {
155
+ continue;
156
+ }
157
+ seen.add(index);
158
+ cleaned.push({ startBlock: index, title: entry.title?.trim() || undefined });
159
+ }
160
+ cleaned.sort((a, b) => a.startBlock - b.startBlock);
161
+ if (cleaned.length > 0 && cleaned[0].startBlock !== 0) {
162
+ cleaned.unshift({ startBlock: 0 });
163
+ }
164
+ return cleaned;
165
+ }
166
+ // ─────────────────────────────────────────────
167
+ // Segment assembly
168
+ // ─────────────────────────────────────────────
169
+ /** Join blocks between consecutive boundaries into one segment each. */
170
+ buildSegments(blocks, boundaries) {
171
+ return boundaries.map((boundary, i) => {
172
+ const endBlock = i + 1 < boundaries.length ? boundaries[i + 1].startBlock : blocks.length;
173
+ const slice = blocks.slice(boundary.startBlock, endBlock);
174
+ const body = slice.map((b) => b.Text).join('\n\n');
175
+ return {
176
+ Modality: 'text',
177
+ Title: boundary.title,
178
+ Text: boundary.title ? `${boundary.title}\n${body}` : body,
179
+ StartOffset: slice[0]?.StartOffset,
180
+ EndOffset: slice[slice.length - 1]?.EndOffset,
181
+ };
182
+ });
183
+ }
184
+ /** Degrade to structural segmentation — never fail an ingestion run over segmentation. */
185
+ async fallback(params) {
186
+ const structural = new StructuralTextSegmenter();
187
+ const result = await structural.Segment({ Text: params.Text, MimeType: params.MimeType, Options: params.Options });
188
+ return result.Segments.map((s) => ({
189
+ Modality: s.Modality,
190
+ Text: s.Text,
191
+ Title: s.Title,
192
+ StartOffset: s.StartOffset,
193
+ EndOffset: s.EndOffset,
194
+ }));
195
+ }
196
+ };
197
+ SemanticTextSegmenter = __decorate([
198
+ RegisterClass(BaseSegmenter, SEMANTIC_TEXT_SEGMENTER_KEY)
199
+ ], SemanticTextSegmenter);
200
+ export { SemanticTextSegmenter };
201
+ //# sourceMappingURL=SemanticTextSegmenter.js.map