smart-slice 0.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- smart_slice/__init__.py +440 -0
- smart_slice/__main__.py +188 -0
- smart_slice/_accel.py +65 -0
- smart_slice/_config.py +60 -0
- smart_slice/_i18n.py +37 -0
- smart_slice/_logging.py +22 -0
- smart_slice/_markdown.py +33 -0
- smart_slice/_optional.py +114 -0
- smart_slice/_speedup_py.py +64 -0
- smart_slice/_uuid.py +40 -0
- smart_slice/_validation.py +174 -0
- smart_slice/chunker.py +884 -0
- smart_slice/chunking/__init__.py +29 -0
- smart_slice/chunking/_base.py +14 -0
- smart_slice/chunking/mark.py +39 -0
- smart_slice/chunking/overlap.py +64 -0
- smart_slice/exceptions.py +60 -0
- smart_slice/files.py +130 -0
- smart_slice/handlers/__init__.py +205 -0
- smart_slice/handlers/_utils.py +55 -0
- smart_slice/handlers/_xlsx_images.py +163 -0
- smart_slice/handlers/archive.py +152 -0
- smart_slice/handlers/base.py +22 -0
- smart_slice/handlers/csv_handler.py +136 -0
- smart_slice/handlers/doc.py +289 -0
- smart_slice/handlers/eml.py +100 -0
- smart_slice/handlers/epub.py +138 -0
- smart_slice/handlers/fb2.py +108 -0
- smart_slice/handlers/html.py +113 -0
- smart_slice/handlers/image.py +90 -0
- smart_slice/handlers/image_text_extract.py +125 -0
- smart_slice/handlers/ipynb.py +126 -0
- smart_slice/handlers/mbox.py +137 -0
- smart_slice/handlers/mhtml.py +96 -0
- smart_slice/handlers/mobi.py +93 -0
- smart_slice/handlers/msg.py +88 -0
- smart_slice/handlers/odf.py +239 -0
- smart_slice/handlers/pdf.py +634 -0
- smart_slice/handlers/ppt.py +134 -0
- smart_slice/handlers/pptx.py +160 -0
- smart_slice/handlers/raster_image.py +52 -0
- smart_slice/handlers/rtf.py +71 -0
- smart_slice/handlers/subtitle.py +103 -0
- smart_slice/handlers/svg.py +100 -0
- smart_slice/handlers/text.py +119 -0
- smart_slice/handlers/vcalendar.py +221 -0
- smart_slice/handlers/wps.py +54 -0
- smart_slice/handlers/xls.py +159 -0
- smart_slice/handlers/xlsx.py +262 -0
- smart_slice/handlers/xmind.py +170 -0
- smart_slice/handlers/zip_handler.py +306 -0
- smart_slice/options.py +225 -0
- smart_slice/patterns.py +69 -0
- smart_slice/qa/__init__.py +56 -0
- smart_slice/qa/_base.py +50 -0
- smart_slice/qa/_table_base.py +21 -0
- smart_slice/qa/csv_qa.py +64 -0
- smart_slice/qa/csv_table.py +78 -0
- smart_slice/qa/md_qa.py +146 -0
- smart_slice/qa/xls_qa.py +67 -0
- smart_slice/qa/xls_table.py +102 -0
- smart_slice/qa/xlsx_qa.py +78 -0
- smart_slice/qa/xlsx_table.py +129 -0
- smart_slice/qa/zip_qa.py +157 -0
- smart_slice/service.py +486 -0
- smart_slice/types.py +55 -0
- smart_slice-0.3.0.dist-info/METADATA +1193 -0
- smart_slice-0.3.0.dist-info/RECORD +71 -0
- smart_slice-0.3.0.dist-info/WHEEL +4 -0
- smart_slice-0.3.0.dist-info/entry_points.txt +2 -0
- smart_slice-0.3.0.dist-info/licenses/LICENSE +674 -0
smart_slice/__init__.py
ADDED
|
@@ -0,0 +1,440 @@
|
|
|
1
|
+
# coding=utf-8
|
|
2
|
+
"""smart-slice - fidelity-first document slicing for RAG pipelines.
|
|
3
|
+
|
|
4
|
+
Turn a document into retrieval-ready paragraphs: extract the text of ~30 file
|
|
5
|
+
formats, cut it along its own structure (heading tree, then blank lines, then a
|
|
6
|
+
length budget), and keep every original character reachable.
|
|
7
|
+
|
|
8
|
+
Quick start::
|
|
9
|
+
|
|
10
|
+
from smart_slice import slice_bytes, slice_text
|
|
11
|
+
|
|
12
|
+
paragraphs = slice_text("# Chapter\\n\\nbody text", limit=1000)
|
|
13
|
+
# [{'title': 'Chapter', 'content': 'body text'}]
|
|
14
|
+
|
|
15
|
+
paragraphs = slice_bytes(open("report.pdf", "rb").read(), "report.pdf", limit=1000)
|
|
16
|
+
|
|
17
|
+
Design notes
|
|
18
|
+
------------
|
|
19
|
+
*Fidelity before cleverness.* No cleaning step deletes source text: heading
|
|
20
|
+
markers are stripped only at line starts and only outside code fences, table
|
|
21
|
+
headers are *appended* to continuation chunks rather than rewritten into them,
|
|
22
|
+
oversized rows are re-split instead of truncated.
|
|
23
|
+
|
|
24
|
+
*Structure before length.* ``limit`` is a budget, not a grid: the slicer walks
|
|
25
|
+
the heading tree first, then falls back to sentence boundaries, then to a hard
|
|
26
|
+
character cut.
|
|
27
|
+
|
|
28
|
+
*No framework.* Pure library: no Django, no ORM, no network, no logging
|
|
29
|
+
configuration. Errors are :class:`~smart_slice.exceptions.SliceError` with an
|
|
30
|
+
HTTP-style ``code``; images extracted from documents are handed to a callback
|
|
31
|
+
instead of being persisted.
|
|
32
|
+
"""
|
|
33
|
+
import os
|
|
34
|
+
from typing import Any, Callable, Dict, Iterable, List, Optional, Sequence, Union
|
|
35
|
+
|
|
36
|
+
from ._validation import ParserLimits
|
|
37
|
+
from .exceptions import (
|
|
38
|
+
ParseError,
|
|
39
|
+
ResourceLimitError,
|
|
40
|
+
SliceError,
|
|
41
|
+
UnsupportedFormatError,
|
|
42
|
+
)
|
|
43
|
+
from .handlers import (
|
|
44
|
+
HANDLER_EXTENSIONS,
|
|
45
|
+
SPLIT_HANDLERS,
|
|
46
|
+
BaseSplitHandle,
|
|
47
|
+
FileBufferHandle,
|
|
48
|
+
missing_dependencies,
|
|
49
|
+
)
|
|
50
|
+
from .options import (
|
|
51
|
+
DEFAULT_LIMIT,
|
|
52
|
+
DEFAULT_OVERLAP,
|
|
53
|
+
ChunkingOptions,
|
|
54
|
+
current_options,
|
|
55
|
+
resolve_options,
|
|
56
|
+
use_options,
|
|
57
|
+
)
|
|
58
|
+
|
|
59
|
+
#: Default embedding-chunk window (characters) for :func:`chunk_paragraphs`.
|
|
60
|
+
DEFAULT_CHUNK_SIZE = 256
|
|
61
|
+
from .patterns import (
|
|
62
|
+
BLANK_LINE,
|
|
63
|
+
DEFAULT_PATTERNS,
|
|
64
|
+
LITERAL_PATTERNS,
|
|
65
|
+
MARKDOWN_HEADINGS,
|
|
66
|
+
patterns_for,
|
|
67
|
+
)
|
|
68
|
+
from .service import (
|
|
69
|
+
BytesSplitFile,
|
|
70
|
+
LazyImageTextExtractor,
|
|
71
|
+
build_image_text_extractor,
|
|
72
|
+
normalize_split_rows,
|
|
73
|
+
replace_image_file_ids,
|
|
74
|
+
split_document,
|
|
75
|
+
)
|
|
76
|
+
from .types import ImageAsset, Paragraph, SplitResult
|
|
77
|
+
|
|
78
|
+
__version__ = "0.3.0"
|
|
79
|
+
|
|
80
|
+
__all__ = [
|
|
81
|
+
"__version__",
|
|
82
|
+
# primary API
|
|
83
|
+
"slice_bytes",
|
|
84
|
+
"slice_text",
|
|
85
|
+
"slice_path",
|
|
86
|
+
"split_document",
|
|
87
|
+
"chunk",
|
|
88
|
+
"chunk_paragraphs",
|
|
89
|
+
# extraction only (no slicing)
|
|
90
|
+
"extract_text",
|
|
91
|
+
# text-level building blocks
|
|
92
|
+
"SplitModel",
|
|
93
|
+
"smart_split_paragraph",
|
|
94
|
+
"filter_special_char",
|
|
95
|
+
"MarkChunkHandle",
|
|
96
|
+
# configuration / introspection
|
|
97
|
+
"ChunkingOptions",
|
|
98
|
+
"DEFAULT_LIMIT",
|
|
99
|
+
"DEFAULT_OVERLAP",
|
|
100
|
+
"DEFAULT_CHUNK_SIZE",
|
|
101
|
+
"resolve_options",
|
|
102
|
+
"current_options",
|
|
103
|
+
"use_options",
|
|
104
|
+
"ParserLimits",
|
|
105
|
+
"SPLIT_HANDLERS",
|
|
106
|
+
"HANDLER_EXTENSIONS",
|
|
107
|
+
"missing_dependencies",
|
|
108
|
+
"supported_extensions",
|
|
109
|
+
"detect_handler",
|
|
110
|
+
"patterns_for",
|
|
111
|
+
"MARKDOWN_HEADINGS",
|
|
112
|
+
"DEFAULT_PATTERNS",
|
|
113
|
+
"BLANK_LINE",
|
|
114
|
+
"LITERAL_PATTERNS",
|
|
115
|
+
# types
|
|
116
|
+
"ImageAsset",
|
|
117
|
+
"Paragraph",
|
|
118
|
+
"SplitResult",
|
|
119
|
+
"BaseSplitHandle",
|
|
120
|
+
"FileBufferHandle",
|
|
121
|
+
"BytesSplitFile",
|
|
122
|
+
"LazyImageTextExtractor",
|
|
123
|
+
"build_image_text_extractor",
|
|
124
|
+
"normalize_split_rows",
|
|
125
|
+
"replace_image_file_ids",
|
|
126
|
+
# errors
|
|
127
|
+
"SliceError",
|
|
128
|
+
"ParseError",
|
|
129
|
+
"UnsupportedFormatError",
|
|
130
|
+
"ResourceLimitError",
|
|
131
|
+
]
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
def slice_text(
|
|
136
|
+
text: str,
|
|
137
|
+
*,
|
|
138
|
+
limit: Optional[int] = None,
|
|
139
|
+
overlap: Optional[int] = None,
|
|
140
|
+
overlap_ratio: Optional[float] = None,
|
|
141
|
+
options: Optional[ChunkingOptions] = None,
|
|
142
|
+
patterns: Optional[Sequence[Any]] = None,
|
|
143
|
+
with_filter: bool = False,
|
|
144
|
+
name: str = "document.md",
|
|
145
|
+
) -> List[Paragraph]:
|
|
146
|
+
"""Slice an in-memory string into ``[{title, content}]`` paragraphs.
|
|
147
|
+
|
|
148
|
+
:param text: source text (markdown headings are recognised)
|
|
149
|
+
:param limit: maximum characters per paragraph; ``None`` = default (1000)
|
|
150
|
+
:param overlap: characters of context shared by consecutive paragraphs.
|
|
151
|
+
``None`` = default (0, no overlap)
|
|
152
|
+
:param overlap_ratio: alternative form of ``overlap`` as a fraction of ``limit``
|
|
153
|
+
(``0.15`` means 15%). Ignored when ``overlap`` is given.
|
|
154
|
+
:param options: a :class:`ChunkingOptions` to reuse across calls; keyword
|
|
155
|
+
arguments above override the matching field
|
|
156
|
+
:param patterns: heading/paragraph regexes; ``None`` = the markdown default
|
|
157
|
+
:param with_filter: strip line-initial heading markers and collapse blank runs
|
|
158
|
+
:param name: only used to pick the default pattern family
|
|
159
|
+
|
|
160
|
+
``with_filter`` defaults to False here: slicing your own string is usually a
|
|
161
|
+
deliberate act and the caller keeps the raw text. The file entry points keep
|
|
162
|
+
the platform default of True.
|
|
163
|
+
"""
|
|
164
|
+
opts = resolve_options(options, limit=limit, overlap=overlap, overlap_ratio=overlap_ratio)
|
|
165
|
+
with use_options(opts):
|
|
166
|
+
return split_document(
|
|
167
|
+
name,
|
|
168
|
+
text.encode("utf-8"),
|
|
169
|
+
limit=opts.limit,
|
|
170
|
+
pattern_list=list(patterns) if patterns is not None else None,
|
|
171
|
+
with_filter=with_filter,
|
|
172
|
+
normalize=True,
|
|
173
|
+
)
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
def slice_bytes(
|
|
177
|
+
content: bytes,
|
|
178
|
+
name: str,
|
|
179
|
+
*,
|
|
180
|
+
limit: Optional[int] = None,
|
|
181
|
+
overlap: Optional[int] = None,
|
|
182
|
+
overlap_ratio: Optional[float] = None,
|
|
183
|
+
options: Optional[ChunkingOptions] = None,
|
|
184
|
+
patterns: Optional[Sequence[Any]] = None,
|
|
185
|
+
with_filter: bool = True,
|
|
186
|
+
normalize: bool = True,
|
|
187
|
+
save_image: Optional[Callable[[List[ImageAsset]], Any]] = None,
|
|
188
|
+
image_text_extractor: Optional[Callable[[bytes, str], str]] = None,
|
|
189
|
+
fallback_title: Optional[str] = None,
|
|
190
|
+
progress_hook: Optional[Callable[[], None]] = None,
|
|
191
|
+
) -> Union[List[Paragraph], SplitResult]:
|
|
192
|
+
"""Slice a document from its raw bytes.
|
|
193
|
+
|
|
194
|
+
:param content: complete file bytes
|
|
195
|
+
:param name: file name - decides which handler runs
|
|
196
|
+
:param limit: maximum characters per paragraph; ``None`` = default (1000)
|
|
197
|
+
:param overlap: characters of context shared by consecutive paragraphs.
|
|
198
|
+
``None`` = default (0). A positive value trades the
|
|
199
|
+
"chunks tile the source exactly once" property for
|
|
200
|
+
cross-boundary retrieval recall.
|
|
201
|
+
:param overlap_ratio: alternative form of ``overlap`` as a fraction of ``limit``
|
|
202
|
+
:param options: a :class:`ChunkingOptions` to reuse; keyword arguments
|
|
203
|
+
above override the matching field
|
|
204
|
+
:param patterns: custom heading regexes; ``None`` = per-format default
|
|
205
|
+
:param with_filter: pass through to the slicer's cleaning stage
|
|
206
|
+
:param normalize: True -> flat ``[{title, content}]``;
|
|
207
|
+
False -> the handler's raw structure (nested groups for
|
|
208
|
+
multi-sheet workbooks and archives, which previews want)
|
|
209
|
+
:param save_image: callback receiving :class:`ImageAsset` objects extracted
|
|
210
|
+
from the document; may return ``{new_id: existing_id}``
|
|
211
|
+
to remap references after deduplication
|
|
212
|
+
:param image_text_extractor: ``(image_bytes, image_name) -> str`` OCR hook.
|
|
213
|
+
Not installed by default; see
|
|
214
|
+
:func:`build_image_text_extractor`.
|
|
215
|
+
:param fallback_title: title used for paragraphs that have none (only with
|
|
216
|
+
``normalize=True``)
|
|
217
|
+
:param progress_hook: zero-argument callback fired before each handler, each
|
|
218
|
+
archive member and each OCR call - use it as a heartbeat
|
|
219
|
+
while parsing large files
|
|
220
|
+
:raises SliceError: nothing supports the format, or parsing failed
|
|
221
|
+
|
|
222
|
+
This is the entry point every other layer in the package funnels through.
|
|
223
|
+
|
|
224
|
+
``overlap`` reaches the slicer through a contextvar (see
|
|
225
|
+
:mod:`smart_slice.options`), so the generated format handlers keep their
|
|
226
|
+
original signature while still honouring the caller's chunking configuration.
|
|
227
|
+
"""
|
|
228
|
+
opts = resolve_options(options, limit=limit, overlap=overlap, overlap_ratio=overlap_ratio)
|
|
229
|
+
with use_options(opts):
|
|
230
|
+
return split_document(
|
|
231
|
+
name,
|
|
232
|
+
content,
|
|
233
|
+
limit=opts.limit,
|
|
234
|
+
pattern_list=list(patterns) if patterns is not None else None,
|
|
235
|
+
with_filter=with_filter,
|
|
236
|
+
save_image=save_image,
|
|
237
|
+
image_text_extractor=image_text_extractor,
|
|
238
|
+
normalize=normalize,
|
|
239
|
+
fallback_title=fallback_title,
|
|
240
|
+
progress_hook=progress_hook,
|
|
241
|
+
)
|
|
242
|
+
|
|
243
|
+
|
|
244
|
+
def slice_path(
|
|
245
|
+
path: Union[str, "os.PathLike[str]"],
|
|
246
|
+
*,
|
|
247
|
+
limit: Optional[int] = None,
|
|
248
|
+
overlap: Optional[int] = None,
|
|
249
|
+
overlap_ratio: Optional[float] = None,
|
|
250
|
+
options: Optional[ChunkingOptions] = None,
|
|
251
|
+
name: Optional[str] = None,
|
|
252
|
+
**kwargs: Any,
|
|
253
|
+
) -> Union[List[Paragraph], SplitResult]:
|
|
254
|
+
"""Read a file from disk and slice it. See :func:`slice_bytes` for keywords.
|
|
255
|
+
|
|
256
|
+
:param name: override the file name used for handler dispatch (defaults to the
|
|
257
|
+
path's basename)
|
|
258
|
+
"""
|
|
259
|
+
resolved = os.fspath(path)
|
|
260
|
+
with open(resolved, "rb") as handle:
|
|
261
|
+
content = handle.read()
|
|
262
|
+
return slice_bytes(
|
|
263
|
+
content,
|
|
264
|
+
name or os.path.basename(resolved),
|
|
265
|
+
limit=limit,
|
|
266
|
+
overlap=overlap,
|
|
267
|
+
overlap_ratio=overlap_ratio,
|
|
268
|
+
options=options,
|
|
269
|
+
**kwargs,
|
|
270
|
+
)
|
|
271
|
+
|
|
272
|
+
|
|
273
|
+
def extract_text(content: bytes, name: str, *, save_image: Optional[Callable] = None) -> str:
|
|
274
|
+
"""Extraction only: return the document's text without slicing it.
|
|
275
|
+
|
|
276
|
+
Mirrors ``get_content`` on the handlers - useful for previews, for feeding an
|
|
277
|
+
LLM a whole (small) document, or for bundling archive members into one file.
|
|
278
|
+
"""
|
|
279
|
+
file = BytesSplitFile(name, content)
|
|
280
|
+
buf = FileBufferHandle()
|
|
281
|
+
buf.buffer = content
|
|
282
|
+
get_buffer = buf.get_buffer
|
|
283
|
+
for handler in SPLIT_HANDLERS:
|
|
284
|
+
if not handler.support(file, get_buffer):
|
|
285
|
+
continue
|
|
286
|
+
sink = save_image or (lambda images: None)
|
|
287
|
+
from .handlers.image import ImageSplitHandle
|
|
288
|
+
|
|
289
|
+
if isinstance(handler, ImageSplitHandle):
|
|
290
|
+
# extended signature: OCR extractor + buffer access are keyword args
|
|
291
|
+
return handler.get_content(file, sink, get_buffer=get_buffer)
|
|
292
|
+
return handler.get_content(file, sink)
|
|
293
|
+
extension = ("." + name.rsplit(".", 1)[1].lower()) if "." in name else ""
|
|
294
|
+
raise UnsupportedFormatError(f"Unsupported file format{f': {extension}' if extension else ''}")
|
|
295
|
+
|
|
296
|
+
|
|
297
|
+
def detect_handler(name: str, content: Optional[bytes] = None) -> Optional[str]:
|
|
298
|
+
"""Return the handler class name that would claim this input, or ``None``.
|
|
299
|
+
|
|
300
|
+
Introspection helper for CLIs, tests and "why did my file parse like that"
|
|
301
|
+
debugging. Passing ``content`` enables the content-sniffing handlers.
|
|
302
|
+
"""
|
|
303
|
+
file = BytesSplitFile(name, content or b"")
|
|
304
|
+
buf = FileBufferHandle()
|
|
305
|
+
if content is not None:
|
|
306
|
+
buf.buffer = content
|
|
307
|
+
get_buffer = buf.get_buffer
|
|
308
|
+
for handler in SPLIT_HANDLERS:
|
|
309
|
+
try:
|
|
310
|
+
if handler.support(file, get_buffer):
|
|
311
|
+
return type(handler).__name__
|
|
312
|
+
except Exception: # noqa: BLE001 - a failed sniff must not break detection
|
|
313
|
+
continue
|
|
314
|
+
return None
|
|
315
|
+
|
|
316
|
+
|
|
317
|
+
def supported_extensions() -> Dict[str, List[str]]:
|
|
318
|
+
"""Map handler name -> extensions it is documented for."""
|
|
319
|
+
return {name: list(exts) for name, exts in HANDLER_EXTENSIONS.items()}
|
|
320
|
+
|
|
321
|
+
|
|
322
|
+
def chunk_paragraphs(
|
|
323
|
+
paragraphs: Iterable[Paragraph],
|
|
324
|
+
*,
|
|
325
|
+
chunk_size: Optional[int] = None,
|
|
326
|
+
chunk_overlap: Optional[int] = None,
|
|
327
|
+
chunk_overlap_ratio: Optional[float] = None,
|
|
328
|
+
options: Optional[ChunkingOptions] = None,
|
|
329
|
+
carry_title: Optional[bool] = None,
|
|
330
|
+
handler: Optional[Any] = None,
|
|
331
|
+
) -> List[str]:
|
|
332
|
+
"""Chunk paragraphs down to an embedding window, optionally with overlap.
|
|
333
|
+
|
|
334
|
+
Slicing yields semantic paragraphs; embedding models still need fixed-size
|
|
335
|
+
input. The paragraph stays the retrieval unit, the chunk the vector unit.
|
|
336
|
+
|
|
337
|
+
Defaults mirror the historical behaviour (``chunk_size=256``, no overlap, no
|
|
338
|
+
title prefix); every one of them is overridable per call or through an
|
|
339
|
+
:class:`~smart_slice.options.ChunkingOptions`.
|
|
340
|
+
|
|
341
|
+
:param paragraphs: ``[{title, content}]`` from :func:`slice_bytes`
|
|
342
|
+
:param chunk_size: target chunk length in characters (default 256)
|
|
343
|
+
:param chunk_overlap: characters of context carried from the previous chunk
|
|
344
|
+
:param chunk_overlap_ratio: ``chunk_overlap`` as a fraction of ``chunk_size``
|
|
345
|
+
:param options: a :class:`ChunkingOptions` to reuse
|
|
346
|
+
:param carry_title: prefix **every** chunk with its paragraph's heading
|
|
347
|
+
chain (default False). Without this, chunking a long
|
|
348
|
+
paragraph leaves only its first chunk carrying section
|
|
349
|
+
context, which measurably hurts retrieval for the rest.
|
|
350
|
+
The prefix is additive: ``chunk_size`` still governs
|
|
351
|
+
the body length, so budget for the title in your
|
|
352
|
+
embedding window.
|
|
353
|
+
:param handler: any :class:`~smart_slice.chunking.IChunkHandle`
|
|
354
|
+
"""
|
|
355
|
+
from .chunking import MarkChunkHandle, OverlapChunkHandle
|
|
356
|
+
|
|
357
|
+
opts = resolve_options(
|
|
358
|
+
options,
|
|
359
|
+
limit=chunk_size if chunk_size is not None else DEFAULT_CHUNK_SIZE,
|
|
360
|
+
overlap=chunk_overlap,
|
|
361
|
+
overlap_ratio=chunk_overlap_ratio,
|
|
362
|
+
)
|
|
363
|
+
if carry_title is not None:
|
|
364
|
+
opts = opts.with_(carry_title=carry_title)
|
|
365
|
+
|
|
366
|
+
handle = handler or (MarkChunkHandle() if opts.effective_overlap == 0 else OverlapChunkHandle())
|
|
367
|
+
|
|
368
|
+
# Pair each paragraph's text with its heading chain so the chain can be
|
|
369
|
+
# re-applied to *every* chunk produced from that paragraph (see below).
|
|
370
|
+
pairs = []
|
|
371
|
+
for row in paragraphs:
|
|
372
|
+
if isinstance(row, dict):
|
|
373
|
+
content = row.get("content")
|
|
374
|
+
title = str(row.get("title") or "").strip()
|
|
375
|
+
else:
|
|
376
|
+
content, title = row, ""
|
|
377
|
+
if isinstance(content, str) and content.strip():
|
|
378
|
+
pairs.append((content, title))
|
|
379
|
+
|
|
380
|
+
# Chunk each paragraph on its own: a paragraph is the semantic unit, and
|
|
381
|
+
# blending two sections into one vector hurts retrieval more than a little
|
|
382
|
+
# lost context helps.
|
|
383
|
+
chunks: List[str] = []
|
|
384
|
+
for text, title in pairs:
|
|
385
|
+
if isinstance(handle, OverlapChunkHandle):
|
|
386
|
+
pieces = handle.handle([text], opts)
|
|
387
|
+
else:
|
|
388
|
+
pieces = handle.handle([text], opts.limit)
|
|
389
|
+
if opts.carry_title and title:
|
|
390
|
+
prefix = f"# {title}\n"
|
|
391
|
+
chunks.extend(prefix + piece for piece in pieces)
|
|
392
|
+
else:
|
|
393
|
+
chunks.extend(pieces)
|
|
394
|
+
return chunks
|
|
395
|
+
|
|
396
|
+
|
|
397
|
+
def chunk(
|
|
398
|
+
text: str,
|
|
399
|
+
*,
|
|
400
|
+
chunk_size: Optional[int] = None,
|
|
401
|
+
chunk_overlap: Optional[int] = None,
|
|
402
|
+
chunk_overlap_ratio: Optional[float] = None,
|
|
403
|
+
options: Optional[ChunkingOptions] = None,
|
|
404
|
+
handler: Optional[Any] = None,
|
|
405
|
+
) -> List[str]:
|
|
406
|
+
"""Chunk a single string (convenience wrapper over :func:`chunk_paragraphs`)."""
|
|
407
|
+
return chunk_paragraphs(
|
|
408
|
+
[{"title": "", "content": text}],
|
|
409
|
+
chunk_size=chunk_size,
|
|
410
|
+
chunk_overlap=chunk_overlap,
|
|
411
|
+
chunk_overlap_ratio=chunk_overlap_ratio,
|
|
412
|
+
options=options,
|
|
413
|
+
handler=handler,
|
|
414
|
+
)
|
|
415
|
+
|
|
416
|
+
|
|
417
|
+
# re-exported lazily so ``import smart_slice`` stays cheap (jieba / parsers are
|
|
418
|
+
# only imported when the caller actually reaches for them)
|
|
419
|
+
def __getattr__(name: str) -> Any:
|
|
420
|
+
if name == "SplitModel":
|
|
421
|
+
from .chunker import SplitModel
|
|
422
|
+
|
|
423
|
+
return SplitModel
|
|
424
|
+
if name == "smart_split_paragraph":
|
|
425
|
+
from .chunker import smart_split_paragraph
|
|
426
|
+
|
|
427
|
+
return smart_split_paragraph
|
|
428
|
+
if name == "filter_special_char":
|
|
429
|
+
from .chunker import filter_special_char
|
|
430
|
+
|
|
431
|
+
return filter_special_char
|
|
432
|
+
if name == "MarkChunkHandle":
|
|
433
|
+
from .chunking import MarkChunkHandle
|
|
434
|
+
|
|
435
|
+
return MarkChunkHandle
|
|
436
|
+
if name == "OverlapChunkHandle":
|
|
437
|
+
from .chunking import OverlapChunkHandle
|
|
438
|
+
|
|
439
|
+
return OverlapChunkHandle
|
|
440
|
+
raise AttributeError(f"module {__name__!r} has no attribute {name!r}")
|
smart_slice/__main__.py
ADDED
|
@@ -0,0 +1,188 @@
|
|
|
1
|
+
# coding=utf-8
|
|
2
|
+
"""Command line interface: ``python -m smart_slice`` / ``smart-slice``."""
|
|
3
|
+
import argparse
|
|
4
|
+
import json
|
|
5
|
+
import os
|
|
6
|
+
import sys
|
|
7
|
+
from typing import List, Optional, Sequence
|
|
8
|
+
|
|
9
|
+
from . import (
|
|
10
|
+
DEFAULT_LIMIT,
|
|
11
|
+
ChunkingOptions,
|
|
12
|
+
__version__,
|
|
13
|
+
detect_handler,
|
|
14
|
+
missing_dependencies,
|
|
15
|
+
slice_path,
|
|
16
|
+
supported_extensions,
|
|
17
|
+
)
|
|
18
|
+
from .exceptions import SliceError
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def _build_parser() -> argparse.ArgumentParser:
|
|
22
|
+
parser = argparse.ArgumentParser(
|
|
23
|
+
prog="smart-slice",
|
|
24
|
+
description="Fidelity-first document slicing for RAG pipelines.",
|
|
25
|
+
)
|
|
26
|
+
parser.add_argument("--version", action="version", version=f"smart-slice {__version__}")
|
|
27
|
+
sub = parser.add_subparsers(dest="command", required=True)
|
|
28
|
+
|
|
29
|
+
run = sub.add_parser("slice", help="slice a file into paragraphs")
|
|
30
|
+
run.add_argument("path", help="document to slice")
|
|
31
|
+
run.add_argument("--limit", type=int, default=DEFAULT_LIMIT, help=f"max chars per paragraph (default {DEFAULT_LIMIT})")
|
|
32
|
+
run.add_argument(
|
|
33
|
+
"--overlap", type=int, default=None,
|
|
34
|
+
help="characters of context shared by consecutive paragraphs (default 0 = none)",
|
|
35
|
+
)
|
|
36
|
+
run.add_argument(
|
|
37
|
+
"--overlap-ratio", type=float, default=None,
|
|
38
|
+
help="alternative to --overlap, as a fraction of --limit (e.g. 0.15 for 15%%)",
|
|
39
|
+
)
|
|
40
|
+
run.add_argument(
|
|
41
|
+
"--overlap-section-only", action="store_true",
|
|
42
|
+
help="carry context only between paragraphs in the same section",
|
|
43
|
+
)
|
|
44
|
+
run.add_argument("--no-filter", action="store_true", help="keep heading markers / blank runs verbatim")
|
|
45
|
+
run.add_argument(
|
|
46
|
+
"--format",
|
|
47
|
+
choices=("json", "jsonl", "text", "md"),
|
|
48
|
+
default="json",
|
|
49
|
+
help="output encoding (default json)",
|
|
50
|
+
)
|
|
51
|
+
run.add_argument("--output", "-o", help="write to a file instead of stdout")
|
|
52
|
+
run.add_argument("--title-prefix", action="store_true", help="prefix each paragraph with its title chain")
|
|
53
|
+
run.add_argument("--stats", action="store_true", help="print a paragraph/character summary to stderr")
|
|
54
|
+
|
|
55
|
+
chunk = sub.add_parser("chunk", help="slice, then chunk paragraphs for an embedding window")
|
|
56
|
+
chunk.add_argument("path", help="document to slice and chunk")
|
|
57
|
+
chunk.add_argument("--limit", type=int, default=DEFAULT_LIMIT, help="paragraph budget (default %(default)s)")
|
|
58
|
+
chunk.add_argument("--chunk-size", type=int, default=256, help="embedding chunk size (default %(default)s)")
|
|
59
|
+
chunk.add_argument("--chunk-overlap", type=int, default=None, help="chars shared by consecutive chunks")
|
|
60
|
+
chunk.add_argument("--carry-title", action="store_true", help="prefix each chunk with its heading chain")
|
|
61
|
+
chunk.add_argument("--output", "-o", help="write JSON lines to a file instead of stdout")
|
|
62
|
+
chunk.add_argument("--stats", action="store_true", help="print a chunk summary to stderr")
|
|
63
|
+
|
|
64
|
+
probe = sub.add_parser("detect", help="show which handler claims a file")
|
|
65
|
+
probe.add_argument("path", help="document to inspect")
|
|
66
|
+
|
|
67
|
+
info = sub.add_parser("formats", help="list supported extensions and missing optional deps")
|
|
68
|
+
info.add_argument("--json", action="store_true", help="machine readable output")
|
|
69
|
+
return parser
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def _emit(rows: List[dict], fmt: str, output: Optional[str], title_prefix: bool) -> None:
|
|
73
|
+
def encode(row: dict) -> str:
|
|
74
|
+
content = row.get("content", "")
|
|
75
|
+
if title_prefix and row.get("title"):
|
|
76
|
+
content = f"# {row['title']}\n\n{content}"
|
|
77
|
+
return content
|
|
78
|
+
|
|
79
|
+
if fmt == "json":
|
|
80
|
+
payload = json.dumps(rows, ensure_ascii=False, indent=2)
|
|
81
|
+
elif fmt == "jsonl":
|
|
82
|
+
payload = "\n".join(json.dumps(row, ensure_ascii=False) for row in rows)
|
|
83
|
+
elif fmt == "text":
|
|
84
|
+
payload = "\n\n".join(encode(row) for row in rows)
|
|
85
|
+
else: # md
|
|
86
|
+
payload = "\n\n".join(
|
|
87
|
+
(f"# {row.get('title')}\n\n" if row.get("title") else "") + row.get("content", "") for row in rows
|
|
88
|
+
)
|
|
89
|
+
|
|
90
|
+
if output:
|
|
91
|
+
directory = os.path.dirname(os.path.abspath(output))
|
|
92
|
+
os.makedirs(directory, exist_ok=True)
|
|
93
|
+
with open(output, "w", encoding="utf-8", newline="\n") as handle:
|
|
94
|
+
handle.write(payload)
|
|
95
|
+
else:
|
|
96
|
+
sys.stdout.write(payload + "\n")
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
def main(argv: Optional[Sequence[str]] = None) -> int:
|
|
100
|
+
args = _build_parser().parse_args(argv)
|
|
101
|
+
|
|
102
|
+
if args.command == "formats":
|
|
103
|
+
extensions = supported_extensions()
|
|
104
|
+
gaps = missing_dependencies()
|
|
105
|
+
if args.json:
|
|
106
|
+
sys.stdout.write(json.dumps({"handlers": extensions, "missing_optional": gaps}, ensure_ascii=False, indent=2) + "\n")
|
|
107
|
+
return 0
|
|
108
|
+
for handler, exts in extensions.items():
|
|
109
|
+
sys.stdout.write(f"{handler:<22} {', '.join(exts)}\n")
|
|
110
|
+
if gaps:
|
|
111
|
+
sys.stdout.write("\nOptional extras not installed:\n")
|
|
112
|
+
for gap in gaps:
|
|
113
|
+
sys.stdout.write(f" smart-slice[{gap['extra']}] -> {gap['requirement']}\n")
|
|
114
|
+
return 0
|
|
115
|
+
|
|
116
|
+
if args.command == "detect":
|
|
117
|
+
with open(args.path, "rb") as handle:
|
|
118
|
+
content = handle.read()
|
|
119
|
+
handler = detect_handler(os.path.basename(args.path), content)
|
|
120
|
+
sys.stdout.write((handler or "<no handler>") + "\n")
|
|
121
|
+
return 0 if handler else 2
|
|
122
|
+
|
|
123
|
+
if args.command == "chunk":
|
|
124
|
+
from . import chunk_paragraphs
|
|
125
|
+
|
|
126
|
+
try:
|
|
127
|
+
paragraphs = slice_path(args.path, limit=args.limit)
|
|
128
|
+
except SliceError as error:
|
|
129
|
+
sys.stderr.write(f"smart-slice: {error.message} (code {error.code})\n")
|
|
130
|
+
return 1
|
|
131
|
+
except OSError as error:
|
|
132
|
+
sys.stderr.write(f"smart-slice: cannot read {args.path}: {error}\n")
|
|
133
|
+
return 1
|
|
134
|
+
|
|
135
|
+
pieces = chunk_paragraphs(
|
|
136
|
+
paragraphs,
|
|
137
|
+
chunk_size=args.chunk_size,
|
|
138
|
+
chunk_overlap=args.chunk_overlap,
|
|
139
|
+
carry_title=args.carry_title or None,
|
|
140
|
+
)
|
|
141
|
+
payload = "\n".join(json.dumps({"content": piece}, ensure_ascii=False) for piece in pieces)
|
|
142
|
+
if args.output:
|
|
143
|
+
directory = os.path.dirname(os.path.abspath(args.output))
|
|
144
|
+
os.makedirs(directory, exist_ok=True)
|
|
145
|
+
with open(args.output, "w", encoding="utf-8", newline="\n") as handle:
|
|
146
|
+
handle.write(payload + ("\n" if pieces else ""))
|
|
147
|
+
else:
|
|
148
|
+
sys.stdout.write(payload + ("\n" if pieces else ""))
|
|
149
|
+
if args.stats:
|
|
150
|
+
characters = sum(len(piece) for piece in pieces)
|
|
151
|
+
sys.stderr.write(
|
|
152
|
+
f"smart-slice: {len(pieces)} chunks from {len(paragraphs)} paragraphs, "
|
|
153
|
+
f"{characters} characters, chunk_size={args.chunk_size}, "
|
|
154
|
+
f"chunk_overlap={args.chunk_overlap or 0}\n"
|
|
155
|
+
)
|
|
156
|
+
return 0
|
|
157
|
+
|
|
158
|
+
# slice
|
|
159
|
+
try:
|
|
160
|
+
opts = ChunkingOptions(
|
|
161
|
+
limit=args.limit,
|
|
162
|
+
overlap=args.overlap,
|
|
163
|
+
overlap_ratio=args.overlap_ratio,
|
|
164
|
+
overlap_within_section=args.overlap_section_only,
|
|
165
|
+
)
|
|
166
|
+
result = slice_path(args.path, limit=args.limit, options=opts,
|
|
167
|
+
with_filter=not args.no_filter)
|
|
168
|
+
except SliceError as error:
|
|
169
|
+
sys.stderr.write(f"smart-slice: {error.message} (code {error.code})\n")
|
|
170
|
+
return 1
|
|
171
|
+
except OSError as error:
|
|
172
|
+
sys.stderr.write(f"smart-slice: cannot read {args.path}: {error}\n")
|
|
173
|
+
return 1
|
|
174
|
+
|
|
175
|
+
rows = [row for row in result if isinstance(row, dict)]
|
|
176
|
+
_emit(rows, args.format, args.output, args.title_prefix)
|
|
177
|
+
if args.stats:
|
|
178
|
+
characters = sum(len(row.get("content", "")) for row in rows)
|
|
179
|
+
overlap = opts.effective_overlap
|
|
180
|
+
sys.stderr.write(
|
|
181
|
+
f"smart-slice: {len(rows)} paragraphs, {characters} characters, "
|
|
182
|
+
f"limit={args.limit}, overlap={overlap}\n"
|
|
183
|
+
)
|
|
184
|
+
return 0
|
|
185
|
+
|
|
186
|
+
|
|
187
|
+
if __name__ == "__main__":
|
|
188
|
+
raise SystemExit(main())
|
smart_slice/_accel.py
ADDED
|
@@ -0,0 +1,65 @@
|
|
|
1
|
+
# coding=utf-8
|
|
2
|
+
"""Accelerator resolution: prefer the C scan, fall back to pure Python.
|
|
3
|
+
|
|
4
|
+
The chunker's hot path is "find the heading lines of this block". The regular
|
|
5
|
+
expression path runs up to six whole-block scans per recursion level (one per
|
|
6
|
+
heading level, cascading until one matches). A single linear scan that reports
|
|
7
|
+
every candidate heading line with its hash count replaces all of them.
|
|
8
|
+
|
|
9
|
+
This module resolves which scan implementation to use:
|
|
10
|
+
|
|
11
|
+
1. ``smart_slice._speedup`` - the optional C extension (see csrc/_speedup.c);
|
|
12
|
+
2. ``smart_slice._speedup_py`` - an identical pure-Python scan, always present.
|
|
13
|
+
|
|
14
|
+
Both return ``[(line_start, line_end, hashes), ...]`` with the same acceptance
|
|
15
|
+
rule, so the chunker is agnostic. ``ACCELERATOR`` records which one is active for
|
|
16
|
+
diagnostics; it never affects results (the equivalence test asserts that).
|
|
17
|
+
"""
|
|
18
|
+
from typing import Callable, List, Optional, Tuple
|
|
19
|
+
|
|
20
|
+
from smart_slice._speedup_py import scan_heading_candidates as _py_scan
|
|
21
|
+
|
|
22
|
+
__all__ = [
|
|
23
|
+
"scan_heading_candidates",
|
|
24
|
+
"ACCELERATOR",
|
|
25
|
+
"CANONICAL_HEADING_LEVELS",
|
|
26
|
+
"heading_level_of",
|
|
27
|
+
]
|
|
28
|
+
|
|
29
|
+
try: # optional C extension
|
|
30
|
+
from smart_slice._speedup import scan_heading_candidates as _c_scan # type: ignore
|
|
31
|
+
|
|
32
|
+
scan_heading_candidates: Callable[[str], List[Tuple[int, int, int]]] = _c_scan
|
|
33
|
+
ACCELERATOR = "c"
|
|
34
|
+
except Exception: # noqa: BLE001 - any import/build problem falls back cleanly
|
|
35
|
+
scan_heading_candidates = _py_scan
|
|
36
|
+
ACCELERATOR = "python"
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
# The canonical markdown heading patterns, by exact source string, mapped to their
|
|
40
|
+
# 1-based level. A pattern is eligible for the scan fast path only if its
|
|
41
|
+
# `.pattern` matches one of these verbatim - any custom or edited pattern takes the
|
|
42
|
+
# regular-expression path, so the optimisation can never change custom behaviour.
|
|
43
|
+
_CANONICAL_PATTERNS = (
|
|
44
|
+
r'(?<=^)# (?!-\*- coding:).*|(?<=\n)# (?!-\*- coding:).*',
|
|
45
|
+
r'(?<=\n)(?<!#)## (?!#).*|(?<=^)(?<!#)## (?!#).*',
|
|
46
|
+
r"(?<=\n)(?<!#)### (?!#).*|(?<=^)(?<!#)### (?!#).*",
|
|
47
|
+
r"(?<=\n)(?<!#)#### (?!#).*|(?<=^)(?<!#)#### (?!#).*",
|
|
48
|
+
r"(?<=\n)(?<!#)##### (?!#).*|(?<=^)(?<!#)##### (?!#).*",
|
|
49
|
+
r"(?<=\n)(?<!#)###### (?!#).*|(?<=^)(?<!#)###### (?!#).*",
|
|
50
|
+
)
|
|
51
|
+
|
|
52
|
+
CANONICAL_HEADING_LEVELS = {pattern: level + 1 for level, pattern in enumerate(_CANONICAL_PATTERNS)}
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def heading_level_of(pattern) -> Optional[int]:
|
|
56
|
+
"""Return the heading level (1-6) for a canonical heading pattern, else None.
|
|
57
|
+
|
|
58
|
+
``None`` means "not a canonical markdown heading" - the caller must use the
|
|
59
|
+
regular-expression path for that pattern (custom schemes, the blank-line rule,
|
|
60
|
+
etc.). Accepts either a compiled pattern or a raw string.
|
|
61
|
+
"""
|
|
62
|
+
source = getattr(pattern, "pattern", pattern)
|
|
63
|
+
if not isinstance(source, str):
|
|
64
|
+
return None
|
|
65
|
+
return CANONICAL_HEADING_LEVELS.get(source)
|