smart-slice 0.3.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (71) hide show
  1. smart_slice/__init__.py +440 -0
  2. smart_slice/__main__.py +188 -0
  3. smart_slice/_accel.py +65 -0
  4. smart_slice/_config.py +60 -0
  5. smart_slice/_i18n.py +37 -0
  6. smart_slice/_logging.py +22 -0
  7. smart_slice/_markdown.py +33 -0
  8. smart_slice/_optional.py +114 -0
  9. smart_slice/_speedup_py.py +64 -0
  10. smart_slice/_uuid.py +40 -0
  11. smart_slice/_validation.py +174 -0
  12. smart_slice/chunker.py +884 -0
  13. smart_slice/chunking/__init__.py +29 -0
  14. smart_slice/chunking/_base.py +14 -0
  15. smart_slice/chunking/mark.py +39 -0
  16. smart_slice/chunking/overlap.py +64 -0
  17. smart_slice/exceptions.py +60 -0
  18. smart_slice/files.py +130 -0
  19. smart_slice/handlers/__init__.py +205 -0
  20. smart_slice/handlers/_utils.py +55 -0
  21. smart_slice/handlers/_xlsx_images.py +163 -0
  22. smart_slice/handlers/archive.py +152 -0
  23. smart_slice/handlers/base.py +22 -0
  24. smart_slice/handlers/csv_handler.py +136 -0
  25. smart_slice/handlers/doc.py +289 -0
  26. smart_slice/handlers/eml.py +100 -0
  27. smart_slice/handlers/epub.py +138 -0
  28. smart_slice/handlers/fb2.py +108 -0
  29. smart_slice/handlers/html.py +113 -0
  30. smart_slice/handlers/image.py +90 -0
  31. smart_slice/handlers/image_text_extract.py +125 -0
  32. smart_slice/handlers/ipynb.py +126 -0
  33. smart_slice/handlers/mbox.py +137 -0
  34. smart_slice/handlers/mhtml.py +96 -0
  35. smart_slice/handlers/mobi.py +93 -0
  36. smart_slice/handlers/msg.py +88 -0
  37. smart_slice/handlers/odf.py +239 -0
  38. smart_slice/handlers/pdf.py +634 -0
  39. smart_slice/handlers/ppt.py +134 -0
  40. smart_slice/handlers/pptx.py +160 -0
  41. smart_slice/handlers/raster_image.py +52 -0
  42. smart_slice/handlers/rtf.py +71 -0
  43. smart_slice/handlers/subtitle.py +103 -0
  44. smart_slice/handlers/svg.py +100 -0
  45. smart_slice/handlers/text.py +119 -0
  46. smart_slice/handlers/vcalendar.py +221 -0
  47. smart_slice/handlers/wps.py +54 -0
  48. smart_slice/handlers/xls.py +159 -0
  49. smart_slice/handlers/xlsx.py +262 -0
  50. smart_slice/handlers/xmind.py +170 -0
  51. smart_slice/handlers/zip_handler.py +306 -0
  52. smart_slice/options.py +225 -0
  53. smart_slice/patterns.py +69 -0
  54. smart_slice/qa/__init__.py +56 -0
  55. smart_slice/qa/_base.py +50 -0
  56. smart_slice/qa/_table_base.py +21 -0
  57. smart_slice/qa/csv_qa.py +64 -0
  58. smart_slice/qa/csv_table.py +78 -0
  59. smart_slice/qa/md_qa.py +146 -0
  60. smart_slice/qa/xls_qa.py +67 -0
  61. smart_slice/qa/xls_table.py +102 -0
  62. smart_slice/qa/xlsx_qa.py +78 -0
  63. smart_slice/qa/xlsx_table.py +129 -0
  64. smart_slice/qa/zip_qa.py +157 -0
  65. smart_slice/service.py +486 -0
  66. smart_slice/types.py +55 -0
  67. smart_slice-0.3.0.dist-info/METADATA +1193 -0
  68. smart_slice-0.3.0.dist-info/RECORD +71 -0
  69. smart_slice-0.3.0.dist-info/WHEEL +4 -0
  70. smart_slice-0.3.0.dist-info/entry_points.txt +2 -0
  71. smart_slice-0.3.0.dist-info/licenses/LICENSE +674 -0
@@ -0,0 +1,440 @@
1
+ # coding=utf-8
2
+ """smart-slice - fidelity-first document slicing for RAG pipelines.
3
+
4
+ Turn a document into retrieval-ready paragraphs: extract the text of ~30 file
5
+ formats, cut it along its own structure (heading tree, then blank lines, then a
6
+ length budget), and keep every original character reachable.
7
+
8
+ Quick start::
9
+
10
+ from smart_slice import slice_bytes, slice_text
11
+
12
+ paragraphs = slice_text("# Chapter\\n\\nbody text", limit=1000)
13
+ # [{'title': 'Chapter', 'content': 'body text'}]
14
+
15
+ paragraphs = slice_bytes(open("report.pdf", "rb").read(), "report.pdf", limit=1000)
16
+
17
+ Design notes
18
+ ------------
19
+ *Fidelity before cleverness.* No cleaning step deletes source text: heading
20
+ markers are stripped only at line starts and only outside code fences, table
21
+ headers are *appended* to continuation chunks rather than rewritten into them,
22
+ oversized rows are re-split instead of truncated.
23
+
24
+ *Structure before length.* ``limit`` is a budget, not a grid: the slicer walks
25
+ the heading tree first, then falls back to sentence boundaries, then to a hard
26
+ character cut.
27
+
28
+ *No framework.* Pure library: no Django, no ORM, no network, no logging
29
+ configuration. Errors are :class:`~smart_slice.exceptions.SliceError` with an
30
+ HTTP-style ``code``; images extracted from documents are handed to a callback
31
+ instead of being persisted.
32
+ """
33
+ import os
34
+ from typing import Any, Callable, Dict, Iterable, List, Optional, Sequence, Union
35
+
36
+ from ._validation import ParserLimits
37
+ from .exceptions import (
38
+ ParseError,
39
+ ResourceLimitError,
40
+ SliceError,
41
+ UnsupportedFormatError,
42
+ )
43
+ from .handlers import (
44
+ HANDLER_EXTENSIONS,
45
+ SPLIT_HANDLERS,
46
+ BaseSplitHandle,
47
+ FileBufferHandle,
48
+ missing_dependencies,
49
+ )
50
+ from .options import (
51
+ DEFAULT_LIMIT,
52
+ DEFAULT_OVERLAP,
53
+ ChunkingOptions,
54
+ current_options,
55
+ resolve_options,
56
+ use_options,
57
+ )
58
+
59
+ #: Default embedding-chunk window (characters) for :func:`chunk_paragraphs`.
60
+ DEFAULT_CHUNK_SIZE = 256
61
+ from .patterns import (
62
+ BLANK_LINE,
63
+ DEFAULT_PATTERNS,
64
+ LITERAL_PATTERNS,
65
+ MARKDOWN_HEADINGS,
66
+ patterns_for,
67
+ )
68
+ from .service import (
69
+ BytesSplitFile,
70
+ LazyImageTextExtractor,
71
+ build_image_text_extractor,
72
+ normalize_split_rows,
73
+ replace_image_file_ids,
74
+ split_document,
75
+ )
76
+ from .types import ImageAsset, Paragraph, SplitResult
77
+
78
+ __version__ = "0.3.0"
79
+
80
+ __all__ = [
81
+ "__version__",
82
+ # primary API
83
+ "slice_bytes",
84
+ "slice_text",
85
+ "slice_path",
86
+ "split_document",
87
+ "chunk",
88
+ "chunk_paragraphs",
89
+ # extraction only (no slicing)
90
+ "extract_text",
91
+ # text-level building blocks
92
+ "SplitModel",
93
+ "smart_split_paragraph",
94
+ "filter_special_char",
95
+ "MarkChunkHandle",
96
+ # configuration / introspection
97
+ "ChunkingOptions",
98
+ "DEFAULT_LIMIT",
99
+ "DEFAULT_OVERLAP",
100
+ "DEFAULT_CHUNK_SIZE",
101
+ "resolve_options",
102
+ "current_options",
103
+ "use_options",
104
+ "ParserLimits",
105
+ "SPLIT_HANDLERS",
106
+ "HANDLER_EXTENSIONS",
107
+ "missing_dependencies",
108
+ "supported_extensions",
109
+ "detect_handler",
110
+ "patterns_for",
111
+ "MARKDOWN_HEADINGS",
112
+ "DEFAULT_PATTERNS",
113
+ "BLANK_LINE",
114
+ "LITERAL_PATTERNS",
115
+ # types
116
+ "ImageAsset",
117
+ "Paragraph",
118
+ "SplitResult",
119
+ "BaseSplitHandle",
120
+ "FileBufferHandle",
121
+ "BytesSplitFile",
122
+ "LazyImageTextExtractor",
123
+ "build_image_text_extractor",
124
+ "normalize_split_rows",
125
+ "replace_image_file_ids",
126
+ # errors
127
+ "SliceError",
128
+ "ParseError",
129
+ "UnsupportedFormatError",
130
+ "ResourceLimitError",
131
+ ]
132
+
133
+
134
+
135
+ def slice_text(
136
+ text: str,
137
+ *,
138
+ limit: Optional[int] = None,
139
+ overlap: Optional[int] = None,
140
+ overlap_ratio: Optional[float] = None,
141
+ options: Optional[ChunkingOptions] = None,
142
+ patterns: Optional[Sequence[Any]] = None,
143
+ with_filter: bool = False,
144
+ name: str = "document.md",
145
+ ) -> List[Paragraph]:
146
+ """Slice an in-memory string into ``[{title, content}]`` paragraphs.
147
+
148
+ :param text: source text (markdown headings are recognised)
149
+ :param limit: maximum characters per paragraph; ``None`` = default (1000)
150
+ :param overlap: characters of context shared by consecutive paragraphs.
151
+ ``None`` = default (0, no overlap)
152
+ :param overlap_ratio: alternative form of ``overlap`` as a fraction of ``limit``
153
+ (``0.15`` means 15%). Ignored when ``overlap`` is given.
154
+ :param options: a :class:`ChunkingOptions` to reuse across calls; keyword
155
+ arguments above override the matching field
156
+ :param patterns: heading/paragraph regexes; ``None`` = the markdown default
157
+ :param with_filter: strip line-initial heading markers and collapse blank runs
158
+ :param name: only used to pick the default pattern family
159
+
160
+ ``with_filter`` defaults to False here: slicing your own string is usually a
161
+ deliberate act and the caller keeps the raw text. The file entry points keep
162
+ the platform default of True.
163
+ """
164
+ opts = resolve_options(options, limit=limit, overlap=overlap, overlap_ratio=overlap_ratio)
165
+ with use_options(opts):
166
+ return split_document(
167
+ name,
168
+ text.encode("utf-8"),
169
+ limit=opts.limit,
170
+ pattern_list=list(patterns) if patterns is not None else None,
171
+ with_filter=with_filter,
172
+ normalize=True,
173
+ )
174
+
175
+
176
+ def slice_bytes(
177
+ content: bytes,
178
+ name: str,
179
+ *,
180
+ limit: Optional[int] = None,
181
+ overlap: Optional[int] = None,
182
+ overlap_ratio: Optional[float] = None,
183
+ options: Optional[ChunkingOptions] = None,
184
+ patterns: Optional[Sequence[Any]] = None,
185
+ with_filter: bool = True,
186
+ normalize: bool = True,
187
+ save_image: Optional[Callable[[List[ImageAsset]], Any]] = None,
188
+ image_text_extractor: Optional[Callable[[bytes, str], str]] = None,
189
+ fallback_title: Optional[str] = None,
190
+ progress_hook: Optional[Callable[[], None]] = None,
191
+ ) -> Union[List[Paragraph], SplitResult]:
192
+ """Slice a document from its raw bytes.
193
+
194
+ :param content: complete file bytes
195
+ :param name: file name - decides which handler runs
196
+ :param limit: maximum characters per paragraph; ``None`` = default (1000)
197
+ :param overlap: characters of context shared by consecutive paragraphs.
198
+ ``None`` = default (0). A positive value trades the
199
+ "chunks tile the source exactly once" property for
200
+ cross-boundary retrieval recall.
201
+ :param overlap_ratio: alternative form of ``overlap`` as a fraction of ``limit``
202
+ :param options: a :class:`ChunkingOptions` to reuse; keyword arguments
203
+ above override the matching field
204
+ :param patterns: custom heading regexes; ``None`` = per-format default
205
+ :param with_filter: pass through to the slicer's cleaning stage
206
+ :param normalize: True -> flat ``[{title, content}]``;
207
+ False -> the handler's raw structure (nested groups for
208
+ multi-sheet workbooks and archives, which previews want)
209
+ :param save_image: callback receiving :class:`ImageAsset` objects extracted
210
+ from the document; may return ``{new_id: existing_id}``
211
+ to remap references after deduplication
212
+ :param image_text_extractor: ``(image_bytes, image_name) -> str`` OCR hook.
213
+ Not installed by default; see
214
+ :func:`build_image_text_extractor`.
215
+ :param fallback_title: title used for paragraphs that have none (only with
216
+ ``normalize=True``)
217
+ :param progress_hook: zero-argument callback fired before each handler, each
218
+ archive member and each OCR call - use it as a heartbeat
219
+ while parsing large files
220
+ :raises SliceError: nothing supports the format, or parsing failed
221
+
222
+ This is the entry point every other layer in the package funnels through.
223
+
224
+ ``overlap`` reaches the slicer through a contextvar (see
225
+ :mod:`smart_slice.options`), so the generated format handlers keep their
226
+ original signature while still honouring the caller's chunking configuration.
227
+ """
228
+ opts = resolve_options(options, limit=limit, overlap=overlap, overlap_ratio=overlap_ratio)
229
+ with use_options(opts):
230
+ return split_document(
231
+ name,
232
+ content,
233
+ limit=opts.limit,
234
+ pattern_list=list(patterns) if patterns is not None else None,
235
+ with_filter=with_filter,
236
+ save_image=save_image,
237
+ image_text_extractor=image_text_extractor,
238
+ normalize=normalize,
239
+ fallback_title=fallback_title,
240
+ progress_hook=progress_hook,
241
+ )
242
+
243
+
244
+ def slice_path(
245
+ path: Union[str, "os.PathLike[str]"],
246
+ *,
247
+ limit: Optional[int] = None,
248
+ overlap: Optional[int] = None,
249
+ overlap_ratio: Optional[float] = None,
250
+ options: Optional[ChunkingOptions] = None,
251
+ name: Optional[str] = None,
252
+ **kwargs: Any,
253
+ ) -> Union[List[Paragraph], SplitResult]:
254
+ """Read a file from disk and slice it. See :func:`slice_bytes` for keywords.
255
+
256
+ :param name: override the file name used for handler dispatch (defaults to the
257
+ path's basename)
258
+ """
259
+ resolved = os.fspath(path)
260
+ with open(resolved, "rb") as handle:
261
+ content = handle.read()
262
+ return slice_bytes(
263
+ content,
264
+ name or os.path.basename(resolved),
265
+ limit=limit,
266
+ overlap=overlap,
267
+ overlap_ratio=overlap_ratio,
268
+ options=options,
269
+ **kwargs,
270
+ )
271
+
272
+
273
+ def extract_text(content: bytes, name: str, *, save_image: Optional[Callable] = None) -> str:
274
+ """Extraction only: return the document's text without slicing it.
275
+
276
+ Mirrors ``get_content`` on the handlers - useful for previews, for feeding an
277
+ LLM a whole (small) document, or for bundling archive members into one file.
278
+ """
279
+ file = BytesSplitFile(name, content)
280
+ buf = FileBufferHandle()
281
+ buf.buffer = content
282
+ get_buffer = buf.get_buffer
283
+ for handler in SPLIT_HANDLERS:
284
+ if not handler.support(file, get_buffer):
285
+ continue
286
+ sink = save_image or (lambda images: None)
287
+ from .handlers.image import ImageSplitHandle
288
+
289
+ if isinstance(handler, ImageSplitHandle):
290
+ # extended signature: OCR extractor + buffer access are keyword args
291
+ return handler.get_content(file, sink, get_buffer=get_buffer)
292
+ return handler.get_content(file, sink)
293
+ extension = ("." + name.rsplit(".", 1)[1].lower()) if "." in name else ""
294
+ raise UnsupportedFormatError(f"Unsupported file format{f': {extension}' if extension else ''}")
295
+
296
+
297
+ def detect_handler(name: str, content: Optional[bytes] = None) -> Optional[str]:
298
+ """Return the handler class name that would claim this input, or ``None``.
299
+
300
+ Introspection helper for CLIs, tests and "why did my file parse like that"
301
+ debugging. Passing ``content`` enables the content-sniffing handlers.
302
+ """
303
+ file = BytesSplitFile(name, content or b"")
304
+ buf = FileBufferHandle()
305
+ if content is not None:
306
+ buf.buffer = content
307
+ get_buffer = buf.get_buffer
308
+ for handler in SPLIT_HANDLERS:
309
+ try:
310
+ if handler.support(file, get_buffer):
311
+ return type(handler).__name__
312
+ except Exception: # noqa: BLE001 - a failed sniff must not break detection
313
+ continue
314
+ return None
315
+
316
+
317
+ def supported_extensions() -> Dict[str, List[str]]:
318
+ """Map handler name -> extensions it is documented for."""
319
+ return {name: list(exts) for name, exts in HANDLER_EXTENSIONS.items()}
320
+
321
+
322
+ def chunk_paragraphs(
323
+ paragraphs: Iterable[Paragraph],
324
+ *,
325
+ chunk_size: Optional[int] = None,
326
+ chunk_overlap: Optional[int] = None,
327
+ chunk_overlap_ratio: Optional[float] = None,
328
+ options: Optional[ChunkingOptions] = None,
329
+ carry_title: Optional[bool] = None,
330
+ handler: Optional[Any] = None,
331
+ ) -> List[str]:
332
+ """Chunk paragraphs down to an embedding window, optionally with overlap.
333
+
334
+ Slicing yields semantic paragraphs; embedding models still need fixed-size
335
+ input. The paragraph stays the retrieval unit, the chunk the vector unit.
336
+
337
+ Defaults mirror the historical behaviour (``chunk_size=256``, no overlap, no
338
+ title prefix); every one of them is overridable per call or through an
339
+ :class:`~smart_slice.options.ChunkingOptions`.
340
+
341
+ :param paragraphs: ``[{title, content}]`` from :func:`slice_bytes`
342
+ :param chunk_size: target chunk length in characters (default 256)
343
+ :param chunk_overlap: characters of context carried from the previous chunk
344
+ :param chunk_overlap_ratio: ``chunk_overlap`` as a fraction of ``chunk_size``
345
+ :param options: a :class:`ChunkingOptions` to reuse
346
+ :param carry_title: prefix **every** chunk with its paragraph's heading
347
+ chain (default False). Without this, chunking a long
348
+ paragraph leaves only its first chunk carrying section
349
+ context, which measurably hurts retrieval for the rest.
350
+ The prefix is additive: ``chunk_size`` still governs
351
+ the body length, so budget for the title in your
352
+ embedding window.
353
+ :param handler: any :class:`~smart_slice.chunking.IChunkHandle`
354
+ """
355
+ from .chunking import MarkChunkHandle, OverlapChunkHandle
356
+
357
+ opts = resolve_options(
358
+ options,
359
+ limit=chunk_size if chunk_size is not None else DEFAULT_CHUNK_SIZE,
360
+ overlap=chunk_overlap,
361
+ overlap_ratio=chunk_overlap_ratio,
362
+ )
363
+ if carry_title is not None:
364
+ opts = opts.with_(carry_title=carry_title)
365
+
366
+ handle = handler or (MarkChunkHandle() if opts.effective_overlap == 0 else OverlapChunkHandle())
367
+
368
+ # Pair each paragraph's text with its heading chain so the chain can be
369
+ # re-applied to *every* chunk produced from that paragraph (see below).
370
+ pairs = []
371
+ for row in paragraphs:
372
+ if isinstance(row, dict):
373
+ content = row.get("content")
374
+ title = str(row.get("title") or "").strip()
375
+ else:
376
+ content, title = row, ""
377
+ if isinstance(content, str) and content.strip():
378
+ pairs.append((content, title))
379
+
380
+ # Chunk each paragraph on its own: a paragraph is the semantic unit, and
381
+ # blending two sections into one vector hurts retrieval more than a little
382
+ # lost context helps.
383
+ chunks: List[str] = []
384
+ for text, title in pairs:
385
+ if isinstance(handle, OverlapChunkHandle):
386
+ pieces = handle.handle([text], opts)
387
+ else:
388
+ pieces = handle.handle([text], opts.limit)
389
+ if opts.carry_title and title:
390
+ prefix = f"# {title}\n"
391
+ chunks.extend(prefix + piece for piece in pieces)
392
+ else:
393
+ chunks.extend(pieces)
394
+ return chunks
395
+
396
+
397
+ def chunk(
398
+ text: str,
399
+ *,
400
+ chunk_size: Optional[int] = None,
401
+ chunk_overlap: Optional[int] = None,
402
+ chunk_overlap_ratio: Optional[float] = None,
403
+ options: Optional[ChunkingOptions] = None,
404
+ handler: Optional[Any] = None,
405
+ ) -> List[str]:
406
+ """Chunk a single string (convenience wrapper over :func:`chunk_paragraphs`)."""
407
+ return chunk_paragraphs(
408
+ [{"title": "", "content": text}],
409
+ chunk_size=chunk_size,
410
+ chunk_overlap=chunk_overlap,
411
+ chunk_overlap_ratio=chunk_overlap_ratio,
412
+ options=options,
413
+ handler=handler,
414
+ )
415
+
416
+
417
+ # re-exported lazily so ``import smart_slice`` stays cheap (jieba / parsers are
418
+ # only imported when the caller actually reaches for them)
419
+ def __getattr__(name: str) -> Any:
420
+ if name == "SplitModel":
421
+ from .chunker import SplitModel
422
+
423
+ return SplitModel
424
+ if name == "smart_split_paragraph":
425
+ from .chunker import smart_split_paragraph
426
+
427
+ return smart_split_paragraph
428
+ if name == "filter_special_char":
429
+ from .chunker import filter_special_char
430
+
431
+ return filter_special_char
432
+ if name == "MarkChunkHandle":
433
+ from .chunking import MarkChunkHandle
434
+
435
+ return MarkChunkHandle
436
+ if name == "OverlapChunkHandle":
437
+ from .chunking import OverlapChunkHandle
438
+
439
+ return OverlapChunkHandle
440
+ raise AttributeError(f"module {__name__!r} has no attribute {name!r}")
@@ -0,0 +1,188 @@
1
+ # coding=utf-8
2
+ """Command line interface: ``python -m smart_slice`` / ``smart-slice``."""
3
+ import argparse
4
+ import json
5
+ import os
6
+ import sys
7
+ from typing import List, Optional, Sequence
8
+
9
+ from . import (
10
+ DEFAULT_LIMIT,
11
+ ChunkingOptions,
12
+ __version__,
13
+ detect_handler,
14
+ missing_dependencies,
15
+ slice_path,
16
+ supported_extensions,
17
+ )
18
+ from .exceptions import SliceError
19
+
20
+
21
+ def _build_parser() -> argparse.ArgumentParser:
22
+ parser = argparse.ArgumentParser(
23
+ prog="smart-slice",
24
+ description="Fidelity-first document slicing for RAG pipelines.",
25
+ )
26
+ parser.add_argument("--version", action="version", version=f"smart-slice {__version__}")
27
+ sub = parser.add_subparsers(dest="command", required=True)
28
+
29
+ run = sub.add_parser("slice", help="slice a file into paragraphs")
30
+ run.add_argument("path", help="document to slice")
31
+ run.add_argument("--limit", type=int, default=DEFAULT_LIMIT, help=f"max chars per paragraph (default {DEFAULT_LIMIT})")
32
+ run.add_argument(
33
+ "--overlap", type=int, default=None,
34
+ help="characters of context shared by consecutive paragraphs (default 0 = none)",
35
+ )
36
+ run.add_argument(
37
+ "--overlap-ratio", type=float, default=None,
38
+ help="alternative to --overlap, as a fraction of --limit (e.g. 0.15 for 15%%)",
39
+ )
40
+ run.add_argument(
41
+ "--overlap-section-only", action="store_true",
42
+ help="carry context only between paragraphs in the same section",
43
+ )
44
+ run.add_argument("--no-filter", action="store_true", help="keep heading markers / blank runs verbatim")
45
+ run.add_argument(
46
+ "--format",
47
+ choices=("json", "jsonl", "text", "md"),
48
+ default="json",
49
+ help="output encoding (default json)",
50
+ )
51
+ run.add_argument("--output", "-o", help="write to a file instead of stdout")
52
+ run.add_argument("--title-prefix", action="store_true", help="prefix each paragraph with its title chain")
53
+ run.add_argument("--stats", action="store_true", help="print a paragraph/character summary to stderr")
54
+
55
+ chunk = sub.add_parser("chunk", help="slice, then chunk paragraphs for an embedding window")
56
+ chunk.add_argument("path", help="document to slice and chunk")
57
+ chunk.add_argument("--limit", type=int, default=DEFAULT_LIMIT, help="paragraph budget (default %(default)s)")
58
+ chunk.add_argument("--chunk-size", type=int, default=256, help="embedding chunk size (default %(default)s)")
59
+ chunk.add_argument("--chunk-overlap", type=int, default=None, help="chars shared by consecutive chunks")
60
+ chunk.add_argument("--carry-title", action="store_true", help="prefix each chunk with its heading chain")
61
+ chunk.add_argument("--output", "-o", help="write JSON lines to a file instead of stdout")
62
+ chunk.add_argument("--stats", action="store_true", help="print a chunk summary to stderr")
63
+
64
+ probe = sub.add_parser("detect", help="show which handler claims a file")
65
+ probe.add_argument("path", help="document to inspect")
66
+
67
+ info = sub.add_parser("formats", help="list supported extensions and missing optional deps")
68
+ info.add_argument("--json", action="store_true", help="machine readable output")
69
+ return parser
70
+
71
+
72
+ def _emit(rows: List[dict], fmt: str, output: Optional[str], title_prefix: bool) -> None:
73
+ def encode(row: dict) -> str:
74
+ content = row.get("content", "")
75
+ if title_prefix and row.get("title"):
76
+ content = f"# {row['title']}\n\n{content}"
77
+ return content
78
+
79
+ if fmt == "json":
80
+ payload = json.dumps(rows, ensure_ascii=False, indent=2)
81
+ elif fmt == "jsonl":
82
+ payload = "\n".join(json.dumps(row, ensure_ascii=False) for row in rows)
83
+ elif fmt == "text":
84
+ payload = "\n\n".join(encode(row) for row in rows)
85
+ else: # md
86
+ payload = "\n\n".join(
87
+ (f"# {row.get('title')}\n\n" if row.get("title") else "") + row.get("content", "") for row in rows
88
+ )
89
+
90
+ if output:
91
+ directory = os.path.dirname(os.path.abspath(output))
92
+ os.makedirs(directory, exist_ok=True)
93
+ with open(output, "w", encoding="utf-8", newline="\n") as handle:
94
+ handle.write(payload)
95
+ else:
96
+ sys.stdout.write(payload + "\n")
97
+
98
+
99
+ def main(argv: Optional[Sequence[str]] = None) -> int:
100
+ args = _build_parser().parse_args(argv)
101
+
102
+ if args.command == "formats":
103
+ extensions = supported_extensions()
104
+ gaps = missing_dependencies()
105
+ if args.json:
106
+ sys.stdout.write(json.dumps({"handlers": extensions, "missing_optional": gaps}, ensure_ascii=False, indent=2) + "\n")
107
+ return 0
108
+ for handler, exts in extensions.items():
109
+ sys.stdout.write(f"{handler:<22} {', '.join(exts)}\n")
110
+ if gaps:
111
+ sys.stdout.write("\nOptional extras not installed:\n")
112
+ for gap in gaps:
113
+ sys.stdout.write(f" smart-slice[{gap['extra']}] -> {gap['requirement']}\n")
114
+ return 0
115
+
116
+ if args.command == "detect":
117
+ with open(args.path, "rb") as handle:
118
+ content = handle.read()
119
+ handler = detect_handler(os.path.basename(args.path), content)
120
+ sys.stdout.write((handler or "<no handler>") + "\n")
121
+ return 0 if handler else 2
122
+
123
+ if args.command == "chunk":
124
+ from . import chunk_paragraphs
125
+
126
+ try:
127
+ paragraphs = slice_path(args.path, limit=args.limit)
128
+ except SliceError as error:
129
+ sys.stderr.write(f"smart-slice: {error.message} (code {error.code})\n")
130
+ return 1
131
+ except OSError as error:
132
+ sys.stderr.write(f"smart-slice: cannot read {args.path}: {error}\n")
133
+ return 1
134
+
135
+ pieces = chunk_paragraphs(
136
+ paragraphs,
137
+ chunk_size=args.chunk_size,
138
+ chunk_overlap=args.chunk_overlap,
139
+ carry_title=args.carry_title or None,
140
+ )
141
+ payload = "\n".join(json.dumps({"content": piece}, ensure_ascii=False) for piece in pieces)
142
+ if args.output:
143
+ directory = os.path.dirname(os.path.abspath(args.output))
144
+ os.makedirs(directory, exist_ok=True)
145
+ with open(args.output, "w", encoding="utf-8", newline="\n") as handle:
146
+ handle.write(payload + ("\n" if pieces else ""))
147
+ else:
148
+ sys.stdout.write(payload + ("\n" if pieces else ""))
149
+ if args.stats:
150
+ characters = sum(len(piece) for piece in pieces)
151
+ sys.stderr.write(
152
+ f"smart-slice: {len(pieces)} chunks from {len(paragraphs)} paragraphs, "
153
+ f"{characters} characters, chunk_size={args.chunk_size}, "
154
+ f"chunk_overlap={args.chunk_overlap or 0}\n"
155
+ )
156
+ return 0
157
+
158
+ # slice
159
+ try:
160
+ opts = ChunkingOptions(
161
+ limit=args.limit,
162
+ overlap=args.overlap,
163
+ overlap_ratio=args.overlap_ratio,
164
+ overlap_within_section=args.overlap_section_only,
165
+ )
166
+ result = slice_path(args.path, limit=args.limit, options=opts,
167
+ with_filter=not args.no_filter)
168
+ except SliceError as error:
169
+ sys.stderr.write(f"smart-slice: {error.message} (code {error.code})\n")
170
+ return 1
171
+ except OSError as error:
172
+ sys.stderr.write(f"smart-slice: cannot read {args.path}: {error}\n")
173
+ return 1
174
+
175
+ rows = [row for row in result if isinstance(row, dict)]
176
+ _emit(rows, args.format, args.output, args.title_prefix)
177
+ if args.stats:
178
+ characters = sum(len(row.get("content", "")) for row in rows)
179
+ overlap = opts.effective_overlap
180
+ sys.stderr.write(
181
+ f"smart-slice: {len(rows)} paragraphs, {characters} characters, "
182
+ f"limit={args.limit}, overlap={overlap}\n"
183
+ )
184
+ return 0
185
+
186
+
187
+ if __name__ == "__main__":
188
+ raise SystemExit(main())
smart_slice/_accel.py ADDED
@@ -0,0 +1,65 @@
1
+ # coding=utf-8
2
+ """Accelerator resolution: prefer the C scan, fall back to pure Python.
3
+
4
+ The chunker's hot path is "find the heading lines of this block". The regular
5
+ expression path runs up to six whole-block scans per recursion level (one per
6
+ heading level, cascading until one matches). A single linear scan that reports
7
+ every candidate heading line with its hash count replaces all of them.
8
+
9
+ This module resolves which scan implementation to use:
10
+
11
+ 1. ``smart_slice._speedup`` - the optional C extension (see csrc/_speedup.c);
12
+ 2. ``smart_slice._speedup_py`` - an identical pure-Python scan, always present.
13
+
14
+ Both return ``[(line_start, line_end, hashes), ...]`` with the same acceptance
15
+ rule, so the chunker is agnostic. ``ACCELERATOR`` records which one is active for
16
+ diagnostics; it never affects results (the equivalence test asserts that).
17
+ """
18
+ from typing import Callable, List, Optional, Tuple
19
+
20
+ from smart_slice._speedup_py import scan_heading_candidates as _py_scan
21
+
22
+ __all__ = [
23
+ "scan_heading_candidates",
24
+ "ACCELERATOR",
25
+ "CANONICAL_HEADING_LEVELS",
26
+ "heading_level_of",
27
+ ]
28
+
29
+ try: # optional C extension
30
+ from smart_slice._speedup import scan_heading_candidates as _c_scan # type: ignore
31
+
32
+ scan_heading_candidates: Callable[[str], List[Tuple[int, int, int]]] = _c_scan
33
+ ACCELERATOR = "c"
34
+ except Exception: # noqa: BLE001 - any import/build problem falls back cleanly
35
+ scan_heading_candidates = _py_scan
36
+ ACCELERATOR = "python"
37
+
38
+
39
+ # The canonical markdown heading patterns, by exact source string, mapped to their
40
+ # 1-based level. A pattern is eligible for the scan fast path only if its
41
+ # `.pattern` matches one of these verbatim - any custom or edited pattern takes the
42
+ # regular-expression path, so the optimisation can never change custom behaviour.
43
+ _CANONICAL_PATTERNS = (
44
+ r'(?<=^)# (?!-\*- coding:).*|(?<=\n)# (?!-\*- coding:).*',
45
+ r'(?<=\n)(?<!#)## (?!#).*|(?<=^)(?<!#)## (?!#).*',
46
+ r"(?<=\n)(?<!#)### (?!#).*|(?<=^)(?<!#)### (?!#).*",
47
+ r"(?<=\n)(?<!#)#### (?!#).*|(?<=^)(?<!#)#### (?!#).*",
48
+ r"(?<=\n)(?<!#)##### (?!#).*|(?<=^)(?<!#)##### (?!#).*",
49
+ r"(?<=\n)(?<!#)###### (?!#).*|(?<=^)(?<!#)###### (?!#).*",
50
+ )
51
+
52
+ CANONICAL_HEADING_LEVELS = {pattern: level + 1 for level, pattern in enumerate(_CANONICAL_PATTERNS)}
53
+
54
+
55
+ def heading_level_of(pattern) -> Optional[int]:
56
+ """Return the heading level (1-6) for a canonical heading pattern, else None.
57
+
58
+ ``None`` means "not a canonical markdown heading" - the caller must use the
59
+ regular-expression path for that pattern (custom schemes, the blank-line rule,
60
+ etc.). Accepts either a compiled pattern or a raw string.
61
+ """
62
+ source = getattr(pattern, "pattern", pattern)
63
+ if not isinstance(source, str):
64
+ return None
65
+ return CANONICAL_HEADING_LEVELS.get(source)