smart-slice 0.3.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (89) hide show
  1. smart_slice-0.3.0/.gitignore +52 -0
  2. smart_slice-0.3.0/CHANGELOG.md +226 -0
  3. smart_slice-0.3.0/LICENSE +674 -0
  4. smart_slice-0.3.0/PKG-INFO +1193 -0
  5. smart_slice-0.3.0/README.md +402 -0
  6. smart_slice-0.3.0/csrc/_speedup.c +135 -0
  7. smart_slice-0.3.0/docs/ARCHITECTURE.md +273 -0
  8. smart_slice-0.3.0/docs/PERFORMANCE.md +193 -0
  9. smart_slice-0.3.0/docs/PORTING.md +128 -0
  10. smart_slice-0.3.0/examples/basic_text.py +44 -0
  11. smart_slice-0.3.0/examples/cli_and_chunk.py +62 -0
  12. smart_slice-0.3.0/examples/slice_a_file.py +78 -0
  13. smart_slice-0.3.0/pyproject.toml +121 -0
  14. smart_slice-0.3.0/scripts/benchmark.py +176 -0
  15. smart_slice-0.3.0/scripts/build_ext.py +87 -0
  16. smart_slice-0.3.0/setup.py +34 -0
  17. smart_slice-0.3.0/smart_slice/__init__.py +440 -0
  18. smart_slice-0.3.0/smart_slice/__main__.py +188 -0
  19. smart_slice-0.3.0/smart_slice/_accel.py +65 -0
  20. smart_slice-0.3.0/smart_slice/_config.py +60 -0
  21. smart_slice-0.3.0/smart_slice/_i18n.py +37 -0
  22. smart_slice-0.3.0/smart_slice/_logging.py +22 -0
  23. smart_slice-0.3.0/smart_slice/_markdown.py +33 -0
  24. smart_slice-0.3.0/smart_slice/_optional.py +114 -0
  25. smart_slice-0.3.0/smart_slice/_speedup_py.py +64 -0
  26. smart_slice-0.3.0/smart_slice/_uuid.py +40 -0
  27. smart_slice-0.3.0/smart_slice/_validation.py +174 -0
  28. smart_slice-0.3.0/smart_slice/chunker.py +884 -0
  29. smart_slice-0.3.0/smart_slice/chunking/__init__.py +29 -0
  30. smart_slice-0.3.0/smart_slice/chunking/_base.py +14 -0
  31. smart_slice-0.3.0/smart_slice/chunking/mark.py +39 -0
  32. smart_slice-0.3.0/smart_slice/chunking/overlap.py +64 -0
  33. smart_slice-0.3.0/smart_slice/exceptions.py +60 -0
  34. smart_slice-0.3.0/smart_slice/files.py +130 -0
  35. smart_slice-0.3.0/smart_slice/handlers/__init__.py +205 -0
  36. smart_slice-0.3.0/smart_slice/handlers/_utils.py +55 -0
  37. smart_slice-0.3.0/smart_slice/handlers/_xlsx_images.py +163 -0
  38. smart_slice-0.3.0/smart_slice/handlers/archive.py +152 -0
  39. smart_slice-0.3.0/smart_slice/handlers/base.py +22 -0
  40. smart_slice-0.3.0/smart_slice/handlers/csv_handler.py +136 -0
  41. smart_slice-0.3.0/smart_slice/handlers/doc.py +289 -0
  42. smart_slice-0.3.0/smart_slice/handlers/eml.py +100 -0
  43. smart_slice-0.3.0/smart_slice/handlers/epub.py +138 -0
  44. smart_slice-0.3.0/smart_slice/handlers/fb2.py +108 -0
  45. smart_slice-0.3.0/smart_slice/handlers/html.py +113 -0
  46. smart_slice-0.3.0/smart_slice/handlers/image.py +90 -0
  47. smart_slice-0.3.0/smart_slice/handlers/image_text_extract.py +125 -0
  48. smart_slice-0.3.0/smart_slice/handlers/ipynb.py +126 -0
  49. smart_slice-0.3.0/smart_slice/handlers/mbox.py +137 -0
  50. smart_slice-0.3.0/smart_slice/handlers/mhtml.py +96 -0
  51. smart_slice-0.3.0/smart_slice/handlers/mobi.py +93 -0
  52. smart_slice-0.3.0/smart_slice/handlers/msg.py +88 -0
  53. smart_slice-0.3.0/smart_slice/handlers/odf.py +239 -0
  54. smart_slice-0.3.0/smart_slice/handlers/pdf.py +634 -0
  55. smart_slice-0.3.0/smart_slice/handlers/ppt.py +134 -0
  56. smart_slice-0.3.0/smart_slice/handlers/pptx.py +160 -0
  57. smart_slice-0.3.0/smart_slice/handlers/raster_image.py +52 -0
  58. smart_slice-0.3.0/smart_slice/handlers/rtf.py +71 -0
  59. smart_slice-0.3.0/smart_slice/handlers/subtitle.py +103 -0
  60. smart_slice-0.3.0/smart_slice/handlers/svg.py +100 -0
  61. smart_slice-0.3.0/smart_slice/handlers/text.py +119 -0
  62. smart_slice-0.3.0/smart_slice/handlers/vcalendar.py +221 -0
  63. smart_slice-0.3.0/smart_slice/handlers/wps.py +54 -0
  64. smart_slice-0.3.0/smart_slice/handlers/xls.py +159 -0
  65. smart_slice-0.3.0/smart_slice/handlers/xlsx.py +262 -0
  66. smart_slice-0.3.0/smart_slice/handlers/xmind.py +170 -0
  67. smart_slice-0.3.0/smart_slice/handlers/zip_handler.py +306 -0
  68. smart_slice-0.3.0/smart_slice/options.py +225 -0
  69. smart_slice-0.3.0/smart_slice/patterns.py +69 -0
  70. smart_slice-0.3.0/smart_slice/qa/__init__.py +56 -0
  71. smart_slice-0.3.0/smart_slice/qa/_base.py +50 -0
  72. smart_slice-0.3.0/smart_slice/qa/_table_base.py +21 -0
  73. smart_slice-0.3.0/smart_slice/qa/csv_qa.py +64 -0
  74. smart_slice-0.3.0/smart_slice/qa/csv_table.py +78 -0
  75. smart_slice-0.3.0/smart_slice/qa/md_qa.py +146 -0
  76. smart_slice-0.3.0/smart_slice/qa/xls_qa.py +67 -0
  77. smart_slice-0.3.0/smart_slice/qa/xls_table.py +102 -0
  78. smart_slice-0.3.0/smart_slice/qa/xlsx_qa.py +78 -0
  79. smart_slice-0.3.0/smart_slice/qa/xlsx_table.py +129 -0
  80. smart_slice-0.3.0/smart_slice/qa/zip_qa.py +157 -0
  81. smart_slice-0.3.0/smart_slice/service.py +486 -0
  82. smart_slice-0.3.0/smart_slice/types.py +55 -0
  83. smart_slice-0.3.0/tests/test_c_speedup.py +272 -0
  84. smart_slice-0.3.0/tests/test_fidelity.py +782 -0
  85. smart_slice-0.3.0/tests/test_format_expansion.py +364 -0
  86. smart_slice-0.3.0/tests/test_full_format.py +440 -0
  87. smart_slice-0.3.0/tests/test_options_overlap.py +341 -0
  88. smart_slice-0.3.0/tests/test_public_api.py +158 -0
  89. smart_slice-0.3.0/tests/test_service.py +485 -0
@@ -0,0 +1,52 @@
1
+ # Byte-compiled / cache
2
+ __pycache__/
3
+ *.py[cod]
4
+ *$py.class
5
+
6
+ # Distribution / packaging
7
+ build/
8
+ dist/
9
+ downloads/
10
+ eggs/
11
+ .eggs/
12
+ *.egg-info/
13
+ *.egg
14
+ .installed.cfg
15
+ wheels/
16
+ share/python-wheels/
17
+
18
+ # Virtual environments
19
+ .venv/
20
+ venv/
21
+ env/
22
+ ENV/
23
+
24
+ # Test / lint / type caches
25
+ .pytest_cache/
26
+ .ruff_cache/
27
+ .mypy_cache/
28
+ .tox/
29
+ .coverage
30
+ .coverage.*
31
+ htmlcov/
32
+ coverage.xml
33
+ *.cover
34
+
35
+ # Editors / OS
36
+ .idea/
37
+ .vscode/
38
+ *.swp
39
+ .DS_Store
40
+ Thumbs.db
41
+
42
+ # Compiled extension / build artefacts
43
+ *.pyd
44
+ *.so
45
+ *.dylib
46
+ *.obj
47
+ *.o
48
+ *.exp
49
+ *.lib
50
+ *.pdb
51
+ *.ilk
52
+ build/
@@ -0,0 +1,226 @@
1
+ # Changelog
2
+
3
+ All notable changes to this project are documented in this file.
4
+
5
+ The format follows [Keep a Changelog](https://keepachangelog.com/en/1.1.0/),
6
+ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
7
+
8
+ ## [0.3.0] - 2026-09-27
9
+
10
+ Two themes: a **correctness fix for the published wheel** (a clean install could not
11
+ even `import smart_slice`) and a **broad format expansion** (78 -> 197 declared
12
+ extensions, 22 -> 30 handlers).
13
+
14
+ ### Fixed
15
+
16
+ - **The wheel crashed on import in a clean environment.** 34 module-level imports of
17
+ optional parsers (`py7zr`, `python-docx`, `openpyxl`, `pypdf`, `markdownify`, ...)
18
+ were unguarded, so `pip install smart-slice` (core only) raised
19
+ `ModuleNotFoundError` at `import smart_slice` - directly contradicting the documented
20
+ "a handler whose parser is missing simply reports unsupported instead of crashing".
21
+ Every optional import is now wrapped, and each handler's `support()` returns False
22
+ when its parser is absent, so the format degrades to `400 Unsupported file format:
23
+ .pdf` instead of breaking the package. Verified in a bare venv with only
24
+ `charset-normalizer` installed.
25
+ - **Legacy binary `.doc` (OLE compound file) reported `500` instead of `400`.**
26
+ `DocSplitHandle` claimed `.doc` by extension, then handed it to `python-docx`, which
27
+ raised `BadZipFile`. A binary `.doc` is now detected by its OLE magic and rejected
28
+ with `400 ... convert the document to .docx first`, matching the existing
29
+ `.wps`/`.et` legacy-binary behaviour. `DocSplitHandle.handle` also gained the
30
+ `except (SliceError, ResourceLimitError): raise` passthrough the other handlers have,
31
+ so a 400 is no longer swallowed and re-raised as 500.
32
+ - **Source/config extensions were silently mis-sliced.** `.java`, `.go`, `.c`, `.rs`,
33
+ `.properties`, `.adoc`, `.diff` and ~50 more were only reachable through the
34
+ "decodes as text" fallback, which used the *markdown* heading patterns - so a
35
+ `# comment` line in source code was extracted as a section title. They are now first-
36
+ class members of `TEXT_EXTENSIONS`, which routes them to `LITERAL_EXTENSIONS`
37
+ (blank-line/length splitting only); a `#` in code stays in the body.
38
+ - **mbox bodies were dropped.** `mailbox.mbox` builds legacy `Compat32` messages that
39
+ have no `get_content()`, so every body part failed and was skipped. The mailbox is now
40
+ parsed with `policy=email.policy.default` (matching `EmlSplitHandle`); multi-message
41
+ archives produce one correctly-titled paragraph per message.
42
+
43
+ ### Added
44
+
45
+ - **8 new handlers**, all pure-stdlib so they work even in a bare install:
46
+ - `SvgSplitHandle` - `.svg` `.svgz` (gzip-aware): extracts visible `<text>`/`<tspan>`,
47
+ drops `<script>`/`<style>`/`<defs>`.
48
+ - `IpynbSplitHandle` - `.ipynb`: markdown cells kept, code cells fenced (so their `#`
49
+ comments are not mistaken for headings), `text/plain` outputs collected.
50
+ - `SubtitleSplitHandle` - `.srt` `.vtt` `.ass` `.ssa` `.sub`: strips indices,
51
+ timecodes and ASS `Script Info`/`Format` metadata, keeps only cue text.
52
+ - `Fb2SplitHandle` - `.fb2` `.fb2.zip`: FictionBook XML; section titles map to
53
+ Markdown headings. Registered *before* `ZipSplitHandle` so `.fb2.zip` is not treated
54
+ as a generic archive.
55
+ - `MboxSplitHandle` - `.mbox`: per-message subject + body.
56
+ - `VcalendarSplitHandle` / `VcardSplitHandle` - `.ics` `.ifb` / `.vcf`: unfold
57
+ RFC 5545/6350 lines, keep only human-readable fields (event summary -> title;
58
+ time/location/description, or name/phone/email/address -> key/value rows).
59
+ - `ExtendedImageSplitHandle` - 25 more Pillow-decodable raster containers
60
+ (`.ico` `.tga` `.pcx` `.dds` `.sgi` `.ppm`/`.pgm`/`.pbm` `.im` `.icns` `.qoi`
61
+ `.jfif` `.apng` `.xbm` `.psd` + aliases). Subclasses `ImageSplitHandle`, so the OCR
62
+ branch and `save_image` contract are inherited unchanged.
63
+ - **OOXML variants that were declared but rejected now parse:** `.pptm` `.ppsm` `.potm`
64
+ (added to the presentation content-type map) and `.dotx` `.dotm` `.xltx` `.xltm`
65
+ (declared and claimed). `.ppsx`/`.potx` content types were also completed.
66
+ - **`.tar.xz` / `.txz`** (stdlib `tarfile` already supported xz), **`.azw1`/`.azw4`/
67
+ `.prc`** Kindle variants, **`.xhtml`/`.shtml`**, **`.tab`** (CSV), **`.xlt`** (binary
68
+ Excel template) are now recognised and declared.
69
+ - **`smart_slice._optional`** - single source of truth mapping each optional import to
70
+ its pip extra, powering both the `support()` gates and the "run pip install
71
+ smart-slice[office] to enable it" error text.
72
+ - Tests: `tests/test_format_expansion.py` (32 cases) covering every new handler, the
73
+ bare-environment degradation, the source-fidelity fix, and the OOXML variant
74
+ declarations. Suite is now 216 tests.
75
+
76
+ ### Changed
77
+
78
+ - `HANDLER_EXTENSIONS` grew from 78 to 197 declared extensions across 30 handlers;
79
+ README / README_CN format tables were regenerated from the live registry and now
80
+ include a parser column and an explicit "not supported (400)" table.
81
+ - The zip/tar/7z **inner-file** dispatch list was extended with the new structured
82
+ text handlers (`Fb2`, `Svg`, `Ipynb`, `Subtitle`, `Mbox`, `Vcalendar`, `Vcard`), so
83
+ those formats are also recognised inside archives. Image handlers stay out of that
84
+ list on purpose: excluding `ImageSplitHandle` from archive members is a deliberate
85
+ Phase 2-B decision (embedded images travel through the zip handler's own markdown
86
+ reference collection + `save_image` path), pinned by
87
+ `test_zip_inner_images_do_not_receive_extractor`. Adding it back was attempted and
88
+ reverted; changing embedded-image semantics needs its own reviewed change.
89
+
90
+ ## [0.2.0] - 2026-09-27
91
+
92
+ Adds configurable chunking (size **and** overlap) in both default and fully
93
+ user-controlled modes, plus a substantial performance pass on the slicing engine.
94
+
95
+ ### Added
96
+
97
+ - **`ChunkingOptions`** (`smart_slice.options`) - one immutable object describing
98
+ chunk size and overlap behaviour: `limit`, `overlap`, `overlap_ratio`,
99
+ `boundary`, `lookback`, `overlap_boundary`, `min_chunk`, `carry_title`,
100
+ `overlap_within_section`, and `length_fn` (measure size in tokens instead of
101
+ characters by passing a tokenizer).
102
+ - **Paragraph overlap** - consecutive paragraphs can share trailing context.
103
+ Exposed at every entry point as `overlap=` / `overlap_ratio=` / `options=`,
104
+ with `overlap=0` (the default) preserving the previous exact-tiling behaviour.
105
+ - **`apply_paragraph_overlap`** - overlap is applied as a distinct pass *after*
106
+ the heading tree is assembled, never inside it (see Fixed for why). It is purely
107
+ additive: no paragraph's own text is altered, dropped or reordered.
108
+ - **Second-stage chunk overlap** - `chunk_paragraphs(..., chunk_overlap=,
109
+ chunk_overlap_ratio=)` and `OverlapChunkHandle`, so embedding chunks can also
110
+ carry context.
111
+ - **`carry_title`** - prefix *every* chunk with its paragraph's heading chain, so
112
+ section context survives chunking instead of appearing only in the first chunk.
113
+ - **`overlap_within_section`** - restrict overlap to paragraphs inside the same
114
+ heading chain.
115
+ - **CLI**: `slice --overlap N`, `--overlap-ratio F`, `--overlap-section-only`; a
116
+ new `chunk` subcommand (`--chunk-size`, `--chunk-overlap`, `--carry-title`).
117
+ - **Optional C accelerator** (`csrc/_speedup.c`, extra `accel`) - a single-pass
118
+ heading scan. Purely an optimisation: `smart_slice._speedup_py` provides an
119
+ identical pure-Python scan and is used whenever the extension is absent.
120
+ - **`scripts/build_ext.py`** - builds the extension in place, using `zig cc` on
121
+ Windows where setuptools ignores `CC` and insists on `cl.exe`.
122
+ - Tests: `tests/test_options_overlap.py` (45) and `tests/test_c_speedup.py` (19),
123
+ including the equivalence property that the heading prefilter never changes
124
+ slicing output.
125
+
126
+ ### Changed
127
+
128
+ - **Performance: ~2.7x faster slicing** on a 550 KB structured corpus
129
+ (394 ms -> 144 ms; 1.4 -> 3.8 MB/s), from algorithmic fixes rather than the C
130
+ extension:
131
+ - `parse_title_level` no longer runs up to six whole-document regex scans per
132
+ recursion level. One linear scan establishes which heading levels are present
133
+ and provably-empty levels are skipped. Regex calls: 48,636 -> 11,604 per
134
+ document.
135
+ - `mask_code_blocks` returns the input unchanged when there is no code fence
136
+ (previously it always allocated a full character list and re-joined it).
137
+ Masking calls: 97,272 -> 11,604.
138
+ - Masked text is memoised across heading levels for the same block.
139
+ - Compiled patterns bypass `re.findall`'s dispatch, removing 194,544
140
+ `re._compile` cache lookups per document.
141
+ - `re_findall` flattens match groups in one pass instead of
142
+ `reduce([*x, *y], ...)`, which was O(n^2) in the number of matches.
143
+ - Total Python calls per document: 4.33M -> 1.55M (-64%).
144
+ - The C extension adds ~4% on top of that (139 vs 145 ms end-to-end), because the
145
+ algorithmic pass had already removed most of the scan cost. Measured, not
146
+ assumed - the extension is worth having for heading-dense documents but is not
147
+ where the win came from.
148
+ - `smart_slice/chunker.py` **graduated from generated to first-party source**:
149
+ chunking policy now lives there, so it is no longer regenerated from the source
150
+ platform. Documented in `docs/PORTING.md`; `CONTRIBUTING.md` explains which
151
+ files are generated and which are editable.
152
+ - Sentence boundaries now include the fullwidth `!` (U+FF01) and `?` (U+FF1F).
153
+ The previous implementation listed ASCII `!` and `?` twice, so CJK text ending in
154
+ a fullwidth mark was never treated as a sentence end (see Fixed).
155
+
156
+ ### Fixed
157
+
158
+ - **Overlap inside the heading tree corrupted paragraph boundaries.** `parse_to_tree`
159
+ recovers block positions with `str.index()` over produced content, so chunks must
160
+ stay disjoint substrings of the source; overlapping them makes the lookup
161
+ ambiguous and silently reshuffles boundaries (symptom: identical total character
162
+ count, different paragraph split). Overlap is therefore applied after assembly.
163
+ - **Fullwidth `!`/`?` were not sentence boundaries.** The source `split_chars`
164
+ list contained ASCII `!` and `?` twice each, never U+FF01/U+FF1F, despite the
165
+ comment claiming "中英文感叹号/问号". Corrected, with a regression test.
166
+ - **`post_handler_paragraph` raised `NameError`.** It called `functools.reduce`,
167
+ which the module never imported. The function had no callers in the source
168
+ platform or in this package, so the defect stayed latent; the `reduce` flatten
169
+ (O(n^2)) is now a single pass and the two `if len(x) > 4096: pass` no-op
170
+ statements are gone.
171
+ - **`ChunkingOptions.with_(overlap_ratio=...)` was a no-op.** `__post_init__`
172
+ materialised `overlap=None` to `0`, which then outranked the ratio. `overlap`
173
+ keeps its `None` sentinel and resolution happens in `effective_overlap`.
174
+ - **`carry_title` only prefixed the first chunk of a paragraph.** It now prefixes
175
+ every chunk, which is the point of the option.
176
+ - **Importing the package mutated global Pillow state.** `_xlsx_images` set
177
+ `ImageFile.LOAD_TRUNCATED_IMAGES = True` and `Image.MAX_IMAGE_PIXELS = None` at
178
+ import time, silently disabling the *host process's* decompression-bomb guard.
179
+ Both are now scoped to `_permissive_pil()` around the image parse and restored
180
+ afterwards.
181
+
182
+ ## [0.1.0] - 2026-09-26
183
+
184
+ Initial public release, extracted from an enterprise agent platform's document
185
+ ingestion layer. The parsing and slicing behaviour is carried over verbatim; the
186
+ web-framework coupling is removed.
187
+
188
+ ### Added
189
+
190
+ - **Slicing engine** (`smart_slice.chunker`): heading-tree parsing, blank-line and
191
+ sentence-boundary splitting, a length budget (`limit`), heading-chain `title`
192
+ output, fidelity-preserving cleaning (line-start, code-fence-aware heading
193
+ marker removal), markdown table header re-insertion across cut boundaries, and
194
+ oversized-row re-splitting.
195
+ - **~30 format handlers** dispatched by a single ordered registry
196
+ (`SPLIT_HANDLERS`): HTML, MHTML, DOCX/DOCM, PDF, XLSX/XLSM, XLS, CSV/TSV, ZIP,
197
+ XMind, PPTX family, PPT, WPS/ET, RTF, ODF family, EPUB, EML, MSG, MOBI/AZW,
198
+ TAR family, 7z, images, and a text/source-code fallback.
199
+ - **Public API**: `slice_text`, `slice_bytes`, `slice_path`, `extract_text`,
200
+ `detect_handler`, `supported_extensions`, `missing_dependencies`,
201
+ `chunk_paragraphs`, `chunk`, plus the low-level `split_document`.
202
+ - **Command line interface** (`smart-slice` / `python -m smart_slice`) with
203
+ `slice`, `detect`, and `formats` subcommands and json/jsonl/text/md output.
204
+ - **QA and table parsers** (`smart_slice.qa`): question/answer extraction from
205
+ markdown, CSV and spreadsheets, and header-aware table row expansion.
206
+ - **Embedding chunking** (`smart_slice.chunking`): second-stage fixed-size
207
+ chunking of already-sliced paragraphs.
208
+ - **Image pipeline**: images extracted from documents are delivered through a
209
+ `save_image` callback as `ImageAsset` objects, with reference remapping for
210
+ deduplication and an optional local-OCR text-extraction hook.
211
+ - **Defensive parsing guards** (`ParserLimits`): byte, XML node/depth, archive
212
+ member, MIME part/depth and table-cell caps, configurable via
213
+ `SMART_SLICE_PARSER_*` environment variables.
214
+ - **Optional-dependency model**: a minimal core install plus per-format extras;
215
+ handlers degrade to "unsupported" rather than crashing when a parser is absent.
216
+ - **Test suite**: the source platform's slicing tests ported to run standalone
217
+ plus public-API and CLI tests.
218
+
219
+ ### Notes
220
+
221
+ - Local image OCR is disabled by default; set `SMART_SLICE_OCR_ENABLED=1` and
222
+ install the `ocr` extra to enable it.
223
+ - Licensed under GPL-3.0, matching the upstream platform.
224
+
225
+ [0.2.0]: https://github.com/guangxiangdebizi/smart-slice/compare/v0.1.0...v0.2.0
226
+ [0.1.0]: https://github.com/guangxiangdebizi/smart-slice/releases/tag/v0.1.0