smart-slice 0.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- smart_slice-0.3.0/.gitignore +52 -0
- smart_slice-0.3.0/CHANGELOG.md +226 -0
- smart_slice-0.3.0/LICENSE +674 -0
- smart_slice-0.3.0/PKG-INFO +1193 -0
- smart_slice-0.3.0/README.md +402 -0
- smart_slice-0.3.0/csrc/_speedup.c +135 -0
- smart_slice-0.3.0/docs/ARCHITECTURE.md +273 -0
- smart_slice-0.3.0/docs/PERFORMANCE.md +193 -0
- smart_slice-0.3.0/docs/PORTING.md +128 -0
- smart_slice-0.3.0/examples/basic_text.py +44 -0
- smart_slice-0.3.0/examples/cli_and_chunk.py +62 -0
- smart_slice-0.3.0/examples/slice_a_file.py +78 -0
- smart_slice-0.3.0/pyproject.toml +121 -0
- smart_slice-0.3.0/scripts/benchmark.py +176 -0
- smart_slice-0.3.0/scripts/build_ext.py +87 -0
- smart_slice-0.3.0/setup.py +34 -0
- smart_slice-0.3.0/smart_slice/__init__.py +440 -0
- smart_slice-0.3.0/smart_slice/__main__.py +188 -0
- smart_slice-0.3.0/smart_slice/_accel.py +65 -0
- smart_slice-0.3.0/smart_slice/_config.py +60 -0
- smart_slice-0.3.0/smart_slice/_i18n.py +37 -0
- smart_slice-0.3.0/smart_slice/_logging.py +22 -0
- smart_slice-0.3.0/smart_slice/_markdown.py +33 -0
- smart_slice-0.3.0/smart_slice/_optional.py +114 -0
- smart_slice-0.3.0/smart_slice/_speedup_py.py +64 -0
- smart_slice-0.3.0/smart_slice/_uuid.py +40 -0
- smart_slice-0.3.0/smart_slice/_validation.py +174 -0
- smart_slice-0.3.0/smart_slice/chunker.py +884 -0
- smart_slice-0.3.0/smart_slice/chunking/__init__.py +29 -0
- smart_slice-0.3.0/smart_slice/chunking/_base.py +14 -0
- smart_slice-0.3.0/smart_slice/chunking/mark.py +39 -0
- smart_slice-0.3.0/smart_slice/chunking/overlap.py +64 -0
- smart_slice-0.3.0/smart_slice/exceptions.py +60 -0
- smart_slice-0.3.0/smart_slice/files.py +130 -0
- smart_slice-0.3.0/smart_slice/handlers/__init__.py +205 -0
- smart_slice-0.3.0/smart_slice/handlers/_utils.py +55 -0
- smart_slice-0.3.0/smart_slice/handlers/_xlsx_images.py +163 -0
- smart_slice-0.3.0/smart_slice/handlers/archive.py +152 -0
- smart_slice-0.3.0/smart_slice/handlers/base.py +22 -0
- smart_slice-0.3.0/smart_slice/handlers/csv_handler.py +136 -0
- smart_slice-0.3.0/smart_slice/handlers/doc.py +289 -0
- smart_slice-0.3.0/smart_slice/handlers/eml.py +100 -0
- smart_slice-0.3.0/smart_slice/handlers/epub.py +138 -0
- smart_slice-0.3.0/smart_slice/handlers/fb2.py +108 -0
- smart_slice-0.3.0/smart_slice/handlers/html.py +113 -0
- smart_slice-0.3.0/smart_slice/handlers/image.py +90 -0
- smart_slice-0.3.0/smart_slice/handlers/image_text_extract.py +125 -0
- smart_slice-0.3.0/smart_slice/handlers/ipynb.py +126 -0
- smart_slice-0.3.0/smart_slice/handlers/mbox.py +137 -0
- smart_slice-0.3.0/smart_slice/handlers/mhtml.py +96 -0
- smart_slice-0.3.0/smart_slice/handlers/mobi.py +93 -0
- smart_slice-0.3.0/smart_slice/handlers/msg.py +88 -0
- smart_slice-0.3.0/smart_slice/handlers/odf.py +239 -0
- smart_slice-0.3.0/smart_slice/handlers/pdf.py +634 -0
- smart_slice-0.3.0/smart_slice/handlers/ppt.py +134 -0
- smart_slice-0.3.0/smart_slice/handlers/pptx.py +160 -0
- smart_slice-0.3.0/smart_slice/handlers/raster_image.py +52 -0
- smart_slice-0.3.0/smart_slice/handlers/rtf.py +71 -0
- smart_slice-0.3.0/smart_slice/handlers/subtitle.py +103 -0
- smart_slice-0.3.0/smart_slice/handlers/svg.py +100 -0
- smart_slice-0.3.0/smart_slice/handlers/text.py +119 -0
- smart_slice-0.3.0/smart_slice/handlers/vcalendar.py +221 -0
- smart_slice-0.3.0/smart_slice/handlers/wps.py +54 -0
- smart_slice-0.3.0/smart_slice/handlers/xls.py +159 -0
- smart_slice-0.3.0/smart_slice/handlers/xlsx.py +262 -0
- smart_slice-0.3.0/smart_slice/handlers/xmind.py +170 -0
- smart_slice-0.3.0/smart_slice/handlers/zip_handler.py +306 -0
- smart_slice-0.3.0/smart_slice/options.py +225 -0
- smart_slice-0.3.0/smart_slice/patterns.py +69 -0
- smart_slice-0.3.0/smart_slice/qa/__init__.py +56 -0
- smart_slice-0.3.0/smart_slice/qa/_base.py +50 -0
- smart_slice-0.3.0/smart_slice/qa/_table_base.py +21 -0
- smart_slice-0.3.0/smart_slice/qa/csv_qa.py +64 -0
- smart_slice-0.3.0/smart_slice/qa/csv_table.py +78 -0
- smart_slice-0.3.0/smart_slice/qa/md_qa.py +146 -0
- smart_slice-0.3.0/smart_slice/qa/xls_qa.py +67 -0
- smart_slice-0.3.0/smart_slice/qa/xls_table.py +102 -0
- smart_slice-0.3.0/smart_slice/qa/xlsx_qa.py +78 -0
- smart_slice-0.3.0/smart_slice/qa/xlsx_table.py +129 -0
- smart_slice-0.3.0/smart_slice/qa/zip_qa.py +157 -0
- smart_slice-0.3.0/smart_slice/service.py +486 -0
- smart_slice-0.3.0/smart_slice/types.py +55 -0
- smart_slice-0.3.0/tests/test_c_speedup.py +272 -0
- smart_slice-0.3.0/tests/test_fidelity.py +782 -0
- smart_slice-0.3.0/tests/test_format_expansion.py +364 -0
- smart_slice-0.3.0/tests/test_full_format.py +440 -0
- smart_slice-0.3.0/tests/test_options_overlap.py +341 -0
- smart_slice-0.3.0/tests/test_public_api.py +158 -0
- smart_slice-0.3.0/tests/test_service.py +485 -0
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
# Byte-compiled / cache
|
|
2
|
+
__pycache__/
|
|
3
|
+
*.py[cod]
|
|
4
|
+
*$py.class
|
|
5
|
+
|
|
6
|
+
# Distribution / packaging
|
|
7
|
+
build/
|
|
8
|
+
dist/
|
|
9
|
+
downloads/
|
|
10
|
+
eggs/
|
|
11
|
+
.eggs/
|
|
12
|
+
*.egg-info/
|
|
13
|
+
*.egg
|
|
14
|
+
.installed.cfg
|
|
15
|
+
wheels/
|
|
16
|
+
share/python-wheels/
|
|
17
|
+
|
|
18
|
+
# Virtual environments
|
|
19
|
+
.venv/
|
|
20
|
+
venv/
|
|
21
|
+
env/
|
|
22
|
+
ENV/
|
|
23
|
+
|
|
24
|
+
# Test / lint / type caches
|
|
25
|
+
.pytest_cache/
|
|
26
|
+
.ruff_cache/
|
|
27
|
+
.mypy_cache/
|
|
28
|
+
.tox/
|
|
29
|
+
.coverage
|
|
30
|
+
.coverage.*
|
|
31
|
+
htmlcov/
|
|
32
|
+
coverage.xml
|
|
33
|
+
*.cover
|
|
34
|
+
|
|
35
|
+
# Editors / OS
|
|
36
|
+
.idea/
|
|
37
|
+
.vscode/
|
|
38
|
+
*.swp
|
|
39
|
+
.DS_Store
|
|
40
|
+
Thumbs.db
|
|
41
|
+
|
|
42
|
+
# Compiled extension / build artefacts
|
|
43
|
+
*.pyd
|
|
44
|
+
*.so
|
|
45
|
+
*.dylib
|
|
46
|
+
*.obj
|
|
47
|
+
*.o
|
|
48
|
+
*.exp
|
|
49
|
+
*.lib
|
|
50
|
+
*.pdb
|
|
51
|
+
*.ilk
|
|
52
|
+
build/
|
|
@@ -0,0 +1,226 @@
|
|
|
1
|
+
# Changelog
|
|
2
|
+
|
|
3
|
+
All notable changes to this project are documented in this file.
|
|
4
|
+
|
|
5
|
+
The format follows [Keep a Changelog](https://keepachangelog.com/en/1.1.0/),
|
|
6
|
+
and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
|
|
7
|
+
|
|
8
|
+
## [0.3.0] - 2026-09-27
|
|
9
|
+
|
|
10
|
+
Two themes: a **correctness fix for the published wheel** (a clean install could not
|
|
11
|
+
even `import smart_slice`) and a **broad format expansion** (78 -> 197 declared
|
|
12
|
+
extensions, 22 -> 30 handlers).
|
|
13
|
+
|
|
14
|
+
### Fixed
|
|
15
|
+
|
|
16
|
+
- **The wheel crashed on import in a clean environment.** 34 module-level imports of
|
|
17
|
+
optional parsers (`py7zr`, `python-docx`, `openpyxl`, `pypdf`, `markdownify`, ...)
|
|
18
|
+
were unguarded, so `pip install smart-slice` (core only) raised
|
|
19
|
+
`ModuleNotFoundError` at `import smart_slice` - directly contradicting the documented
|
|
20
|
+
"a handler whose parser is missing simply reports unsupported instead of crashing".
|
|
21
|
+
Every optional import is now wrapped, and each handler's `support()` returns False
|
|
22
|
+
when its parser is absent, so the format degrades to `400 Unsupported file format:
|
|
23
|
+
.pdf` instead of breaking the package. Verified in a bare venv with only
|
|
24
|
+
`charset-normalizer` installed.
|
|
25
|
+
- **Legacy binary `.doc` (OLE compound file) reported `500` instead of `400`.**
|
|
26
|
+
`DocSplitHandle` claimed `.doc` by extension, then handed it to `python-docx`, which
|
|
27
|
+
raised `BadZipFile`. A binary `.doc` is now detected by its OLE magic and rejected
|
|
28
|
+
with `400 ... convert the document to .docx first`, matching the existing
|
|
29
|
+
`.wps`/`.et` legacy-binary behaviour. `DocSplitHandle.handle` also gained the
|
|
30
|
+
`except (SliceError, ResourceLimitError): raise` passthrough the other handlers have,
|
|
31
|
+
so a 400 is no longer swallowed and re-raised as 500.
|
|
32
|
+
- **Source/config extensions were silently mis-sliced.** `.java`, `.go`, `.c`, `.rs`,
|
|
33
|
+
`.properties`, `.adoc`, `.diff` and ~50 more were only reachable through the
|
|
34
|
+
"decodes as text" fallback, which used the *markdown* heading patterns - so a
|
|
35
|
+
`# comment` line in source code was extracted as a section title. They are now first-
|
|
36
|
+
class members of `TEXT_EXTENSIONS`, which routes them to `LITERAL_EXTENSIONS`
|
|
37
|
+
(blank-line/length splitting only); a `#` in code stays in the body.
|
|
38
|
+
- **mbox bodies were dropped.** `mailbox.mbox` builds legacy `Compat32` messages that
|
|
39
|
+
have no `get_content()`, so every body part failed and was skipped. The mailbox is now
|
|
40
|
+
parsed with `policy=email.policy.default` (matching `EmlSplitHandle`); multi-message
|
|
41
|
+
archives produce one correctly-titled paragraph per message.
|
|
42
|
+
|
|
43
|
+
### Added
|
|
44
|
+
|
|
45
|
+
- **8 new handlers**, all pure-stdlib so they work even in a bare install:
|
|
46
|
+
- `SvgSplitHandle` - `.svg` `.svgz` (gzip-aware): extracts visible `<text>`/`<tspan>`,
|
|
47
|
+
drops `<script>`/`<style>`/`<defs>`.
|
|
48
|
+
- `IpynbSplitHandle` - `.ipynb`: markdown cells kept, code cells fenced (so their `#`
|
|
49
|
+
comments are not mistaken for headings), `text/plain` outputs collected.
|
|
50
|
+
- `SubtitleSplitHandle` - `.srt` `.vtt` `.ass` `.ssa` `.sub`: strips indices,
|
|
51
|
+
timecodes and ASS `Script Info`/`Format` metadata, keeps only cue text.
|
|
52
|
+
- `Fb2SplitHandle` - `.fb2` `.fb2.zip`: FictionBook XML; section titles map to
|
|
53
|
+
Markdown headings. Registered *before* `ZipSplitHandle` so `.fb2.zip` is not treated
|
|
54
|
+
as a generic archive.
|
|
55
|
+
- `MboxSplitHandle` - `.mbox`: per-message subject + body.
|
|
56
|
+
- `VcalendarSplitHandle` / `VcardSplitHandle` - `.ics` `.ifb` / `.vcf`: unfold
|
|
57
|
+
RFC 5545/6350 lines, keep only human-readable fields (event summary -> title;
|
|
58
|
+
time/location/description, or name/phone/email/address -> key/value rows).
|
|
59
|
+
- `ExtendedImageSplitHandle` - 25 more Pillow-decodable raster containers
|
|
60
|
+
(`.ico` `.tga` `.pcx` `.dds` `.sgi` `.ppm`/`.pgm`/`.pbm` `.im` `.icns` `.qoi`
|
|
61
|
+
`.jfif` `.apng` `.xbm` `.psd` + aliases). Subclasses `ImageSplitHandle`, so the OCR
|
|
62
|
+
branch and `save_image` contract are inherited unchanged.
|
|
63
|
+
- **OOXML variants that were declared but rejected now parse:** `.pptm` `.ppsm` `.potm`
|
|
64
|
+
(added to the presentation content-type map) and `.dotx` `.dotm` `.xltx` `.xltm`
|
|
65
|
+
(declared and claimed). `.ppsx`/`.potx` content types were also completed.
|
|
66
|
+
- **`.tar.xz` / `.txz`** (stdlib `tarfile` already supported xz), **`.azw1`/`.azw4`/
|
|
67
|
+
`.prc`** Kindle variants, **`.xhtml`/`.shtml`**, **`.tab`** (CSV), **`.xlt`** (binary
|
|
68
|
+
Excel template) are now recognised and declared.
|
|
69
|
+
- **`smart_slice._optional`** - single source of truth mapping each optional import to
|
|
70
|
+
its pip extra, powering both the `support()` gates and the "run pip install
|
|
71
|
+
smart-slice[office] to enable it" error text.
|
|
72
|
+
- Tests: `tests/test_format_expansion.py` (32 cases) covering every new handler, the
|
|
73
|
+
bare-environment degradation, the source-fidelity fix, and the OOXML variant
|
|
74
|
+
declarations. Suite is now 216 tests.
|
|
75
|
+
|
|
76
|
+
### Changed
|
|
77
|
+
|
|
78
|
+
- `HANDLER_EXTENSIONS` grew from 78 to 197 declared extensions across 30 handlers;
|
|
79
|
+
README / README_CN format tables were regenerated from the live registry and now
|
|
80
|
+
include a parser column and an explicit "not supported (400)" table.
|
|
81
|
+
- The zip/tar/7z **inner-file** dispatch list was extended with the new structured
|
|
82
|
+
text handlers (`Fb2`, `Svg`, `Ipynb`, `Subtitle`, `Mbox`, `Vcalendar`, `Vcard`), so
|
|
83
|
+
those formats are also recognised inside archives. Image handlers stay out of that
|
|
84
|
+
list on purpose: excluding `ImageSplitHandle` from archive members is a deliberate
|
|
85
|
+
Phase 2-B decision (embedded images travel through the zip handler's own markdown
|
|
86
|
+
reference collection + `save_image` path), pinned by
|
|
87
|
+
`test_zip_inner_images_do_not_receive_extractor`. Adding it back was attempted and
|
|
88
|
+
reverted; changing embedded-image semantics needs its own reviewed change.
|
|
89
|
+
|
|
90
|
+
## [0.2.0] - 2026-09-27
|
|
91
|
+
|
|
92
|
+
Adds configurable chunking (size **and** overlap) in both default and fully
|
|
93
|
+
user-controlled modes, plus a substantial performance pass on the slicing engine.
|
|
94
|
+
|
|
95
|
+
### Added
|
|
96
|
+
|
|
97
|
+
- **`ChunkingOptions`** (`smart_slice.options`) - one immutable object describing
|
|
98
|
+
chunk size and overlap behaviour: `limit`, `overlap`, `overlap_ratio`,
|
|
99
|
+
`boundary`, `lookback`, `overlap_boundary`, `min_chunk`, `carry_title`,
|
|
100
|
+
`overlap_within_section`, and `length_fn` (measure size in tokens instead of
|
|
101
|
+
characters by passing a tokenizer).
|
|
102
|
+
- **Paragraph overlap** - consecutive paragraphs can share trailing context.
|
|
103
|
+
Exposed at every entry point as `overlap=` / `overlap_ratio=` / `options=`,
|
|
104
|
+
with `overlap=0` (the default) preserving the previous exact-tiling behaviour.
|
|
105
|
+
- **`apply_paragraph_overlap`** - overlap is applied as a distinct pass *after*
|
|
106
|
+
the heading tree is assembled, never inside it (see Fixed for why). It is purely
|
|
107
|
+
additive: no paragraph's own text is altered, dropped or reordered.
|
|
108
|
+
- **Second-stage chunk overlap** - `chunk_paragraphs(..., chunk_overlap=,
|
|
109
|
+
chunk_overlap_ratio=)` and `OverlapChunkHandle`, so embedding chunks can also
|
|
110
|
+
carry context.
|
|
111
|
+
- **`carry_title`** - prefix *every* chunk with its paragraph's heading chain, so
|
|
112
|
+
section context survives chunking instead of appearing only in the first chunk.
|
|
113
|
+
- **`overlap_within_section`** - restrict overlap to paragraphs inside the same
|
|
114
|
+
heading chain.
|
|
115
|
+
- **CLI**: `slice --overlap N`, `--overlap-ratio F`, `--overlap-section-only`; a
|
|
116
|
+
new `chunk` subcommand (`--chunk-size`, `--chunk-overlap`, `--carry-title`).
|
|
117
|
+
- **Optional C accelerator** (`csrc/_speedup.c`, extra `accel`) - a single-pass
|
|
118
|
+
heading scan. Purely an optimisation: `smart_slice._speedup_py` provides an
|
|
119
|
+
identical pure-Python scan and is used whenever the extension is absent.
|
|
120
|
+
- **`scripts/build_ext.py`** - builds the extension in place, using `zig cc` on
|
|
121
|
+
Windows where setuptools ignores `CC` and insists on `cl.exe`.
|
|
122
|
+
- Tests: `tests/test_options_overlap.py` (45) and `tests/test_c_speedup.py` (19),
|
|
123
|
+
including the equivalence property that the heading prefilter never changes
|
|
124
|
+
slicing output.
|
|
125
|
+
|
|
126
|
+
### Changed
|
|
127
|
+
|
|
128
|
+
- **Performance: ~2.7x faster slicing** on a 550 KB structured corpus
|
|
129
|
+
(394 ms -> 144 ms; 1.4 -> 3.8 MB/s), from algorithmic fixes rather than the C
|
|
130
|
+
extension:
|
|
131
|
+
- `parse_title_level` no longer runs up to six whole-document regex scans per
|
|
132
|
+
recursion level. One linear scan establishes which heading levels are present
|
|
133
|
+
and provably-empty levels are skipped. Regex calls: 48,636 -> 11,604 per
|
|
134
|
+
document.
|
|
135
|
+
- `mask_code_blocks` returns the input unchanged when there is no code fence
|
|
136
|
+
(previously it always allocated a full character list and re-joined it).
|
|
137
|
+
Masking calls: 97,272 -> 11,604.
|
|
138
|
+
- Masked text is memoised across heading levels for the same block.
|
|
139
|
+
- Compiled patterns bypass `re.findall`'s dispatch, removing 194,544
|
|
140
|
+
`re._compile` cache lookups per document.
|
|
141
|
+
- `re_findall` flattens match groups in one pass instead of
|
|
142
|
+
`reduce([*x, *y], ...)`, which was O(n^2) in the number of matches.
|
|
143
|
+
- Total Python calls per document: 4.33M -> 1.55M (-64%).
|
|
144
|
+
- The C extension adds ~4% on top of that (139 vs 145 ms end-to-end), because the
|
|
145
|
+
algorithmic pass had already removed most of the scan cost. Measured, not
|
|
146
|
+
assumed - the extension is worth having for heading-dense documents but is not
|
|
147
|
+
where the win came from.
|
|
148
|
+
- `smart_slice/chunker.py` **graduated from generated to first-party source**:
|
|
149
|
+
chunking policy now lives there, so it is no longer regenerated from the source
|
|
150
|
+
platform. Documented in `docs/PORTING.md`; `CONTRIBUTING.md` explains which
|
|
151
|
+
files are generated and which are editable.
|
|
152
|
+
- Sentence boundaries now include the fullwidth `!` (U+FF01) and `?` (U+FF1F).
|
|
153
|
+
The previous implementation listed ASCII `!` and `?` twice, so CJK text ending in
|
|
154
|
+
a fullwidth mark was never treated as a sentence end (see Fixed).
|
|
155
|
+
|
|
156
|
+
### Fixed
|
|
157
|
+
|
|
158
|
+
- **Overlap inside the heading tree corrupted paragraph boundaries.** `parse_to_tree`
|
|
159
|
+
recovers block positions with `str.index()` over produced content, so chunks must
|
|
160
|
+
stay disjoint substrings of the source; overlapping them makes the lookup
|
|
161
|
+
ambiguous and silently reshuffles boundaries (symptom: identical total character
|
|
162
|
+
count, different paragraph split). Overlap is therefore applied after assembly.
|
|
163
|
+
- **Fullwidth `!`/`?` were not sentence boundaries.** The source `split_chars`
|
|
164
|
+
list contained ASCII `!` and `?` twice each, never U+FF01/U+FF1F, despite the
|
|
165
|
+
comment claiming "中英文感叹号/问号". Corrected, with a regression test.
|
|
166
|
+
- **`post_handler_paragraph` raised `NameError`.** It called `functools.reduce`,
|
|
167
|
+
which the module never imported. The function had no callers in the source
|
|
168
|
+
platform or in this package, so the defect stayed latent; the `reduce` flatten
|
|
169
|
+
(O(n^2)) is now a single pass and the two `if len(x) > 4096: pass` no-op
|
|
170
|
+
statements are gone.
|
|
171
|
+
- **`ChunkingOptions.with_(overlap_ratio=...)` was a no-op.** `__post_init__`
|
|
172
|
+
materialised `overlap=None` to `0`, which then outranked the ratio. `overlap`
|
|
173
|
+
keeps its `None` sentinel and resolution happens in `effective_overlap`.
|
|
174
|
+
- **`carry_title` only prefixed the first chunk of a paragraph.** It now prefixes
|
|
175
|
+
every chunk, which is the point of the option.
|
|
176
|
+
- **Importing the package mutated global Pillow state.** `_xlsx_images` set
|
|
177
|
+
`ImageFile.LOAD_TRUNCATED_IMAGES = True` and `Image.MAX_IMAGE_PIXELS = None` at
|
|
178
|
+
import time, silently disabling the *host process's* decompression-bomb guard.
|
|
179
|
+
Both are now scoped to `_permissive_pil()` around the image parse and restored
|
|
180
|
+
afterwards.
|
|
181
|
+
|
|
182
|
+
## [0.1.0] - 2026-09-26
|
|
183
|
+
|
|
184
|
+
Initial public release, extracted from an enterprise agent platform's document
|
|
185
|
+
ingestion layer. The parsing and slicing behaviour is carried over verbatim; the
|
|
186
|
+
web-framework coupling is removed.
|
|
187
|
+
|
|
188
|
+
### Added
|
|
189
|
+
|
|
190
|
+
- **Slicing engine** (`smart_slice.chunker`): heading-tree parsing, blank-line and
|
|
191
|
+
sentence-boundary splitting, a length budget (`limit`), heading-chain `title`
|
|
192
|
+
output, fidelity-preserving cleaning (line-start, code-fence-aware heading
|
|
193
|
+
marker removal), markdown table header re-insertion across cut boundaries, and
|
|
194
|
+
oversized-row re-splitting.
|
|
195
|
+
- **~30 format handlers** dispatched by a single ordered registry
|
|
196
|
+
(`SPLIT_HANDLERS`): HTML, MHTML, DOCX/DOCM, PDF, XLSX/XLSM, XLS, CSV/TSV, ZIP,
|
|
197
|
+
XMind, PPTX family, PPT, WPS/ET, RTF, ODF family, EPUB, EML, MSG, MOBI/AZW,
|
|
198
|
+
TAR family, 7z, images, and a text/source-code fallback.
|
|
199
|
+
- **Public API**: `slice_text`, `slice_bytes`, `slice_path`, `extract_text`,
|
|
200
|
+
`detect_handler`, `supported_extensions`, `missing_dependencies`,
|
|
201
|
+
`chunk_paragraphs`, `chunk`, plus the low-level `split_document`.
|
|
202
|
+
- **Command line interface** (`smart-slice` / `python -m smart_slice`) with
|
|
203
|
+
`slice`, `detect`, and `formats` subcommands and json/jsonl/text/md output.
|
|
204
|
+
- **QA and table parsers** (`smart_slice.qa`): question/answer extraction from
|
|
205
|
+
markdown, CSV and spreadsheets, and header-aware table row expansion.
|
|
206
|
+
- **Embedding chunking** (`smart_slice.chunking`): second-stage fixed-size
|
|
207
|
+
chunking of already-sliced paragraphs.
|
|
208
|
+
- **Image pipeline**: images extracted from documents are delivered through a
|
|
209
|
+
`save_image` callback as `ImageAsset` objects, with reference remapping for
|
|
210
|
+
deduplication and an optional local-OCR text-extraction hook.
|
|
211
|
+
- **Defensive parsing guards** (`ParserLimits`): byte, XML node/depth, archive
|
|
212
|
+
member, MIME part/depth and table-cell caps, configurable via
|
|
213
|
+
`SMART_SLICE_PARSER_*` environment variables.
|
|
214
|
+
- **Optional-dependency model**: a minimal core install plus per-format extras;
|
|
215
|
+
handlers degrade to "unsupported" rather than crashing when a parser is absent.
|
|
216
|
+
- **Test suite**: the source platform's slicing tests ported to run standalone
|
|
217
|
+
plus public-API and CLI tests.
|
|
218
|
+
|
|
219
|
+
### Notes
|
|
220
|
+
|
|
221
|
+
- Local image OCR is disabled by default; set `SMART_SLICE_OCR_ENABLED=1` and
|
|
222
|
+
install the `ocr` extra to enable it.
|
|
223
|
+
- Licensed under GPL-3.0, matching the upstream platform.
|
|
224
|
+
|
|
225
|
+
[0.2.0]: https://github.com/guangxiangdebizi/smart-slice/compare/v0.1.0...v0.2.0
|
|
226
|
+
[0.1.0]: https://github.com/guangxiangdebizi/smart-slice/releases/tag/v0.1.0
|