meltify 0.2.0__tar.gz → 0.2.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {meltify-0.2.0 → meltify-0.2.2}/PKG-INFO +63 -21
- meltify-0.2.2/README.md +182 -0
- {meltify-0.2.0 → meltify-0.2.2}/pyproject.toml +10 -2
- {meltify-0.2.0 → meltify-0.2.2}/pyproject.toml.orig +13 -3
- {meltify-0.2.0 → meltify-0.2.2}/src/meltify/commands/doctor.py +126 -16
- {meltify-0.2.0 → meltify-0.2.2}/src/meltify/commands/media.py +6 -37
- {meltify-0.2.0 → meltify-0.2.2}/src/meltify/commands/ocr.py +2 -2
- meltify-0.2.2/src/meltify/commands/read.py +711 -0
- {meltify-0.2.0 → meltify-0.2.2}/src/meltify/commands/submit.py +3 -2
- meltify-0.2.2/src/meltify/converters/__init__.py +424 -0
- meltify-0.2.2/src/meltify/converters/archive.py +665 -0
- meltify-0.2.2/src/meltify/converters/blips.py +187 -0
- meltify-0.2.2/src/meltify/converters/doc.py +342 -0
- meltify-0.2.2/src/meltify/converters/embeds.py +285 -0
- meltify-0.2.2/src/meltify/converters/epub.py +106 -0
- meltify-0.2.2/src/meltify/converters/fallback.py +304 -0
- meltify-0.2.2/src/meltify/converters/hwp.py +457 -0
- meltify-0.2.2/src/meltify/converters/iwa.py +249 -0
- meltify-0.2.2/src/meltify/converters/iwork.py +883 -0
- meltify-0.2.2/src/meltify/converters/limits.py +69 -0
- meltify-0.2.2/src/meltify/converters/metafile.py +530 -0
- meltify-0.2.2/src/meltify/converters/msg.py +147 -0
- meltify-0.2.2/src/meltify/converters/odf.py +316 -0
- meltify-0.2.2/src/meltify/converters/office.py +334 -0
- meltify-0.2.2/src/meltify/converters/ooxml.py +253 -0
- meltify-0.2.2/src/meltify/converters/ooxml_charts.py +334 -0
- meltify-0.2.2/src/meltify/converters/parquet.py +103 -0
- meltify-0.2.2/src/meltify/converters/pdf.py +192 -0
- meltify-0.2.2/src/meltify/converters/ppt.py +437 -0
- meltify-0.2.2/src/meltify/converters/quicklook.py +347 -0
- meltify-0.2.2/src/meltify/converters/raster.py +21 -0
- meltify-0.2.2/src/meltify/converters/render.py +243 -0
- meltify-0.2.2/src/meltify/converters/rtf.py +118 -0
- meltify-0.2.2/src/meltify/converters/run.py +131 -0
- meltify-0.2.2/src/meltify/converters/sheet.py +345 -0
- {meltify-0.2.0 → meltify-0.2.2}/src/meltify/converters/slack.py +2 -2
- {meltify-0.2.0 → meltify-0.2.2}/src/meltify/converters/sqlite.py +2 -11
- {meltify-0.2.0 → meltify-0.2.2}/src/meltify/converters/subtitle.py +33 -3
- {meltify-0.2.0 → meltify-0.2.2}/src/meltify/converters/svg.py +12 -5
- meltify-0.2.2/src/meltify/converters/tables.py +48 -0
- meltify-0.2.2/src/meltify/converters/text.py +170 -0
- meltify-0.2.2/src/meltify/converters/unlock.py +255 -0
- meltify-0.2.2/src/meltify/converters/web.py +292 -0
- meltify-0.2.2/src/meltify/converters/webarchive.py +160 -0
- meltify-0.2.2/src/meltify/converters/wordperfect.py +253 -0
- meltify-0.2.2/src/meltify/converters/xmlsafe.py +28 -0
- {meltify-0.2.0 → meltify-0.2.2}/src/meltify/data/defaults.toml +15 -0
- {meltify-0.2.0 → meltify-0.2.2}/src/meltify/evidence.py +8 -5
- {meltify-0.2.0 → meltify-0.2.2}/src/meltify/fetch.py +4 -0
- {meltify-0.2.0 → meltify-0.2.2}/src/meltify/files.py +40 -11
- meltify-0.2.2/src/meltify/forensics/pdf.py +200 -0
- {meltify-0.2.0 → meltify-0.2.2}/src/meltify/imaging.py +46 -0
- meltify-0.2.2/src/meltify/needs.py +20 -0
- meltify-0.2.2/src/meltify/passwords.py +32 -0
- {meltify-0.2.0 → meltify-0.2.2}/src/meltify/recognize.py +100 -28
- meltify-0.2.2/src/meltify/safe.py +121 -0
- meltify-0.2.2/src/meltify/tools.py +292 -0
- meltify-0.2.0/README.md +0 -148
- meltify-0.2.0/src/meltify/commands/read.py +0 -423
- meltify-0.2.0/src/meltify/converters/__init__.py +0 -277
- meltify-0.2.0/src/meltify/converters/archive.py +0 -383
- meltify-0.2.0/src/meltify/converters/hwp.py +0 -98
- meltify-0.2.0/src/meltify/converters/iwork.py +0 -90
- meltify-0.2.0/src/meltify/converters/legacy.py +0 -147
- meltify-0.2.0/src/meltify/converters/odf.py +0 -208
- meltify-0.2.0/src/meltify/converters/office.py +0 -183
- meltify-0.2.0/src/meltify/converters/pdf.py +0 -95
- meltify-0.2.0/src/meltify/converters/rtf.py +0 -17
- meltify-0.2.0/src/meltify/converters/sheet.py +0 -106
- meltify-0.2.0/src/meltify/converters/text.py +0 -79
- meltify-0.2.0/src/meltify/converters/web.py +0 -163
- meltify-0.2.0/src/meltify/forensics/pdf.py +0 -120
- meltify-0.2.0/src/meltify/safe.py +0 -56
- {meltify-0.2.0 → meltify-0.2.2}/LICENSE +0 -0
- {meltify-0.2.0 → meltify-0.2.2}/src/meltify/__init__.py +0 -0
- {meltify-0.2.0 → meltify-0.2.2}/src/meltify/__main__.py +0 -0
- {meltify-0.2.0 → meltify-0.2.2}/src/meltify/cli.py +0 -0
- {meltify-0.2.0 → meltify-0.2.2}/src/meltify/commands/__init__.py +0 -0
- {meltify-0.2.0 → meltify-0.2.2}/src/meltify/commands/brief.py +0 -0
- {meltify-0.2.0 → meltify-0.2.2}/src/meltify/commands/check.py +0 -0
- {meltify-0.2.0 → meltify-0.2.2}/src/meltify/commands/hidden.py +0 -0
- {meltify-0.2.0 → meltify-0.2.2}/src/meltify/config.py +0 -0
- {meltify-0.2.0 → meltify-0.2.2}/src/meltify/converters/kakao.py +0 -0
- {meltify-0.2.0 → meltify-0.2.2}/src/meltify/converters/mail.py +0 -0
- {meltify-0.2.0 → meltify-0.2.2}/src/meltify/converters/mbox.py +0 -0
- {meltify-0.2.0 → meltify-0.2.2}/src/meltify/converters/notebook.py +0 -0
- {meltify-0.2.0 → meltify-0.2.2}/src/meltify/data/__init__.py +0 -0
- {meltify-0.2.0 → meltify-0.2.2}/src/meltify/data/brief_en.md +0 -0
- {meltify-0.2.0 → meltify-0.2.2}/src/meltify/data/brief_ko.md +0 -0
- {meltify-0.2.0 → meltify-0.2.2}/src/meltify/engines/__init__.py +0 -0
- {meltify-0.2.0 → meltify-0.2.2}/src/meltify/engines/asr.py +0 -0
- {meltify-0.2.0 → meltify-0.2.2}/src/meltify/engines/consensus.py +0 -0
- {meltify-0.2.0 → meltify-0.2.2}/src/meltify/engines/ocr.py +0 -0
- {meltify-0.2.0 → meltify-0.2.2}/src/meltify/ffmpeg.py +0 -0
- {meltify-0.2.0 → meltify-0.2.2}/src/meltify/forensics/__init__.py +0 -0
- {meltify-0.2.0 → meltify-0.2.2}/src/meltify/lang.py +0 -0
- {meltify-0.2.0 → meltify-0.2.2}/src/meltify/llm.py +0 -0
- {meltify-0.2.0 → meltify-0.2.2}/src/meltify/output.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: meltify
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.2
|
|
4
4
|
Summary: Melt files of any format into prompt-ready text where every line cites its source
|
|
5
5
|
Keywords: llm,prompt,ocr,pdf,transcription,claude-code,agent-skills
|
|
6
6
|
Author: jeon-jihyeon
|
|
@@ -17,9 +17,12 @@ Requires-Dist: ocrmac>=1.0.1 ; sys_platform == 'darwin'
|
|
|
17
17
|
Requires-Dist: trafilatura>=2
|
|
18
18
|
Requires-Dist: python-hwpx>=1
|
|
19
19
|
Requires-Dist: striprtf>=0.0.29
|
|
20
|
-
Requires-Dist: meltify[office,media,asr-mlx,archive] ; extra == 'all'
|
|
20
|
+
Requires-Dist: meltify[office,media,asr-mlx,archive,parquet,crypto] ; extra == 'all'
|
|
21
21
|
Requires-Dist: libarchive-c>=5 ; extra == 'archive'
|
|
22
22
|
Requires-Dist: mlx-whisper>=0.4 ; platform_machine == 'arm64' and sys_platform == 'darwin' and extra == 'asr-mlx'
|
|
23
|
+
Requires-Dist: msoffcrypto-tool>=6 ; extra == 'crypto'
|
|
24
|
+
Requires-Dist: pyzipper>=0.4 ; extra == 'crypto'
|
|
25
|
+
Requires-Dist: cryptography>=42 ; extra == 'crypto'
|
|
23
26
|
Requires-Dist: numbers-parser>=4 ; extra == 'iwork'
|
|
24
27
|
Requires-Dist: yt-dlp>=2026.8.19 ; extra == 'media'
|
|
25
28
|
Requires-Dist: paddleocr>=3.2 ; extra == 'ocr-paddle'
|
|
@@ -27,16 +30,21 @@ Requires-Dist: paddlepaddle>=3.2 ; extra == 'ocr-paddle'
|
|
|
27
30
|
Requires-Dist: markitdown[docx,pptx,outlook]>=0.1.8 ; extra == 'office'
|
|
28
31
|
Requires-Dist: python-calamine>=0.4 ; extra == 'office'
|
|
29
32
|
Requires-Dist: legacy-doc>=0.2 ; extra == 'office'
|
|
33
|
+
Requires-Dist: olefile>=0.47 ; extra == 'office'
|
|
34
|
+
Requires-Dist: metafile-render>=0.3 ; extra == 'office'
|
|
35
|
+
Requires-Dist: pyarrow>=17 ; extra == 'parquet'
|
|
30
36
|
Requires-Dist: playwright>=1.50 ; extra == 'render'
|
|
31
37
|
Requires-Python: >=3.11.9
|
|
32
38
|
Project-URL: Repository, https://github.com/jeon-jihyeon/meltify
|
|
33
39
|
Provides-Extra: all
|
|
34
40
|
Provides-Extra: archive
|
|
35
41
|
Provides-Extra: asr-mlx
|
|
42
|
+
Provides-Extra: crypto
|
|
36
43
|
Provides-Extra: iwork
|
|
37
44
|
Provides-Extra: media
|
|
38
45
|
Provides-Extra: ocr-paddle
|
|
39
46
|
Provides-Extra: office
|
|
47
|
+
Provides-Extra: parquet
|
|
40
48
|
Provides-Extra: render
|
|
41
49
|
Description-Content-Type: text/markdown
|
|
42
50
|
|
|
@@ -93,25 +101,45 @@ Each command has a skill of the same name, such as `meltify-ocr`, that tells the
|
|
|
93
101
|
|
|
94
102
|
| Group | Formats | Needs |
|
|
95
103
|
|---|---|---|
|
|
96
|
-
| Documents | PDF, Word `.docx`
|
|
97
|
-
| Spreadsheets | Excel `.xlsx`
|
|
98
|
-
| Slides | PowerPoint `.pptx`
|
|
104
|
+
| Documents | PDF, Word `.docx` `.docm` `.dotx` `.dotm` `.doc` `.dot`, HWP and HWPX, RTF, ODT, Pages, WordPerfect, EPUB, XPS, FictionBook, MOBI, HTML and Safari `.webarchive`, Markdown and plain text | `office` for Word and EPUB |
|
|
105
|
+
| Spreadsheets | Excel `.xlsx` `.xlsm` `.xltx` `.xltm` `.xlsb` `.xls` `.xlt`, Hancom `.cell`, ODS, Numbers, Parquet, CSV and TSV | `office` for `.xlsb`, `.xls` and `.xlt`, `iwork` for Numbers, `parquet` for Parquet |
|
|
106
|
+
| Slides | PowerPoint `.pptx` `.pptm` `.potx` `.potm` `.ppsx` `.ppsm` `.ppt` `.pps` `.pot`, Hancom `.show`, ODP, Keynote | `office` for PowerPoint |
|
|
107
|
+
| Drawings | ODG, SVG and `.svgz`, EMF and WMF with `.emz` and `.wmz`, comic book `.cbz` | |
|
|
99
108
|
| Mail and chat | `.eml`, `.msg`, mbox, KakaoTalk exports, Slack export zips | `office` for `.msg` |
|
|
100
|
-
| Archives | zip, tar, tgz, tbz2, txz, 7z, rar | `archive` for 7z and rar |
|
|
101
|
-
| Images | PNG, JPEG, WebP, GIF, BMP, TIFF, HEIC, AVIF,
|
|
102
|
-
| Audio and video | mp3, wav, m4a, flac, ogg, mp4, mov, mkv, webm, subtitles | ffmpeg |
|
|
109
|
+
| Archives | zip, tar, tgz, tbz2, txz, 7z, rar, and single files compressed with gz, bz2 or xz | `archive` for 7z and rar, `crypto` for AES zips |
|
|
110
|
+
| Images | PNG, JPEG, WebP, GIF, BMP, TIFF, HEIC, AVIF, JPEG 2000, PSD, ICO, TGA | |
|
|
111
|
+
| Audio and video | mp3, wav, m4a, aac, flac, ogg, opus, wma, aiff, amr, mp4, m4v, mov, mkv, webm, avi, wmv, 3gp, mpg, flv, subtitles | ffmpeg |
|
|
103
112
|
| Web pages | any http(s) URL, video URLs included | `render` for JavaScript-heavy pages, `media` for video sites |
|
|
104
113
|
| Data | JSON, JSONL, XML, YAML, SQLite, Jupyter notebooks, vCard, iCalendar | |
|
|
105
114
|
|
|
106
|
-
Attachments, archive members and downloaded files are melted again as items of their own, up to three levels deep.
|
|
115
|
+
Attachments, archive members, files attached to a PDF and downloaded files are melted again as items of their own, up to three levels deep.
|
|
107
116
|
|
|
108
|
-
|
|
117
|
+
Templates, macro-enabled and slideshow files read like the document they belong to, and a document zip under an unfamiliar name is still recognized by its content.
|
|
118
|
+
|
|
119
|
+
A binary file no converter reads gets one more try, in this order:
|
|
120
|
+
|
|
121
|
+
- Formats LibreOffice imports, such as Works, Lotus 1-2-3, Visio and Publisher, are rendered to PDF and cited by page. These rows show kind `rendered`
|
|
122
|
+
- Any other picture format Pillow decodes, like PCX or ICNS, goes to OCR as kind `image`
|
|
123
|
+
- Audio or video under an odd name is transcribed as kind `media`
|
|
124
|
+
- On macOS, a full Quick Look preview (kind `rendered`), the text Spotlight indexes (kind `text`) or a first-page thumbnail (kind `rendered`) is the last resort
|
|
125
|
+
|
|
126
|
+
Compiled programs and files of one repeated byte skip all of that and return `unsupported format` right away. Word, Excel and PowerPoint binaries, HWP and iWork files whose own parser gives up take the same path, unless they're encrypted or DRM-locked. `--shallow` skips this step, and `read.fallback = false` turns it off.
|
|
127
|
+
|
|
128
|
+
Pictures get read in the same pass. Images, scanned pages, pictures inside PDF, Word, PowerPoint, Excel, ODF, RTF, EPUB, HTML and mail, and the speech and scene frames of recordings all go through local OCR and speech engines, and the text lands right where the picture was. Native text always comes first, and OCR only covers what parsing can't reach:
|
|
129
|
+
|
|
130
|
+
- Charts in Word, PowerPoint, Excel (`.xlsb` too) and ODF become cited tables from the values the file saved. A Word or PowerPoint chart with no cache reads the workbook embedded beside it, and an Excel chart on formulas nobody calculated asks LibreOffice to calculate them. SmartArt text is read as an indented list
|
|
131
|
+
- EMF and WMF pictures, standalone or inside a document, give up their text records directly. Bitmaps and text they can't decode are drawn by LibreOffice, Quick Look on macOS or the pure-Python `metafile-render` and then OCR'd. With none of them, they're listed in `needs`
|
|
132
|
+
- `.ppt` slides, notes and pictures, and `.doc` inline and floating pictures, are read natively. Pages and Keynote text, tables and pictures are read from the document itself, older iWork '09 files and bundle folders included
|
|
133
|
+
- A scan under a visible header still gets OCR'd, PDF sticky notes become cited blocks, and multi-page TIFFs and animated images are read frame by frame up to 50 frames
|
|
109
134
|
|
|
110
135
|
- Each unique picture or recording is read once, and every engine result is cached by content hash under `${XDG_CACHE_HOME:-~/.cache}/meltify`, so a rerun only pays for what changed
|
|
111
136
|
- `--budget SEC` caps the time spent on uncached work (600 seconds by default). Whatever's left shows up as `ocr (budget)` in `needs`, and the next run picks up where this one stopped
|
|
112
|
-
- `--
|
|
137
|
+
- `--jobs N` sets how many files convert at once (8 by default). PDFs convert in worker processes of their own
|
|
138
|
+
- `--shallow` skips OCR, speech, and drawing or recalculation work, and lists what it skipped in `needs`. Native text melts as usual
|
|
113
139
|
- When two OCR engines disagree on a number, the block ends with a `> disputed` line. With only one engine, it ends with `> unchecked: one engine`
|
|
114
140
|
|
|
141
|
+
Encrypted files open with a password from `--password-file PATH` or `MELTIFY_PASSWORD`: PDF, Word, Excel and PowerPoint, HWP and HWPX, Pages, Keynote and Numbers, and zip, 7z and rar archives. HWP distribution documents, which only restrict copying and printing, open without one. Without a password, the row says `encrypted` in `needs` and no markdown file is written, and a wrong one says `wrong password`. An archive still lists the members it could open. `--password` works too, but other users on the machine can see it. With more than one set, `--password-file` wins, then `--password`, then `MELTIFY_PASSWORD`.
|
|
142
|
+
|
|
115
143
|
URLs work like files. `read` saves a web page's main text with its heading anchors (`--whole` keeps the full page, `--render` runs it in a headless browser first), downloads documents and melts them by type, and sends video links through the `media` pipeline. Downloads land under `meltify-out/read/web/` and are revalidated with ETags on the next run. By default it refuses private addresses, follows `robots.txt` and stops at 20 MB or 30 seconds. See [SECURITY.md](SECURITY.md) for the details.
|
|
116
144
|
|
|
117
145
|
## Citations
|
|
@@ -125,7 +153,12 @@ Every result carries `src` and a one-line `cite`:
|
|
|
125
153
|
| Spreadsheet cell | `calendar.xlsx#Calendar!B7` |
|
|
126
154
|
| Mail attachment | `handover.eml#att=cal.xlsx#Calendar` |
|
|
127
155
|
| Archive member | `bundle.zip#att=docs/b.pdf#p3` |
|
|
156
|
+
| Word, ODT or Pages paragraph | `memo.docx:3` |
|
|
128
157
|
| Picture in a document | `memo.docx#para3#img1` |
|
|
158
|
+
| Slide | `deck.pptx#slide4` |
|
|
159
|
+
| Chart in a sheet | `book.xlsx#Sales!D2#img1` |
|
|
160
|
+
| Image frame | `fax.tif#frame2` |
|
|
161
|
+
| PDF sticky note | `report.pdf#p1@pt(300,300,316,316)` |
|
|
129
162
|
| HWP section line | `notice.hwp#s1:12` |
|
|
130
163
|
| Web page line | `https://example.com/guide#install:28` |
|
|
131
164
|
| Text line | `notes.txt:42` |
|
|
@@ -152,15 +185,17 @@ Optional parts install on request with `meltify doctor --install NAME`:
|
|
|
152
185
|
|
|
153
186
|
| Extra | Unlocks |
|
|
154
187
|
|---|---|
|
|
155
|
-
| `office` | Word, PowerPoint, Outlook `.msg
|
|
156
|
-
| `archive` | 7z and rar
|
|
157
|
-
| `iwork` | Numbers tables |
|
|
188
|
+
| `office` | Word, PowerPoint, Outlook `.msg` and EPUB, legacy `.doc`, `.ppt`, `.xls` and `.xlsb`, faster `.xlsx` reading through calamine, and drawing EMF pictures without LibreOffice |
|
|
189
|
+
| `archive` | 7z and rar through the system libarchive, plus a pinned 7-Zip download for encrypted ones |
|
|
190
|
+
| `iwork` | Numbers tables. Pages and Keynote need nothing extra |
|
|
191
|
+
| `parquet` | Parquet files, the first 200 rows and per-column min, max and null counts |
|
|
192
|
+
| `crypto` | encrypted Office files, AES zips, and password-protected HWP and iWork files |
|
|
158
193
|
| `render` | `read --render` for JavaScript-heavy pages, using your Chrome or a downloaded Chromium |
|
|
159
194
|
| `media` | video URLs through yt-dlp |
|
|
160
195
|
| `asr-mlx` | local speech recognition on Apple Silicon |
|
|
161
196
|
| `ocr-paddle` | PaddleOCR, the second local OCR engine |
|
|
162
197
|
|
|
163
|
-
Recordings need ffmpeg
|
|
198
|
+
Recordings need ffmpeg and YouTube downloads need deno. PowerPoint 95, formula recalculation and the rendering fallback need LibreOffice, which `meltify doctor --install libreoffice` downloads as a portable copy on macOS and Linux. `wpd2text` from libwpd reads WordPerfect best, then LibreOffice, and a built-in reader covers the body text without either.
|
|
164
199
|
|
|
165
200
|
## Configuration
|
|
166
201
|
|
|
@@ -172,7 +207,12 @@ Settings are layered, lowest precedence first:
|
|
|
172
207
|
4. `MELTIFY_<KEY>` and `MELTIFY_<SECTION>_<KEY>` environment variables
|
|
173
208
|
5. Command-line flags
|
|
174
209
|
|
|
175
|
-
See [examples/meltify.toml](examples/meltify.toml) for a sample.
|
|
210
|
+
See [examples/meltify.toml](examples/meltify.toml) for a sample. A few keys worth knowing:
|
|
211
|
+
|
|
212
|
+
- `lang` is the language OCR expects (`ko` by default)
|
|
213
|
+
- `asr.lang` is the spoken language for speech engines. It's `auto` by default, separate from `lang`, so Whisper detects each recording's language instead of translating it
|
|
214
|
+
- `read.parquet_rows` sets how many Parquet rows melt per file
|
|
215
|
+
- `render.quicklook = false` keeps Quick Look and Spotlight out of every render
|
|
176
216
|
|
|
177
217
|
## Limits
|
|
178
218
|
|
|
@@ -180,11 +220,13 @@ See [examples/meltify.toml](examples/meltify.toml) for a sample.
|
|
|
180
220
|
- OCR agreement means the engines read the same value, not that the value is right. A value read only by LLM engines is never marked agreed
|
|
181
221
|
- Speech recognition and OCR quality depend on the engine and the input
|
|
182
222
|
- Linux has no Apple Vision, so with PaddleOCR alone every reading is marked `unchecked`
|
|
183
|
-
-
|
|
184
|
-
-
|
|
185
|
-
-
|
|
186
|
-
-
|
|
223
|
+
- Rendered rows, EMF bitmaps and Quick Look previews are only as good as the OCR that reads them, and Quick Look exists only on macOS
|
|
224
|
+
- Pages and Keynote charts are listed in `needs` instead of read, and a Pages or Keynote file with no text records falls back to its embedded preview image
|
|
225
|
+
- DRM-protected HWP files, password-protected WordPerfect files, and Hancom `.cell` and `.show` files from before 2014 and Hanshow `.hpt` files, which are binary, aren't read. Save the Hancom ones again as `.xlsx` or `.pptx`
|
|
226
|
+
- Parquet files show their first 200 rows, and the rest are counted in `needs`
|
|
227
|
+
- Encrypted 7z and rar need 7-Zip, which `meltify doctor --install archive` downloads. Plain ones open with the system libarchive, a separate package on Linux, with 7-Zip, or with `bsdtar` on macOS
|
|
228
|
+
- The plugin launcher is a POSIX shell script and `doctor --install` only downloads macOS and Linux builds, so Windows isn't supported
|
|
187
229
|
|
|
188
230
|
## License
|
|
189
231
|
|
|
190
|
-
MIT. meltify depends on [PyMuPDF](https://github.com/pymupdf/PyMuPDF), which is AGPL-3.0, so distributing a product that bundles it comes with AGPL obligations. The `pi-heif` wheels bundle libheif and libde265 (LGPL-3.0), and the `iwork` extra pulls in `enum-tools` (LGPL-3.0) through `numbers-parser`.
|
|
232
|
+
MIT. meltify depends on [PyMuPDF](https://github.com/pymupdf/PyMuPDF), which is AGPL-3.0, so distributing a product that bundles it comes with AGPL obligations. The `pi-heif` wheels bundle libheif and libde265 (LGPL-3.0), and the `iwork` extra pulls in `enum-tools` (LGPL-3.0) through `numbers-parser`. The programs `doctor --install` downloads keep their own licenses: LibreOffice is MPL-2.0, and 7-Zip is LGPL-2.1 with the unRAR restriction.
|
meltify-0.2.2/README.md
ADDED
|
@@ -0,0 +1,182 @@
|
|
|
1
|
+
<h1 align="center">meltify</h1>
|
|
2
|
+
|
|
3
|
+
<p align="center">Melt files of any format into prompt-ready text where every line cites its source</p>
|
|
4
|
+
|
|
5
|
+
<p align="center">
|
|
6
|
+
<a href="https://github.com/jeon-jihyeon/meltify/actions/workflows/test.yml"><img alt="test" src="https://github.com/jeon-jihyeon/meltify/actions/workflows/test.yml/badge.svg"></a>
|
|
7
|
+
<a href="LICENSE"><img alt="MIT" src="https://img.shields.io/badge/license-MIT-blue"></a>
|
|
8
|
+
</p>
|
|
9
|
+
|
|
10
|
+
Agents misread blurry digits, miss text a PDF hides, can't watch video, and paraphrase the one condition that mattered. meltify turns documents, spreadsheets, slides, mail, chat exports, archives, web pages, images, video and audio into markdown an agent can quote, puts a citation on every block, and checks answers before they go out.
|
|
11
|
+
|
|
12
|
+
## Quickstart
|
|
13
|
+
|
|
14
|
+
Claude Code:
|
|
15
|
+
|
|
16
|
+
```
|
|
17
|
+
/plugin marketplace add jeon-jihyeon/meltify
|
|
18
|
+
/plugin install meltify@meltify
|
|
19
|
+
```
|
|
20
|
+
|
|
21
|
+
Codex, Gemini CLI, Cursor and other agents that read Agent Skills:
|
|
22
|
+
|
|
23
|
+
```
|
|
24
|
+
npx skills add jeon-jihyeon/meltify
|
|
25
|
+
```
|
|
26
|
+
|
|
27
|
+
Command line only:
|
|
28
|
+
|
|
29
|
+
```
|
|
30
|
+
uvx meltify read ./inputs
|
|
31
|
+
```
|
|
32
|
+
|
|
33
|
+
The skills call the `meltify` command. If it isn't installed, the plugin launcher runs it through `uvx`, so [uv](https://docs.astral.sh/uv/) is the only thing you need.
|
|
34
|
+
|
|
35
|
+
## What it does
|
|
36
|
+
|
|
37
|
+
| Command | Input | Output |
|
|
38
|
+
|---|---|---|
|
|
39
|
+
| `read` | files, folders and URLs of mixed formats | one cited markdown file per item, with images, scans and recordings read by local engines, plus a list of what still needs work |
|
|
40
|
+
| `ocr` | images and scanned PDF pages | each engine's lines with positions, and the values the engines disagree on |
|
|
41
|
+
| `hidden` | PDFs | text a reader doesn't see and why, with page and position |
|
|
42
|
+
| `media` | video, audio or a video URL | timestamped full-resolution frames and a timestamped transcript |
|
|
43
|
+
| `submit` | candidate files and a scoring endpoint | rate-limited submissions that skip duplicates and keep the best score |
|
|
44
|
+
| `check` | an answer file or a string | violations of a schema, counts, unique keys and text rules |
|
|
45
|
+
| `brief` | a problem statement | every condition line quoted with its line number, plus a board for juggling several problems |
|
|
46
|
+
| `doctor` | nothing | which engines, binaries and keys are available, and how to add the rest |
|
|
47
|
+
|
|
48
|
+
Each command has a skill of the same name, such as `meltify-ocr`, that tells the agent when to run it and what to do with the result.
|
|
49
|
+
|
|
50
|
+
## What `read` handles
|
|
51
|
+
|
|
52
|
+
| Group | Formats | Needs |
|
|
53
|
+
|---|---|---|
|
|
54
|
+
| Documents | PDF, Word `.docx` `.docm` `.dotx` `.dotm` `.doc` `.dot`, HWP and HWPX, RTF, ODT, Pages, WordPerfect, EPUB, XPS, FictionBook, MOBI, HTML and Safari `.webarchive`, Markdown and plain text | `office` for Word and EPUB |
|
|
55
|
+
| Spreadsheets | Excel `.xlsx` `.xlsm` `.xltx` `.xltm` `.xlsb` `.xls` `.xlt`, Hancom `.cell`, ODS, Numbers, Parquet, CSV and TSV | `office` for `.xlsb`, `.xls` and `.xlt`, `iwork` for Numbers, `parquet` for Parquet |
|
|
56
|
+
| Slides | PowerPoint `.pptx` `.pptm` `.potx` `.potm` `.ppsx` `.ppsm` `.ppt` `.pps` `.pot`, Hancom `.show`, ODP, Keynote | `office` for PowerPoint |
|
|
57
|
+
| Drawings | ODG, SVG and `.svgz`, EMF and WMF with `.emz` and `.wmz`, comic book `.cbz` | |
|
|
58
|
+
| Mail and chat | `.eml`, `.msg`, mbox, KakaoTalk exports, Slack export zips | `office` for `.msg` |
|
|
59
|
+
| Archives | zip, tar, tgz, tbz2, txz, 7z, rar, and single files compressed with gz, bz2 or xz | `archive` for 7z and rar, `crypto` for AES zips |
|
|
60
|
+
| Images | PNG, JPEG, WebP, GIF, BMP, TIFF, HEIC, AVIF, JPEG 2000, PSD, ICO, TGA | |
|
|
61
|
+
| Audio and video | mp3, wav, m4a, aac, flac, ogg, opus, wma, aiff, amr, mp4, m4v, mov, mkv, webm, avi, wmv, 3gp, mpg, flv, subtitles | ffmpeg |
|
|
62
|
+
| Web pages | any http(s) URL, video URLs included | `render` for JavaScript-heavy pages, `media` for video sites |
|
|
63
|
+
| Data | JSON, JSONL, XML, YAML, SQLite, Jupyter notebooks, vCard, iCalendar | |
|
|
64
|
+
|
|
65
|
+
Attachments, archive members, files attached to a PDF and downloaded files are melted again as items of their own, up to three levels deep.
|
|
66
|
+
|
|
67
|
+
Templates, macro-enabled and slideshow files read like the document they belong to, and a document zip under an unfamiliar name is still recognized by its content.
|
|
68
|
+
|
|
69
|
+
A binary file no converter reads gets one more try, in this order:
|
|
70
|
+
|
|
71
|
+
- Formats LibreOffice imports, such as Works, Lotus 1-2-3, Visio and Publisher, are rendered to PDF and cited by page. These rows show kind `rendered`
|
|
72
|
+
- Any other picture format Pillow decodes, like PCX or ICNS, goes to OCR as kind `image`
|
|
73
|
+
- Audio or video under an odd name is transcribed as kind `media`
|
|
74
|
+
- On macOS, a full Quick Look preview (kind `rendered`), the text Spotlight indexes (kind `text`) or a first-page thumbnail (kind `rendered`) is the last resort
|
|
75
|
+
|
|
76
|
+
Compiled programs and files of one repeated byte skip all of that and return `unsupported format` right away. Word, Excel and PowerPoint binaries, HWP and iWork files whose own parser gives up take the same path, unless they're encrypted or DRM-locked. `--shallow` skips this step, and `read.fallback = false` turns it off.
|
|
77
|
+
|
|
78
|
+
Pictures get read in the same pass. Images, scanned pages, pictures inside PDF, Word, PowerPoint, Excel, ODF, RTF, EPUB, HTML and mail, and the speech and scene frames of recordings all go through local OCR and speech engines, and the text lands right where the picture was. Native text always comes first, and OCR only covers what parsing can't reach:
|
|
79
|
+
|
|
80
|
+
- Charts in Word, PowerPoint, Excel (`.xlsb` too) and ODF become cited tables from the values the file saved. A Word or PowerPoint chart with no cache reads the workbook embedded beside it, and an Excel chart on formulas nobody calculated asks LibreOffice to calculate them. SmartArt text is read as an indented list
|
|
81
|
+
- EMF and WMF pictures, standalone or inside a document, give up their text records directly. Bitmaps and text they can't decode are drawn by LibreOffice, Quick Look on macOS or the pure-Python `metafile-render` and then OCR'd. With none of them, they're listed in `needs`
|
|
82
|
+
- `.ppt` slides, notes and pictures, and `.doc` inline and floating pictures, are read natively. Pages and Keynote text, tables and pictures are read from the document itself, older iWork '09 files and bundle folders included
|
|
83
|
+
- A scan under a visible header still gets OCR'd, PDF sticky notes become cited blocks, and multi-page TIFFs and animated images are read frame by frame up to 50 frames
|
|
84
|
+
|
|
85
|
+
- Each unique picture or recording is read once, and every engine result is cached by content hash under `${XDG_CACHE_HOME:-~/.cache}/meltify`, so a rerun only pays for what changed
|
|
86
|
+
- `--budget SEC` caps the time spent on uncached work (600 seconds by default). Whatever's left shows up as `ocr (budget)` in `needs`, and the next run picks up where this one stopped
|
|
87
|
+
- `--jobs N` sets how many files convert at once (8 by default). PDFs convert in worker processes of their own
|
|
88
|
+
- `--shallow` skips OCR, speech, and drawing or recalculation work, and lists what it skipped in `needs`. Native text melts as usual
|
|
89
|
+
- When two OCR engines disagree on a number, the block ends with a `> disputed` line. With only one engine, it ends with `> unchecked: one engine`
|
|
90
|
+
|
|
91
|
+
Encrypted files open with a password from `--password-file PATH` or `MELTIFY_PASSWORD`: PDF, Word, Excel and PowerPoint, HWP and HWPX, Pages, Keynote and Numbers, and zip, 7z and rar archives. HWP distribution documents, which only restrict copying and printing, open without one. Without a password, the row says `encrypted` in `needs` and no markdown file is written, and a wrong one says `wrong password`. An archive still lists the members it could open. `--password` works too, but other users on the machine can see it. With more than one set, `--password-file` wins, then `--password`, then `MELTIFY_PASSWORD`.
|
|
92
|
+
|
|
93
|
+
URLs work like files. `read` saves a web page's main text with its heading anchors (`--whole` keeps the full page, `--render` runs it in a headless browser first), downloads documents and melts them by type, and sends video links through the `media` pipeline. Downloads land under `meltify-out/read/web/` and are revalidated with ETags on the next run. By default it refuses private addresses, follows `robots.txt` and stops at 20 MB or 30 seconds. See [SECURITY.md](SECURITY.md) for the details.
|
|
94
|
+
|
|
95
|
+
## Citations
|
|
96
|
+
|
|
97
|
+
Every result carries `src` and a one-line `cite`:
|
|
98
|
+
|
|
99
|
+
| Input | Cite |
|
|
100
|
+
|---|---|
|
|
101
|
+
| PDF text | `report.pdf#p3@pt(72,95,140,103)` |
|
|
102
|
+
| Image region | `menu.png@px(120,40,380,72)` |
|
|
103
|
+
| Spreadsheet cell | `calendar.xlsx#Calendar!B7` |
|
|
104
|
+
| Mail attachment | `handover.eml#att=cal.xlsx#Calendar` |
|
|
105
|
+
| Archive member | `bundle.zip#att=docs/b.pdf#p3` |
|
|
106
|
+
| Word, ODT or Pages paragraph | `memo.docx:3` |
|
|
107
|
+
| Picture in a document | `memo.docx#para3#img1` |
|
|
108
|
+
| Slide | `deck.pptx#slide4` |
|
|
109
|
+
| Chart in a sheet | `book.xlsx#Sales!D2#img1` |
|
|
110
|
+
| Image frame | `fax.tif#frame2` |
|
|
111
|
+
| PDF sticky note | `report.pdf#p1@pt(300,300,316,316)` |
|
|
112
|
+
| HWP section line | `notice.hwp#s1:12` |
|
|
113
|
+
| Web page line | `https://example.com/guide#install:28` |
|
|
114
|
+
| Text line | `notes.txt:42` |
|
|
115
|
+
| Media span | `call.m4a@00:01:23.4-00:01:27.0` |
|
|
116
|
+
| JSON value | `answers.json#$[3].reason` |
|
|
117
|
+
|
|
118
|
+
Every command prints a table by default, and the full result with `--json`. Exit codes:
|
|
119
|
+
|
|
120
|
+
- `0`: success
|
|
121
|
+
- `1`: violations, failed checks or other errors
|
|
122
|
+
- `2`: usage error or bad input
|
|
123
|
+
- `3`: a needed engine, key, binary or base module is missing
|
|
124
|
+
|
|
125
|
+
## Engines
|
|
126
|
+
|
|
127
|
+
| Task | Local | Paid API |
|
|
128
|
+
|---|---|---|
|
|
129
|
+
| OCR | Apple Vision on macOS, PaddleOCR | Gemini, Claude, any OpenAI-compatible endpoint |
|
|
130
|
+
| Speech | MLX Whisper on Apple Silicon, whisper.cpp | any OpenAI-compatible transcription endpoint |
|
|
131
|
+
|
|
132
|
+
`auto` uses local engines only. Paid engines run only when you name them in `meltify.toml` or on the command line. An agent can also feed in what it reads in an image with `meltify ocr --reading agent=FILE`, so its own reading gets cross-checked against the local engines without any API key.
|
|
133
|
+
|
|
134
|
+
Optional parts install on request with `meltify doctor --install NAME`:
|
|
135
|
+
|
|
136
|
+
| Extra | Unlocks |
|
|
137
|
+
|---|---|
|
|
138
|
+
| `office` | Word, PowerPoint, Outlook `.msg` and EPUB, legacy `.doc`, `.ppt`, `.xls` and `.xlsb`, faster `.xlsx` reading through calamine, and drawing EMF pictures without LibreOffice |
|
|
139
|
+
| `archive` | 7z and rar through the system libarchive, plus a pinned 7-Zip download for encrypted ones |
|
|
140
|
+
| `iwork` | Numbers tables. Pages and Keynote need nothing extra |
|
|
141
|
+
| `parquet` | Parquet files, the first 200 rows and per-column min, max and null counts |
|
|
142
|
+
| `crypto` | encrypted Office files, AES zips, and password-protected HWP and iWork files |
|
|
143
|
+
| `render` | `read --render` for JavaScript-heavy pages, using your Chrome or a downloaded Chromium |
|
|
144
|
+
| `media` | video URLs through yt-dlp |
|
|
145
|
+
| `asr-mlx` | local speech recognition on Apple Silicon |
|
|
146
|
+
| `ocr-paddle` | PaddleOCR, the second local OCR engine |
|
|
147
|
+
|
|
148
|
+
Recordings need ffmpeg and YouTube downloads need deno. PowerPoint 95, formula recalculation and the rendering fallback need LibreOffice, which `meltify doctor --install libreoffice` downloads as a portable copy on macOS and Linux. `wpd2text` from libwpd reads WordPerfect best, then LibreOffice, and a built-in reader covers the body text without either.
|
|
149
|
+
|
|
150
|
+
## Configuration
|
|
151
|
+
|
|
152
|
+
Settings are layered, lowest precedence first:
|
|
153
|
+
|
|
154
|
+
1. The packaged defaults
|
|
155
|
+
2. `$XDG_CONFIG_HOME/meltify/config.toml` (usually `~/.config/meltify/config.toml`)
|
|
156
|
+
3. The nearest `meltify.toml`, searched from the working directory up to the git root, or the file you pass with `--config`. Outside a git repository, only the working directory is searched
|
|
157
|
+
4. `MELTIFY_<KEY>` and `MELTIFY_<SECTION>_<KEY>` environment variables
|
|
158
|
+
5. Command-line flags
|
|
159
|
+
|
|
160
|
+
See [examples/meltify.toml](examples/meltify.toml) for a sample. A few keys worth knowing:
|
|
161
|
+
|
|
162
|
+
- `lang` is the language OCR expects (`ko` by default)
|
|
163
|
+
- `asr.lang` is the spoken language for speech engines. It's `auto` by default, separate from `lang`, so Whisper detects each recording's language instead of translating it
|
|
164
|
+
- `read.parquet_rows` sets how many Parquet rows melt per file
|
|
165
|
+
- `render.quicklook = false` keeps Quick Look and Spotlight out of every render
|
|
166
|
+
|
|
167
|
+
## Limits
|
|
168
|
+
|
|
169
|
+
- Text drawn inside an image is just pixels, so `hidden` finds it only through `--contrast` rendering and OCR
|
|
170
|
+
- OCR agreement means the engines read the same value, not that the value is right. A value read only by LLM engines is never marked agreed
|
|
171
|
+
- Speech recognition and OCR quality depend on the engine and the input
|
|
172
|
+
- Linux has no Apple Vision, so with PaddleOCR alone every reading is marked `unchecked`
|
|
173
|
+
- Rendered rows, EMF bitmaps and Quick Look previews are only as good as the OCR that reads them, and Quick Look exists only on macOS
|
|
174
|
+
- Pages and Keynote charts are listed in `needs` instead of read, and a Pages or Keynote file with no text records falls back to its embedded preview image
|
|
175
|
+
- DRM-protected HWP files, password-protected WordPerfect files, and Hancom `.cell` and `.show` files from before 2014 and Hanshow `.hpt` files, which are binary, aren't read. Save the Hancom ones again as `.xlsx` or `.pptx`
|
|
176
|
+
- Parquet files show their first 200 rows, and the rest are counted in `needs`
|
|
177
|
+
- Encrypted 7z and rar need 7-Zip, which `meltify doctor --install archive` downloads. Plain ones open with the system libarchive, a separate package on Linux, with 7-Zip, or with `bsdtar` on macOS
|
|
178
|
+
- The plugin launcher is a POSIX shell script and `doctor --install` only downloads macOS and Linux builds, so Windows isn't supported
|
|
179
|
+
|
|
180
|
+
## License
|
|
181
|
+
|
|
182
|
+
MIT. meltify depends on [PyMuPDF](https://github.com/pymupdf/PyMuPDF), which is AGPL-3.0, so distributing a product that bundles it comes with AGPL obligations. The `pi-heif` wheels bundle libheif and libde265 (LGPL-3.0), and the `iwork` extra pulls in `enum-tools` (LGPL-3.0) through `numbers-parser`. The programs `doctor --install` downloads keep their own licenses: LibreOffice is MPL-2.0, and 7-Zip is LGPL-2.1 with the unRAR restriction.
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "meltify"
|
|
3
|
-
version = "0.2.
|
|
3
|
+
version = "0.2.2"
|
|
4
4
|
description = "Melt files of any format into prompt-ready text where every line cites its source"
|
|
5
5
|
readme = "README.md"
|
|
6
6
|
requires-python = ">=3.11.9"
|
|
@@ -37,9 +37,17 @@ office = [
|
|
|
37
37
|
"markitdown[docx,pptx,outlook]>=0.1.8",
|
|
38
38
|
"python-calamine>=0.4",
|
|
39
39
|
"legacy-doc>=0.2",
|
|
40
|
+
"olefile>=0.47",
|
|
41
|
+
"metafile-render>=0.3",
|
|
42
|
+
]
|
|
43
|
+
crypto = [
|
|
44
|
+
"msoffcrypto-tool>=6",
|
|
45
|
+
"pyzipper>=0.4",
|
|
46
|
+
"cryptography>=42",
|
|
40
47
|
]
|
|
41
48
|
archive = ["libarchive-c>=5"]
|
|
42
49
|
iwork = ["numbers-parser>=4"]
|
|
50
|
+
parquet = ["pyarrow>=17"]
|
|
43
51
|
render = ["playwright>=1.50"]
|
|
44
52
|
media = ["yt-dlp>=2026.8.19"]
|
|
45
53
|
asr-mlx = ["mlx-whisper>=0.4; sys_platform == 'darwin' and platform_machine == 'arm64'"]
|
|
@@ -47,7 +55,7 @@ ocr-paddle = [
|
|
|
47
55
|
"paddleocr>=3.2",
|
|
48
56
|
"paddlepaddle>=3.2",
|
|
49
57
|
]
|
|
50
|
-
all = ["meltify[office,media,asr-mlx,archive]"]
|
|
58
|
+
all = ["meltify[office,media,asr-mlx,archive,parquet,crypto]"]
|
|
51
59
|
|
|
52
60
|
[project.urls]
|
|
53
61
|
Repository = "https://github.com/jeon-jihyeon/meltify"
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "meltify"
|
|
3
|
-
version = "0.2.
|
|
3
|
+
version = "0.2.2"
|
|
4
4
|
description = "Melt files of any format into prompt-ready text where every line cites its source"
|
|
5
5
|
readme = "README.md"
|
|
6
6
|
requires-python = ">=3.11.9"
|
|
@@ -23,14 +23,24 @@ dependencies = [
|
|
|
23
23
|
]
|
|
24
24
|
|
|
25
25
|
[project.optional-dependencies]
|
|
26
|
-
office = [
|
|
26
|
+
office = [
|
|
27
|
+
"markitdown[docx,pptx,outlook]>=0.1.8",
|
|
28
|
+
"python-calamine>=0.4",
|
|
29
|
+
"legacy-doc>=0.2",
|
|
30
|
+
"olefile>=0.47",
|
|
31
|
+
"metafile-render>=0.3",
|
|
32
|
+
]
|
|
33
|
+
# Password-protected Office, zip, HWP and iWork files
|
|
34
|
+
crypto = ["msoffcrypto-tool>=6", "pyzipper>=0.4", "cryptography>=42"]
|
|
27
35
|
archive = ["libarchive-c>=5"]
|
|
36
|
+
# Only Numbers needs it, since Pages and Keynote are read natively
|
|
28
37
|
iwork = ["numbers-parser>=4"]
|
|
38
|
+
parquet = ["pyarrow>=17"]
|
|
29
39
|
render = ["playwright>=1.50"]
|
|
30
40
|
media = ["yt-dlp>=2026.8.19"]
|
|
31
41
|
asr-mlx = ["mlx-whisper>=0.4; sys_platform == 'darwin' and platform_machine == 'arm64'"]
|
|
32
42
|
ocr-paddle = ["paddleocr>=3.2", "paddlepaddle>=3.2"]
|
|
33
|
-
all = ["meltify[office,media,asr-mlx,archive]"]
|
|
43
|
+
all = ["meltify[office,media,asr-mlx,archive,parquet,crypto]"]
|
|
34
44
|
|
|
35
45
|
[project.urls]
|
|
36
46
|
Repository = "https://github.com/jeon-jihyeon/meltify"
|