meltify 0.2.0__tar.gz → 0.2.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (98) hide show
  1. {meltify-0.2.0 → meltify-0.2.2}/PKG-INFO +63 -21
  2. meltify-0.2.2/README.md +182 -0
  3. {meltify-0.2.0 → meltify-0.2.2}/pyproject.toml +10 -2
  4. {meltify-0.2.0 → meltify-0.2.2}/pyproject.toml.orig +13 -3
  5. {meltify-0.2.0 → meltify-0.2.2}/src/meltify/commands/doctor.py +126 -16
  6. {meltify-0.2.0 → meltify-0.2.2}/src/meltify/commands/media.py +6 -37
  7. {meltify-0.2.0 → meltify-0.2.2}/src/meltify/commands/ocr.py +2 -2
  8. meltify-0.2.2/src/meltify/commands/read.py +711 -0
  9. {meltify-0.2.0 → meltify-0.2.2}/src/meltify/commands/submit.py +3 -2
  10. meltify-0.2.2/src/meltify/converters/__init__.py +424 -0
  11. meltify-0.2.2/src/meltify/converters/archive.py +665 -0
  12. meltify-0.2.2/src/meltify/converters/blips.py +187 -0
  13. meltify-0.2.2/src/meltify/converters/doc.py +342 -0
  14. meltify-0.2.2/src/meltify/converters/embeds.py +285 -0
  15. meltify-0.2.2/src/meltify/converters/epub.py +106 -0
  16. meltify-0.2.2/src/meltify/converters/fallback.py +304 -0
  17. meltify-0.2.2/src/meltify/converters/hwp.py +457 -0
  18. meltify-0.2.2/src/meltify/converters/iwa.py +249 -0
  19. meltify-0.2.2/src/meltify/converters/iwork.py +883 -0
  20. meltify-0.2.2/src/meltify/converters/limits.py +69 -0
  21. meltify-0.2.2/src/meltify/converters/metafile.py +530 -0
  22. meltify-0.2.2/src/meltify/converters/msg.py +147 -0
  23. meltify-0.2.2/src/meltify/converters/odf.py +316 -0
  24. meltify-0.2.2/src/meltify/converters/office.py +334 -0
  25. meltify-0.2.2/src/meltify/converters/ooxml.py +253 -0
  26. meltify-0.2.2/src/meltify/converters/ooxml_charts.py +334 -0
  27. meltify-0.2.2/src/meltify/converters/parquet.py +103 -0
  28. meltify-0.2.2/src/meltify/converters/pdf.py +192 -0
  29. meltify-0.2.2/src/meltify/converters/ppt.py +437 -0
  30. meltify-0.2.2/src/meltify/converters/quicklook.py +347 -0
  31. meltify-0.2.2/src/meltify/converters/raster.py +21 -0
  32. meltify-0.2.2/src/meltify/converters/render.py +243 -0
  33. meltify-0.2.2/src/meltify/converters/rtf.py +118 -0
  34. meltify-0.2.2/src/meltify/converters/run.py +131 -0
  35. meltify-0.2.2/src/meltify/converters/sheet.py +345 -0
  36. {meltify-0.2.0 → meltify-0.2.2}/src/meltify/converters/slack.py +2 -2
  37. {meltify-0.2.0 → meltify-0.2.2}/src/meltify/converters/sqlite.py +2 -11
  38. {meltify-0.2.0 → meltify-0.2.2}/src/meltify/converters/subtitle.py +33 -3
  39. {meltify-0.2.0 → meltify-0.2.2}/src/meltify/converters/svg.py +12 -5
  40. meltify-0.2.2/src/meltify/converters/tables.py +48 -0
  41. meltify-0.2.2/src/meltify/converters/text.py +170 -0
  42. meltify-0.2.2/src/meltify/converters/unlock.py +255 -0
  43. meltify-0.2.2/src/meltify/converters/web.py +292 -0
  44. meltify-0.2.2/src/meltify/converters/webarchive.py +160 -0
  45. meltify-0.2.2/src/meltify/converters/wordperfect.py +253 -0
  46. meltify-0.2.2/src/meltify/converters/xmlsafe.py +28 -0
  47. {meltify-0.2.0 → meltify-0.2.2}/src/meltify/data/defaults.toml +15 -0
  48. {meltify-0.2.0 → meltify-0.2.2}/src/meltify/evidence.py +8 -5
  49. {meltify-0.2.0 → meltify-0.2.2}/src/meltify/fetch.py +4 -0
  50. {meltify-0.2.0 → meltify-0.2.2}/src/meltify/files.py +40 -11
  51. meltify-0.2.2/src/meltify/forensics/pdf.py +200 -0
  52. {meltify-0.2.0 → meltify-0.2.2}/src/meltify/imaging.py +46 -0
  53. meltify-0.2.2/src/meltify/needs.py +20 -0
  54. meltify-0.2.2/src/meltify/passwords.py +32 -0
  55. {meltify-0.2.0 → meltify-0.2.2}/src/meltify/recognize.py +100 -28
  56. meltify-0.2.2/src/meltify/safe.py +121 -0
  57. meltify-0.2.2/src/meltify/tools.py +292 -0
  58. meltify-0.2.0/README.md +0 -148
  59. meltify-0.2.0/src/meltify/commands/read.py +0 -423
  60. meltify-0.2.0/src/meltify/converters/__init__.py +0 -277
  61. meltify-0.2.0/src/meltify/converters/archive.py +0 -383
  62. meltify-0.2.0/src/meltify/converters/hwp.py +0 -98
  63. meltify-0.2.0/src/meltify/converters/iwork.py +0 -90
  64. meltify-0.2.0/src/meltify/converters/legacy.py +0 -147
  65. meltify-0.2.0/src/meltify/converters/odf.py +0 -208
  66. meltify-0.2.0/src/meltify/converters/office.py +0 -183
  67. meltify-0.2.0/src/meltify/converters/pdf.py +0 -95
  68. meltify-0.2.0/src/meltify/converters/rtf.py +0 -17
  69. meltify-0.2.0/src/meltify/converters/sheet.py +0 -106
  70. meltify-0.2.0/src/meltify/converters/text.py +0 -79
  71. meltify-0.2.0/src/meltify/converters/web.py +0 -163
  72. meltify-0.2.0/src/meltify/forensics/pdf.py +0 -120
  73. meltify-0.2.0/src/meltify/safe.py +0 -56
  74. {meltify-0.2.0 → meltify-0.2.2}/LICENSE +0 -0
  75. {meltify-0.2.0 → meltify-0.2.2}/src/meltify/__init__.py +0 -0
  76. {meltify-0.2.0 → meltify-0.2.2}/src/meltify/__main__.py +0 -0
  77. {meltify-0.2.0 → meltify-0.2.2}/src/meltify/cli.py +0 -0
  78. {meltify-0.2.0 → meltify-0.2.2}/src/meltify/commands/__init__.py +0 -0
  79. {meltify-0.2.0 → meltify-0.2.2}/src/meltify/commands/brief.py +0 -0
  80. {meltify-0.2.0 → meltify-0.2.2}/src/meltify/commands/check.py +0 -0
  81. {meltify-0.2.0 → meltify-0.2.2}/src/meltify/commands/hidden.py +0 -0
  82. {meltify-0.2.0 → meltify-0.2.2}/src/meltify/config.py +0 -0
  83. {meltify-0.2.0 → meltify-0.2.2}/src/meltify/converters/kakao.py +0 -0
  84. {meltify-0.2.0 → meltify-0.2.2}/src/meltify/converters/mail.py +0 -0
  85. {meltify-0.2.0 → meltify-0.2.2}/src/meltify/converters/mbox.py +0 -0
  86. {meltify-0.2.0 → meltify-0.2.2}/src/meltify/converters/notebook.py +0 -0
  87. {meltify-0.2.0 → meltify-0.2.2}/src/meltify/data/__init__.py +0 -0
  88. {meltify-0.2.0 → meltify-0.2.2}/src/meltify/data/brief_en.md +0 -0
  89. {meltify-0.2.0 → meltify-0.2.2}/src/meltify/data/brief_ko.md +0 -0
  90. {meltify-0.2.0 → meltify-0.2.2}/src/meltify/engines/__init__.py +0 -0
  91. {meltify-0.2.0 → meltify-0.2.2}/src/meltify/engines/asr.py +0 -0
  92. {meltify-0.2.0 → meltify-0.2.2}/src/meltify/engines/consensus.py +0 -0
  93. {meltify-0.2.0 → meltify-0.2.2}/src/meltify/engines/ocr.py +0 -0
  94. {meltify-0.2.0 → meltify-0.2.2}/src/meltify/ffmpeg.py +0 -0
  95. {meltify-0.2.0 → meltify-0.2.2}/src/meltify/forensics/__init__.py +0 -0
  96. {meltify-0.2.0 → meltify-0.2.2}/src/meltify/lang.py +0 -0
  97. {meltify-0.2.0 → meltify-0.2.2}/src/meltify/llm.py +0 -0
  98. {meltify-0.2.0 → meltify-0.2.2}/src/meltify/output.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: meltify
3
- Version: 0.2.0
3
+ Version: 0.2.2
4
4
  Summary: Melt files of any format into prompt-ready text where every line cites its source
5
5
  Keywords: llm,prompt,ocr,pdf,transcription,claude-code,agent-skills
6
6
  Author: jeon-jihyeon
@@ -17,9 +17,12 @@ Requires-Dist: ocrmac>=1.0.1 ; sys_platform == 'darwin'
17
17
  Requires-Dist: trafilatura>=2
18
18
  Requires-Dist: python-hwpx>=1
19
19
  Requires-Dist: striprtf>=0.0.29
20
- Requires-Dist: meltify[office,media,asr-mlx,archive] ; extra == 'all'
20
+ Requires-Dist: meltify[office,media,asr-mlx,archive,parquet,crypto] ; extra == 'all'
21
21
  Requires-Dist: libarchive-c>=5 ; extra == 'archive'
22
22
  Requires-Dist: mlx-whisper>=0.4 ; platform_machine == 'arm64' and sys_platform == 'darwin' and extra == 'asr-mlx'
23
+ Requires-Dist: msoffcrypto-tool>=6 ; extra == 'crypto'
24
+ Requires-Dist: pyzipper>=0.4 ; extra == 'crypto'
25
+ Requires-Dist: cryptography>=42 ; extra == 'crypto'
23
26
  Requires-Dist: numbers-parser>=4 ; extra == 'iwork'
24
27
  Requires-Dist: yt-dlp>=2026.8.19 ; extra == 'media'
25
28
  Requires-Dist: paddleocr>=3.2 ; extra == 'ocr-paddle'
@@ -27,16 +30,21 @@ Requires-Dist: paddlepaddle>=3.2 ; extra == 'ocr-paddle'
27
30
  Requires-Dist: markitdown[docx,pptx,outlook]>=0.1.8 ; extra == 'office'
28
31
  Requires-Dist: python-calamine>=0.4 ; extra == 'office'
29
32
  Requires-Dist: legacy-doc>=0.2 ; extra == 'office'
33
+ Requires-Dist: olefile>=0.47 ; extra == 'office'
34
+ Requires-Dist: metafile-render>=0.3 ; extra == 'office'
35
+ Requires-Dist: pyarrow>=17 ; extra == 'parquet'
30
36
  Requires-Dist: playwright>=1.50 ; extra == 'render'
31
37
  Requires-Python: >=3.11.9
32
38
  Project-URL: Repository, https://github.com/jeon-jihyeon/meltify
33
39
  Provides-Extra: all
34
40
  Provides-Extra: archive
35
41
  Provides-Extra: asr-mlx
42
+ Provides-Extra: crypto
36
43
  Provides-Extra: iwork
37
44
  Provides-Extra: media
38
45
  Provides-Extra: ocr-paddle
39
46
  Provides-Extra: office
47
+ Provides-Extra: parquet
40
48
  Provides-Extra: render
41
49
  Description-Content-Type: text/markdown
42
50
 
@@ -93,25 +101,45 @@ Each command has a skill of the same name, such as `meltify-ocr`, that tells the
93
101
 
94
102
  | Group | Formats | Needs |
95
103
  |---|---|---|
96
- | Documents | PDF, Word `.docx` and `.doc`, HWP and HWPX, RTF, ODT, Pages, EPUB, HTML, Markdown and plain text | `office` for `.docx`, `.doc`, EPUB and HTML files |
97
- | Spreadsheets | Excel `.xlsx` and `.xls`, ODS, Numbers, CSV and TSV | `office` for `.xls`, `iwork` for Numbers |
98
- | Slides | PowerPoint `.pptx` and `.ppt`, ODP, Keynote | `office` for `.pptx`, LibreOffice for `.ppt` |
104
+ | Documents | PDF, Word `.docx` `.docm` `.dotx` `.dotm` `.doc` `.dot`, HWP and HWPX, RTF, ODT, Pages, WordPerfect, EPUB, XPS, FictionBook, MOBI, HTML and Safari `.webarchive`, Markdown and plain text | `office` for Word and EPUB |
105
+ | Spreadsheets | Excel `.xlsx` `.xlsm` `.xltx` `.xltm` `.xlsb` `.xls` `.xlt`, Hancom `.cell`, ODS, Numbers, Parquet, CSV and TSV | `office` for `.xlsb`, `.xls` and `.xlt`, `iwork` for Numbers, `parquet` for Parquet |
106
+ | Slides | PowerPoint `.pptx` `.pptm` `.potx` `.potm` `.ppsx` `.ppsm` `.ppt` `.pps` `.pot`, Hancom `.show`, ODP, Keynote | `office` for PowerPoint |
107
+ | Drawings | ODG, SVG and `.svgz`, EMF and WMF with `.emz` and `.wmz`, comic book `.cbz` | |
99
108
  | Mail and chat | `.eml`, `.msg`, mbox, KakaoTalk exports, Slack export zips | `office` for `.msg` |
100
- | Archives | zip, tar, tgz, tbz2, txz, 7z, rar | `archive` for 7z and rar |
101
- | Images | PNG, JPEG, WebP, GIF, BMP, TIFF, HEIC, AVIF, SVG | |
102
- | Audio and video | mp3, wav, m4a, flac, ogg, mp4, mov, mkv, webm, subtitles | ffmpeg |
109
+ | Archives | zip, tar, tgz, tbz2, txz, 7z, rar, and single files compressed with gz, bz2 or xz | `archive` for 7z and rar, `crypto` for AES zips |
110
+ | Images | PNG, JPEG, WebP, GIF, BMP, TIFF, HEIC, AVIF, JPEG 2000, PSD, ICO, TGA | |
111
+ | Audio and video | mp3, wav, m4a, aac, flac, ogg, opus, wma, aiff, amr, mp4, m4v, mov, mkv, webm, avi, wmv, 3gp, mpg, flv, subtitles | ffmpeg |
103
112
  | Web pages | any http(s) URL, video URLs included | `render` for JavaScript-heavy pages, `media` for video sites |
104
113
  | Data | JSON, JSONL, XML, YAML, SQLite, Jupyter notebooks, vCard, iCalendar | |
105
114
 
106
- Attachments, archive members and downloaded files are melted again as items of their own, up to three levels deep.
115
+ Attachments, archive members, files attached to a PDF and downloaded files are melted again as items of their own, up to three levels deep.
107
116
 
108
- Pictures get read in the same pass. Images, scanned pages, pictures inside PDF, Word, PowerPoint, Excel and mail, and the speech and scene frames of recordings all go through local OCR and speech engines, and the text lands right where the picture was:
117
+ Templates, macro-enabled and slideshow files read like the document they belong to, and a document zip under an unfamiliar name is still recognized by its content.
118
+
119
+ A binary file no converter reads gets one more try, in this order:
120
+
121
+ - Formats LibreOffice imports, such as Works, Lotus 1-2-3, Visio and Publisher, are rendered to PDF and cited by page. These rows show kind `rendered`
122
+ - Any other picture format Pillow decodes, like PCX or ICNS, goes to OCR as kind `image`
123
+ - Audio or video under an odd name is transcribed as kind `media`
124
+ - On macOS, a full Quick Look preview (kind `rendered`), the text Spotlight indexes (kind `text`) or a first-page thumbnail (kind `rendered`) is the last resort
125
+
126
+ Compiled programs and files of one repeated byte skip all of that and return `unsupported format` right away. Word, Excel and PowerPoint binaries, HWP and iWork files whose own parser gives up take the same path, unless they're encrypted or DRM-locked. `--shallow` skips this step, and `read.fallback = false` turns it off.
127
+
128
+ Pictures get read in the same pass. Images, scanned pages, pictures inside PDF, Word, PowerPoint, Excel, ODF, RTF, EPUB, HTML and mail, and the speech and scene frames of recordings all go through local OCR and speech engines, and the text lands right where the picture was. Native text always comes first, and OCR only covers what parsing can't reach:
129
+
130
+ - Charts in Word, PowerPoint, Excel (`.xlsb` too) and ODF become cited tables from the values the file saved. A Word or PowerPoint chart with no cache reads the workbook embedded beside it, and an Excel chart on formulas nobody calculated asks LibreOffice to calculate them. SmartArt text is read as an indented list
131
+ - EMF and WMF pictures, standalone or inside a document, give up their text records directly. Bitmaps and text they can't decode are drawn by LibreOffice, Quick Look on macOS or the pure-Python `metafile-render` and then OCR'd. With none of them, they're listed in `needs`
132
+ - `.ppt` slides, notes and pictures, and `.doc` inline and floating pictures, are read natively. Pages and Keynote text, tables and pictures are read from the document itself, older iWork '09 files and bundle folders included
133
+ - A scan under a visible header still gets OCR'd, PDF sticky notes become cited blocks, and multi-page TIFFs and animated images are read frame by frame up to 50 frames
109
134
 
110
135
  - Each unique picture or recording is read once, and every engine result is cached by content hash under `${XDG_CACHE_HOME:-~/.cache}/meltify`, so a rerun only pays for what changed
111
136
  - `--budget SEC` caps the time spent on uncached work (600 seconds by default). Whatever's left shows up as `ocr (budget)` in `needs`, and the next run picks up where this one stopped
112
- - `--shallow` skips OCR and speech recognition and just lists what needs them. Everything else melts as usual
137
+ - `--jobs N` sets how many files convert at once (8 by default). PDFs convert in worker processes of their own
138
+ - `--shallow` skips OCR, speech, and drawing or recalculation work, and lists what it skipped in `needs`. Native text melts as usual
113
139
  - When two OCR engines disagree on a number, the block ends with a `> disputed` line. With only one engine, it ends with `> unchecked: one engine`
114
140
 
141
+ Encrypted files open with a password from `--password-file PATH` or `MELTIFY_PASSWORD`: PDF, Word, Excel and PowerPoint, HWP and HWPX, Pages, Keynote and Numbers, and zip, 7z and rar archives. HWP distribution documents, which only restrict copying and printing, open without one. Without a password, the row says `encrypted` in `needs` and no markdown file is written, and a wrong one says `wrong password`. An archive still lists the members it could open. `--password` works too, but other users on the machine can see it. With more than one set, `--password-file` wins, then `--password`, then `MELTIFY_PASSWORD`.
142
+
115
143
  URLs work like files. `read` saves a web page's main text with its heading anchors (`--whole` keeps the full page, `--render` runs it in a headless browser first), downloads documents and melts them by type, and sends video links through the `media` pipeline. Downloads land under `meltify-out/read/web/` and are revalidated with ETags on the next run. By default it refuses private addresses, follows `robots.txt` and stops at 20 MB or 30 seconds. See [SECURITY.md](SECURITY.md) for the details.
116
144
 
117
145
  ## Citations
@@ -125,7 +153,12 @@ Every result carries `src` and a one-line `cite`:
125
153
  | Spreadsheet cell | `calendar.xlsx#Calendar!B7` |
126
154
  | Mail attachment | `handover.eml#att=cal.xlsx#Calendar` |
127
155
  | Archive member | `bundle.zip#att=docs/b.pdf#p3` |
156
+ | Word, ODT or Pages paragraph | `memo.docx:3` |
128
157
  | Picture in a document | `memo.docx#para3#img1` |
158
+ | Slide | `deck.pptx#slide4` |
159
+ | Chart in a sheet | `book.xlsx#Sales!D2#img1` |
160
+ | Image frame | `fax.tif#frame2` |
161
+ | PDF sticky note | `report.pdf#p1@pt(300,300,316,316)` |
129
162
  | HWP section line | `notice.hwp#s1:12` |
130
163
  | Web page line | `https://example.com/guide#install:28` |
131
164
  | Text line | `notes.txt:42` |
@@ -152,15 +185,17 @@ Optional parts install on request with `meltify doctor --install NAME`:
152
185
 
153
186
  | Extra | Unlocks |
154
187
  |---|---|
155
- | `office` | Word, PowerPoint, Outlook `.msg`, EPUB and HTML files, plus legacy `.doc` and `.xls` |
156
- | `archive` | 7z and rar, through the system libarchive |
157
- | `iwork` | Numbers tables |
188
+ | `office` | Word, PowerPoint, Outlook `.msg` and EPUB, legacy `.doc`, `.ppt`, `.xls` and `.xlsb`, faster `.xlsx` reading through calamine, and drawing EMF pictures without LibreOffice |
189
+ | `archive` | 7z and rar through the system libarchive, plus a pinned 7-Zip download for encrypted ones |
190
+ | `iwork` | Numbers tables. Pages and Keynote need nothing extra |
191
+ | `parquet` | Parquet files, the first 200 rows and per-column min, max and null counts |
192
+ | `crypto` | encrypted Office files, AES zips, and password-protected HWP and iWork files |
158
193
  | `render` | `read --render` for JavaScript-heavy pages, using your Chrome or a downloaded Chromium |
159
194
  | `media` | video URLs through yt-dlp |
160
195
  | `asr-mlx` | local speech recognition on Apple Silicon |
161
196
  | `ocr-paddle` | PaddleOCR, the second local OCR engine |
162
197
 
163
- Recordings need ffmpeg, YouTube downloads need deno, and `.ppt` needs LibreOffice.
198
+ Recordings need ffmpeg and YouTube downloads need deno. PowerPoint 95, formula recalculation and the rendering fallback need LibreOffice, which `meltify doctor --install libreoffice` downloads as a portable copy on macOS and Linux. `wpd2text` from libwpd reads WordPerfect best, then LibreOffice, and a built-in reader covers the body text without either.
164
199
 
165
200
  ## Configuration
166
201
 
@@ -172,7 +207,12 @@ Settings are layered, lowest precedence first:
172
207
  4. `MELTIFY_<KEY>` and `MELTIFY_<SECTION>_<KEY>` environment variables
173
208
  5. Command-line flags
174
209
 
175
- See [examples/meltify.toml](examples/meltify.toml) for a sample.
210
+ See [examples/meltify.toml](examples/meltify.toml) for a sample. A few keys worth knowing:
211
+
212
+ - `lang` is the language OCR expects (`ko` by default)
213
+ - `asr.lang` is the spoken language for speech engines. It's `auto` by default, separate from `lang`, so Whisper detects each recording's language instead of translating it
214
+ - `read.parquet_rows` sets how many Parquet rows melt per file
215
+ - `render.quicklook = false` keeps Quick Look and Spotlight out of every render
176
216
 
177
217
  ## Limits
178
218
 
@@ -180,11 +220,13 @@ See [examples/meltify.toml](examples/meltify.toml) for a sample.
180
220
  - OCR agreement means the engines read the same value, not that the value is right. A value read only by LLM engines is never marked agreed
181
221
  - Speech recognition and OCR quality depend on the engine and the input
182
222
  - Linux has no Apple Vision, so with PaddleOCR alone every reading is marked `unchecked`
183
- - `.ppt` slides need [LibreOffice](https://www.libreoffice.org/). Without it, the row says so in `needs`
184
- - 7z and rar need the system libarchive, a separate package on Linux
185
- - Pages and Keynote files are read only through the preview image they embed, and encrypted HWP and Numbers files aren't read at all
186
- - The plugin launcher is a POSIX shell script, so Windows isn't supported
223
+ - Rendered rows, EMF bitmaps and Quick Look previews are only as good as the OCR that reads them, and Quick Look exists only on macOS
224
+ - Pages and Keynote charts are listed in `needs` instead of read, and a Pages or Keynote file with no text records falls back to its embedded preview image
225
+ - DRM-protected HWP files, password-protected WordPerfect files, and Hancom `.cell` and `.show` files from before 2014 and Hanshow `.hpt` files, which are binary, aren't read. Save the Hancom ones again as `.xlsx` or `.pptx`
226
+ - Parquet files show their first 200 rows, and the rest are counted in `needs`
227
+ - Encrypted 7z and rar need 7-Zip, which `meltify doctor --install archive` downloads. Plain ones open with the system libarchive, a separate package on Linux, with 7-Zip, or with `bsdtar` on macOS
228
+ - The plugin launcher is a POSIX shell script and `doctor --install` only downloads macOS and Linux builds, so Windows isn't supported
187
229
 
188
230
  ## License
189
231
 
190
- MIT. meltify depends on [PyMuPDF](https://github.com/pymupdf/PyMuPDF), which is AGPL-3.0, so distributing a product that bundles it comes with AGPL obligations. The `pi-heif` wheels bundle libheif and libde265 (LGPL-3.0), and the `iwork` extra pulls in `enum-tools` (LGPL-3.0) through `numbers-parser`.
232
+ MIT. meltify depends on [PyMuPDF](https://github.com/pymupdf/PyMuPDF), which is AGPL-3.0, so distributing a product that bundles it comes with AGPL obligations. The `pi-heif` wheels bundle libheif and libde265 (LGPL-3.0), and the `iwork` extra pulls in `enum-tools` (LGPL-3.0) through `numbers-parser`. The programs `doctor --install` downloads keep their own licenses: LibreOffice is MPL-2.0, and 7-Zip is LGPL-2.1 with the unRAR restriction.
@@ -0,0 +1,182 @@
1
+ <h1 align="center">meltify</h1>
2
+
3
+ <p align="center">Melt files of any format into prompt-ready text where every line cites its source</p>
4
+
5
+ <p align="center">
6
+ <a href="https://github.com/jeon-jihyeon/meltify/actions/workflows/test.yml"><img alt="test" src="https://github.com/jeon-jihyeon/meltify/actions/workflows/test.yml/badge.svg"></a>
7
+ <a href="LICENSE"><img alt="MIT" src="https://img.shields.io/badge/license-MIT-blue"></a>
8
+ </p>
9
+
10
+ Agents misread blurry digits, miss text a PDF hides, can't watch video, and paraphrase the one condition that mattered. meltify turns documents, spreadsheets, slides, mail, chat exports, archives, web pages, images, video and audio into markdown an agent can quote, puts a citation on every block, and checks answers before they go out.
11
+
12
+ ## Quickstart
13
+
14
+ Claude Code:
15
+
16
+ ```
17
+ /plugin marketplace add jeon-jihyeon/meltify
18
+ /plugin install meltify@meltify
19
+ ```
20
+
21
+ Codex, Gemini CLI, Cursor and other agents that read Agent Skills:
22
+
23
+ ```
24
+ npx skills add jeon-jihyeon/meltify
25
+ ```
26
+
27
+ Command line only:
28
+
29
+ ```
30
+ uvx meltify read ./inputs
31
+ ```
32
+
33
+ The skills call the `meltify` command. If it isn't installed, the plugin launcher runs it through `uvx`, so [uv](https://docs.astral.sh/uv/) is the only thing you need.
34
+
35
+ ## What it does
36
+
37
+ | Command | Input | Output |
38
+ |---|---|---|
39
+ | `read` | files, folders and URLs of mixed formats | one cited markdown file per item, with images, scans and recordings read by local engines, plus a list of what still needs work |
40
+ | `ocr` | images and scanned PDF pages | each engine's lines with positions, and the values the engines disagree on |
41
+ | `hidden` | PDFs | text a reader doesn't see and why, with page and position |
42
+ | `media` | video, audio or a video URL | timestamped full-resolution frames and a timestamped transcript |
43
+ | `submit` | candidate files and a scoring endpoint | rate-limited submissions that skip duplicates and keep the best score |
44
+ | `check` | an answer file or a string | violations of a schema, counts, unique keys and text rules |
45
+ | `brief` | a problem statement | every condition line quoted with its line number, plus a board for juggling several problems |
46
+ | `doctor` | nothing | which engines, binaries and keys are available, and how to add the rest |
47
+
48
+ Each command has a skill of the same name, such as `meltify-ocr`, that tells the agent when to run it and what to do with the result.
49
+
50
+ ## What `read` handles
51
+
52
+ | Group | Formats | Needs |
53
+ |---|---|---|
54
+ | Documents | PDF, Word `.docx` `.docm` `.dotx` `.dotm` `.doc` `.dot`, HWP and HWPX, RTF, ODT, Pages, WordPerfect, EPUB, XPS, FictionBook, MOBI, HTML and Safari `.webarchive`, Markdown and plain text | `office` for Word and EPUB |
55
+ | Spreadsheets | Excel `.xlsx` `.xlsm` `.xltx` `.xltm` `.xlsb` `.xls` `.xlt`, Hancom `.cell`, ODS, Numbers, Parquet, CSV and TSV | `office` for `.xlsb`, `.xls` and `.xlt`, `iwork` for Numbers, `parquet` for Parquet |
56
+ | Slides | PowerPoint `.pptx` `.pptm` `.potx` `.potm` `.ppsx` `.ppsm` `.ppt` `.pps` `.pot`, Hancom `.show`, ODP, Keynote | `office` for PowerPoint |
57
+ | Drawings | ODG, SVG and `.svgz`, EMF and WMF with `.emz` and `.wmz`, comic book `.cbz` | |
58
+ | Mail and chat | `.eml`, `.msg`, mbox, KakaoTalk exports, Slack export zips | `office` for `.msg` |
59
+ | Archives | zip, tar, tgz, tbz2, txz, 7z, rar, and single files compressed with gz, bz2 or xz | `archive` for 7z and rar, `crypto` for AES zips |
60
+ | Images | PNG, JPEG, WebP, GIF, BMP, TIFF, HEIC, AVIF, JPEG 2000, PSD, ICO, TGA | |
61
+ | Audio and video | mp3, wav, m4a, aac, flac, ogg, opus, wma, aiff, amr, mp4, m4v, mov, mkv, webm, avi, wmv, 3gp, mpg, flv, subtitles | ffmpeg |
62
+ | Web pages | any http(s) URL, video URLs included | `render` for JavaScript-heavy pages, `media` for video sites |
63
+ | Data | JSON, JSONL, XML, YAML, SQLite, Jupyter notebooks, vCard, iCalendar | |
64
+
65
+ Attachments, archive members, files attached to a PDF and downloaded files are melted again as items of their own, up to three levels deep.
66
+
67
+ Templates, macro-enabled and slideshow files read like the document they belong to, and a document zip under an unfamiliar name is still recognized by its content.
68
+
69
+ A binary file no converter reads gets one more try, in this order:
70
+
71
+ - Formats LibreOffice imports, such as Works, Lotus 1-2-3, Visio and Publisher, are rendered to PDF and cited by page. These rows show kind `rendered`
72
+ - Any other picture format Pillow decodes, like PCX or ICNS, goes to OCR as kind `image`
73
+ - Audio or video under an odd name is transcribed as kind `media`
74
+ - On macOS, a full Quick Look preview (kind `rendered`), the text Spotlight indexes (kind `text`) or a first-page thumbnail (kind `rendered`) is the last resort
75
+
76
+ Compiled programs and files of one repeated byte skip all of that and return `unsupported format` right away. Word, Excel and PowerPoint binaries, HWP and iWork files whose own parser gives up take the same path, unless they're encrypted or DRM-locked. `--shallow` skips this step, and `read.fallback = false` turns it off.
77
+
78
+ Pictures get read in the same pass. Images, scanned pages, pictures inside PDF, Word, PowerPoint, Excel, ODF, RTF, EPUB, HTML and mail, and the speech and scene frames of recordings all go through local OCR and speech engines, and the text lands right where the picture was. Native text always comes first, and OCR only covers what parsing can't reach:
79
+
80
+ - Charts in Word, PowerPoint, Excel (`.xlsb` too) and ODF become cited tables from the values the file saved. A Word or PowerPoint chart with no cache reads the workbook embedded beside it, and an Excel chart on formulas nobody calculated asks LibreOffice to calculate them. SmartArt text is read as an indented list
81
+ - EMF and WMF pictures, standalone or inside a document, give up their text records directly. Bitmaps and text they can't decode are drawn by LibreOffice, Quick Look on macOS or the pure-Python `metafile-render` and then OCR'd. With none of them, they're listed in `needs`
82
+ - `.ppt` slides, notes and pictures, and `.doc` inline and floating pictures, are read natively. Pages and Keynote text, tables and pictures are read from the document itself, older iWork '09 files and bundle folders included
83
+ - A scan under a visible header still gets OCR'd, PDF sticky notes become cited blocks, and multi-page TIFFs and animated images are read frame by frame up to 50 frames
84
+
85
+ - Each unique picture or recording is read once, and every engine result is cached by content hash under `${XDG_CACHE_HOME:-~/.cache}/meltify`, so a rerun only pays for what changed
86
+ - `--budget SEC` caps the time spent on uncached work (600 seconds by default). Whatever's left shows up as `ocr (budget)` in `needs`, and the next run picks up where this one stopped
87
+ - `--jobs N` sets how many files convert at once (8 by default). PDFs convert in worker processes of their own
88
+ - `--shallow` skips OCR, speech, and drawing or recalculation work, and lists what it skipped in `needs`. Native text melts as usual
89
+ - When two OCR engines disagree on a number, the block ends with a `> disputed` line. With only one engine, it ends with `> unchecked: one engine`
90
+
91
+ Encrypted files open with a password from `--password-file PATH` or `MELTIFY_PASSWORD`: PDF, Word, Excel and PowerPoint, HWP and HWPX, Pages, Keynote and Numbers, and zip, 7z and rar archives. HWP distribution documents, which only restrict copying and printing, open without one. Without a password, the row says `encrypted` in `needs` and no markdown file is written, and a wrong one says `wrong password`. An archive still lists the members it could open. `--password` works too, but other users on the machine can see it. With more than one set, `--password-file` wins, then `--password`, then `MELTIFY_PASSWORD`.
92
+
93
+ URLs work like files. `read` saves a web page's main text with its heading anchors (`--whole` keeps the full page, `--render` runs it in a headless browser first), downloads documents and melts them by type, and sends video links through the `media` pipeline. Downloads land under `meltify-out/read/web/` and are revalidated with ETags on the next run. By default it refuses private addresses, follows `robots.txt` and stops at 20 MB or 30 seconds. See [SECURITY.md](SECURITY.md) for the details.
94
+
95
+ ## Citations
96
+
97
+ Every result carries `src` and a one-line `cite`:
98
+
99
+ | Input | Cite |
100
+ |---|---|
101
+ | PDF text | `report.pdf#p3@pt(72,95,140,103)` |
102
+ | Image region | `menu.png@px(120,40,380,72)` |
103
+ | Spreadsheet cell | `calendar.xlsx#Calendar!B7` |
104
+ | Mail attachment | `handover.eml#att=cal.xlsx#Calendar` |
105
+ | Archive member | `bundle.zip#att=docs/b.pdf#p3` |
106
+ | Word, ODT or Pages paragraph | `memo.docx:3` |
107
+ | Picture in a document | `memo.docx#para3#img1` |
108
+ | Slide | `deck.pptx#slide4` |
109
+ | Chart in a sheet | `book.xlsx#Sales!D2#img1` |
110
+ | Image frame | `fax.tif#frame2` |
111
+ | PDF sticky note | `report.pdf#p1@pt(300,300,316,316)` |
112
+ | HWP section line | `notice.hwp#s1:12` |
113
+ | Web page line | `https://example.com/guide#install:28` |
114
+ | Text line | `notes.txt:42` |
115
+ | Media span | `call.m4a@00:01:23.4-00:01:27.0` |
116
+ | JSON value | `answers.json#$[3].reason` |
117
+
118
+ Every command prints a table by default, and the full result with `--json`. Exit codes:
119
+
120
+ - `0`: success
121
+ - `1`: violations, failed checks or other errors
122
+ - `2`: usage error or bad input
123
+ - `3`: a needed engine, key, binary or base module is missing
124
+
125
+ ## Engines
126
+
127
+ | Task | Local | Paid API |
128
+ |---|---|---|
129
+ | OCR | Apple Vision on macOS, PaddleOCR | Gemini, Claude, any OpenAI-compatible endpoint |
130
+ | Speech | MLX Whisper on Apple Silicon, whisper.cpp | any OpenAI-compatible transcription endpoint |
131
+
132
+ `auto` uses local engines only. Paid engines run only when you name them in `meltify.toml` or on the command line. An agent can also feed in what it reads in an image with `meltify ocr --reading agent=FILE`, so its own reading gets cross-checked against the local engines without any API key.
133
+
134
+ Optional parts install on request with `meltify doctor --install NAME`:
135
+
136
+ | Extra | Unlocks |
137
+ |---|---|
138
+ | `office` | Word, PowerPoint, Outlook `.msg` and EPUB, legacy `.doc`, `.ppt`, `.xls` and `.xlsb`, faster `.xlsx` reading through calamine, and drawing EMF pictures without LibreOffice |
139
+ | `archive` | 7z and rar through the system libarchive, plus a pinned 7-Zip download for encrypted ones |
140
+ | `iwork` | Numbers tables. Pages and Keynote need nothing extra |
141
+ | `parquet` | Parquet files, the first 200 rows and per-column min, max and null counts |
142
+ | `crypto` | encrypted Office files, AES zips, and password-protected HWP and iWork files |
143
+ | `render` | `read --render` for JavaScript-heavy pages, using your Chrome or a downloaded Chromium |
144
+ | `media` | video URLs through yt-dlp |
145
+ | `asr-mlx` | local speech recognition on Apple Silicon |
146
+ | `ocr-paddle` | PaddleOCR, the second local OCR engine |
147
+
148
+ Recordings need ffmpeg and YouTube downloads need deno. PowerPoint 95, formula recalculation and the rendering fallback need LibreOffice, which `meltify doctor --install libreoffice` downloads as a portable copy on macOS and Linux. `wpd2text` from libwpd reads WordPerfect best, then LibreOffice, and a built-in reader covers the body text without either.
149
+
150
+ ## Configuration
151
+
152
+ Settings are layered, lowest precedence first:
153
+
154
+ 1. The packaged defaults
155
+ 2. `$XDG_CONFIG_HOME/meltify/config.toml` (usually `~/.config/meltify/config.toml`)
156
+ 3. The nearest `meltify.toml`, searched from the working directory up to the git root, or the file you pass with `--config`. Outside a git repository, only the working directory is searched
157
+ 4. `MELTIFY_<KEY>` and `MELTIFY_<SECTION>_<KEY>` environment variables
158
+ 5. Command-line flags
159
+
160
+ See [examples/meltify.toml](examples/meltify.toml) for a sample. A few keys worth knowing:
161
+
162
+ - `lang` is the language OCR expects (`ko` by default)
163
+ - `asr.lang` is the spoken language for speech engines. It's `auto` by default, separate from `lang`, so Whisper detects each recording's language instead of translating it
164
+ - `read.parquet_rows` sets how many Parquet rows melt per file
165
+ - `render.quicklook = false` keeps Quick Look and Spotlight out of every render
166
+
167
+ ## Limits
168
+
169
+ - Text drawn inside an image is just pixels, so `hidden` finds it only through `--contrast` rendering and OCR
170
+ - OCR agreement means the engines read the same value, not that the value is right. A value read only by LLM engines is never marked agreed
171
+ - Speech recognition and OCR quality depend on the engine and the input
172
+ - Linux has no Apple Vision, so with PaddleOCR alone every reading is marked `unchecked`
173
+ - Rendered rows, EMF bitmaps and Quick Look previews are only as good as the OCR that reads them, and Quick Look exists only on macOS
174
+ - Pages and Keynote charts are listed in `needs` instead of read, and a Pages or Keynote file with no text records falls back to its embedded preview image
175
+ - DRM-protected HWP files, password-protected WordPerfect files, and Hancom `.cell` and `.show` files from before 2014 and Hanshow `.hpt` files, which are binary, aren't read. Save the Hancom ones again as `.xlsx` or `.pptx`
176
+ - Parquet files show their first 200 rows, and the rest are counted in `needs`
177
+ - Encrypted 7z and rar need 7-Zip, which `meltify doctor --install archive` downloads. Plain ones open with the system libarchive, a separate package on Linux, with 7-Zip, or with `bsdtar` on macOS
178
+ - The plugin launcher is a POSIX shell script and `doctor --install` only downloads macOS and Linux builds, so Windows isn't supported
179
+
180
+ ## License
181
+
182
+ MIT. meltify depends on [PyMuPDF](https://github.com/pymupdf/PyMuPDF), which is AGPL-3.0, so distributing a product that bundles it comes with AGPL obligations. The `pi-heif` wheels bundle libheif and libde265 (LGPL-3.0), and the `iwork` extra pulls in `enum-tools` (LGPL-3.0) through `numbers-parser`. The programs `doctor --install` downloads keep their own licenses: LibreOffice is MPL-2.0, and 7-Zip is LGPL-2.1 with the unRAR restriction.
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "meltify"
3
- version = "0.2.0"
3
+ version = "0.2.2"
4
4
  description = "Melt files of any format into prompt-ready text where every line cites its source"
5
5
  readme = "README.md"
6
6
  requires-python = ">=3.11.9"
@@ -37,9 +37,17 @@ office = [
37
37
  "markitdown[docx,pptx,outlook]>=0.1.8",
38
38
  "python-calamine>=0.4",
39
39
  "legacy-doc>=0.2",
40
+ "olefile>=0.47",
41
+ "metafile-render>=0.3",
42
+ ]
43
+ crypto = [
44
+ "msoffcrypto-tool>=6",
45
+ "pyzipper>=0.4",
46
+ "cryptography>=42",
40
47
  ]
41
48
  archive = ["libarchive-c>=5"]
42
49
  iwork = ["numbers-parser>=4"]
50
+ parquet = ["pyarrow>=17"]
43
51
  render = ["playwright>=1.50"]
44
52
  media = ["yt-dlp>=2026.8.19"]
45
53
  asr-mlx = ["mlx-whisper>=0.4; sys_platform == 'darwin' and platform_machine == 'arm64'"]
@@ -47,7 +55,7 @@ ocr-paddle = [
47
55
  "paddleocr>=3.2",
48
56
  "paddlepaddle>=3.2",
49
57
  ]
50
- all = ["meltify[office,media,asr-mlx,archive]"]
58
+ all = ["meltify[office,media,asr-mlx,archive,parquet,crypto]"]
51
59
 
52
60
  [project.urls]
53
61
  Repository = "https://github.com/jeon-jihyeon/meltify"
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "meltify"
3
- version = "0.2.0"
3
+ version = "0.2.2"
4
4
  description = "Melt files of any format into prompt-ready text where every line cites its source"
5
5
  readme = "README.md"
6
6
  requires-python = ">=3.11.9"
@@ -23,14 +23,24 @@ dependencies = [
23
23
  ]
24
24
 
25
25
  [project.optional-dependencies]
26
- office = ["markitdown[docx,pptx,outlook]>=0.1.8", "python-calamine>=0.4", "legacy-doc>=0.2"]
26
+ office = [
27
+ "markitdown[docx,pptx,outlook]>=0.1.8",
28
+ "python-calamine>=0.4",
29
+ "legacy-doc>=0.2",
30
+ "olefile>=0.47",
31
+ "metafile-render>=0.3",
32
+ ]
33
+ # Password-protected Office, zip, HWP and iWork files
34
+ crypto = ["msoffcrypto-tool>=6", "pyzipper>=0.4", "cryptography>=42"]
27
35
  archive = ["libarchive-c>=5"]
36
+ # Only Numbers needs it, since Pages and Keynote are read natively
28
37
  iwork = ["numbers-parser>=4"]
38
+ parquet = ["pyarrow>=17"]
29
39
  render = ["playwright>=1.50"]
30
40
  media = ["yt-dlp>=2026.8.19"]
31
41
  asr-mlx = ["mlx-whisper>=0.4; sys_platform == 'darwin' and platform_machine == 'arm64'"]
32
42
  ocr-paddle = ["paddleocr>=3.2", "paddlepaddle>=3.2"]
33
- all = ["meltify[office,media,asr-mlx,archive]"]
43
+ all = ["meltify[office,media,asr-mlx,archive,parquet,crypto]"]
34
44
 
35
45
  [project.urls]
36
46
  Repository = "https://github.com/jeon-jihyeon/meltify"