hammerdown 1.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- hammerdown-1.1.0/PKG-INFO +136 -0
- hammerdown-1.1.0/README.md +123 -0
- hammerdown-1.1.0/pyproject.toml +29 -0
- hammerdown-1.1.0/setup.cfg +4 -0
- hammerdown-1.1.0/src/converter.py +666 -0
- hammerdown-1.1.0/src/hammerdown.egg-info/PKG-INFO +136 -0
- hammerdown-1.1.0/src/hammerdown.egg-info/SOURCES.txt +11 -0
- hammerdown-1.1.0/src/hammerdown.egg-info/dependency_links.txt +1 -0
- hammerdown-1.1.0/src/hammerdown.egg-info/entry_points.txt +2 -0
- hammerdown-1.1.0/src/hammerdown.egg-info/requires.txt +6 -0
- hammerdown-1.1.0/src/hammerdown.egg-info/top_level.txt +1 -0
- hammerdown-1.1.0/tests/test_cli.py +38 -0
- hammerdown-1.1.0/tests/test_golden.py +52 -0
|
@@ -0,0 +1,136 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: hammerdown
|
|
3
|
+
Version: 1.1.0
|
|
4
|
+
Summary: Convert documents to Markdown with extracted images
|
|
5
|
+
Requires-Python: >=3.10
|
|
6
|
+
Description-Content-Type: text/markdown
|
|
7
|
+
Requires-Dist: pymupdf
|
|
8
|
+
Requires-Dist: pymupdf4llm
|
|
9
|
+
Requires-Dist: python-docx
|
|
10
|
+
Requires-Dist: openpyxl
|
|
11
|
+
Requires-Dist: python-pptx
|
|
12
|
+
Requires-Dist: onnxruntime
|
|
13
|
+
|
|
14
|
+
# hammerdown
|
|
15
|
+
|
|
16
|
+
[](https://github.com/sergeypugin/hammerdown/actions)
|
|
17
|
+
[](https://pypi.org/project/hammerdown/)
|
|
18
|
+
[](https://github.com/sergeypugin/hammerdown/releases)
|
|
19
|
+
[](https://www.python.org/)
|
|
20
|
+
[](https://github.com/sergeypugin/hammerdown/releases)
|
|
21
|
+
[](CONTRIBUTING.md)
|
|
22
|
+
|
|
23
|
+
`hammerdown` converts PDF, Word, Excel, PowerPoint, and plain-text documents to Markdown. PDF conversion uses PyMuPDF and PyMuPDF4LLM; source images are extracted into the output folder and Markdown image references are generated.
|
|
24
|
+
|
|
25
|
+
## Table of Contents
|
|
26
|
+
|
|
27
|
+
- [Installation](#installation)
|
|
28
|
+
- [Recommended: Python package (pip)](#recommended-python-package-pip)
|
|
29
|
+
- [Direct Download (Standalone binaries)](#direct-download-standalone-binaries)
|
|
30
|
+
- [Context menu integration](#context-menu-integration)
|
|
31
|
+
- [Usage](#usage)
|
|
32
|
+
- [Supported Formats](#supported-formats)
|
|
33
|
+
- [Output Structure](#output-structure)
|
|
34
|
+
- [Contributing](#contributing)
|
|
35
|
+
|
|
36
|
+
## Installation
|
|
37
|
+
|
|
38
|
+
### Recommended: Python package (pip)
|
|
39
|
+
|
|
40
|
+
If Python (3.10+) is installed on your system, installing via pip is the recommended method:
|
|
41
|
+
|
|
42
|
+
```sh
|
|
43
|
+
pip install hammerdown
|
|
44
|
+
```
|
|
45
|
+
|
|
46
|
+
Or after cloning the repository locally:
|
|
47
|
+
|
|
48
|
+
```sh
|
|
49
|
+
git clone https://github.com/sergeypugin/hammerdown.git
|
|
50
|
+
cd hammerdown
|
|
51
|
+
pip install .
|
|
52
|
+
```
|
|
53
|
+
|
|
54
|
+
### Direct Download (Standalone binaries)
|
|
55
|
+
|
|
56
|
+
If you do not have Python installed, precompiled standalone binaries are available on the [GitHub Releases page](https://github.com/sergeypugin/hammerdown/releases) or you can download directly from the links below:
|
|
57
|
+
|
|
58
|
+
| OS | Download |
|
|
59
|
+
| :--- | :--- |
|
|
60
|
+
| **Windows** | [](https://github.com/sergeypugin/hammerdown/releases/latest/download/hammerdown-windows-x64.exe) |
|
|
61
|
+
| **Linux** | [](https://github.com/sergeypugin/hammerdown/releases/latest/download/hammerdown-linux-x64) |
|
|
62
|
+
| **macOS** | [](https://github.com/sergeypugin/hammerdown/releases/latest/download/hammerdown-macos-x64) [](https://github.com/sergeypugin/hammerdown/releases/latest/download/hammerdown-macos-arm64) |
|
|
63
|
+
|
|
64
|
+
On Linux and macOS, make the downloaded binary executable before running:
|
|
65
|
+
|
|
66
|
+
```sh
|
|
67
|
+
chmod +x ./hammerdown-linux-x64
|
|
68
|
+
./hammerdown-linux-x64 --version
|
|
69
|
+
```
|
|
70
|
+
|
|
71
|
+
### Context menu integration
|
|
72
|
+
|
|
73
|
+
Run `--install` to register file manager integrations, or `--uninstall` to remove them:
|
|
74
|
+
|
|
75
|
+
```sh
|
|
76
|
+
hammerdown --install
|
|
77
|
+
hammerdown --uninstall
|
|
78
|
+
```
|
|
79
|
+
|
|
80
|
+
What happens on each platform:
|
|
81
|
+
- windows: moves binary or script into `%LOCALAPPDATA%\Programs\hammerdown\` and adds a "Convert to Markdown" context menu item for PDF files in Windows Explorer
|
|
82
|
+
- linux: creates launcher in `~/.local/bin/hammerdown` and Nautilus script in `~/.local/share/nautilus/scripts/Convert to Markdown`
|
|
83
|
+
- macOS: creates launcher in `~/.local/bin/hammerdown`
|
|
84
|
+
|
|
85
|
+
Update hammerdown to the latest version at any time:
|
|
86
|
+
|
|
87
|
+
```sh
|
|
88
|
+
hammerdown --update
|
|
89
|
+
```
|
|
90
|
+
|
|
91
|
+
This installer is local and does not require administrator privileges.
|
|
92
|
+
|
|
93
|
+
## Usage
|
|
94
|
+
|
|
95
|
+
Convert one or several supported files:
|
|
96
|
+
|
|
97
|
+
```sh
|
|
98
|
+
hammerdown document.pdf
|
|
99
|
+
hammerdown report.docx workbook.xlsx slides.pptx
|
|
100
|
+
```
|
|
101
|
+
|
|
102
|
+
Files can also be dragged onto the executable. Running `hammerdown` without file arguments opens a file-selection dialog when a graphical desktop is available. `--quiet` suppresses routine status messages, which is used by file-manager integrations:
|
|
103
|
+
|
|
104
|
+
```sh
|
|
105
|
+
hammerdown --quiet document.pdf
|
|
106
|
+
hammerdown --version
|
|
107
|
+
```
|
|
108
|
+
|
|
109
|
+
## Supported Formats
|
|
110
|
+
|
|
111
|
+
Supported document extensions:
|
|
112
|
+
- **PDF & E-books**: `.pdf`, `.epub`, `.mobi`, `.fb2`, `.xps`
|
|
113
|
+
- **Word**: `.docx`, `.doc` (legacy `.doc` via LibreOffice / MS Word)
|
|
114
|
+
- **Excel**: `.xlsx`, `.xls` (legacy `.xls` via LibreOffice / MS Excel)
|
|
115
|
+
- **PowerPoint**: `.pptx`, `.ppt` (legacy `.ppt` via LibreOffice / MS PowerPoint)
|
|
116
|
+
- **Plain text & tabular**: `.txt`, `.md`, `.log`, `.csv`
|
|
117
|
+
|
|
118
|
+
## Output Structure
|
|
119
|
+
|
|
120
|
+
For each input file, an output folder `MD_<name>_<ext>` is written next to the source document:
|
|
121
|
+
|
|
122
|
+
```text
|
|
123
|
+
document.pdf
|
|
124
|
+
MD_document_pdf/
|
|
125
|
+
├── document.md
|
|
126
|
+
└── images/
|
|
127
|
+
└── extracted illustration and embedded document images
|
|
128
|
+
```
|
|
129
|
+
|
|
130
|
+
Output folder names consistently include the file extension suffix (for example, `MD_report_doc/` and `MD_report_docx/`) to prevent conflicts between different document formats with the same base name.
|
|
131
|
+
|
|
132
|
+
Images embedded in PDF, Word, or PowerPoint documents are extracted into `MD_<name>_<ext>/images/`, and the generated Markdown contains relative image links. Conversion continues through a batch if an individual file fails; the process exits with status `1` if any input fails.
|
|
133
|
+
|
|
134
|
+
## Contributing
|
|
135
|
+
|
|
136
|
+
Contributions are welcome. Please refer to [CONTRIBUTING.md](CONTRIBUTING.md) for contribution guidelines. For architectural details, processing pipelines, and internal specifications, see [TECHNICAL.md](TECHNICAL.md).
|
|
@@ -0,0 +1,123 @@
|
|
|
1
|
+
# hammerdown
|
|
2
|
+
|
|
3
|
+
[](https://github.com/sergeypugin/hammerdown/actions)
|
|
4
|
+
[](https://pypi.org/project/hammerdown/)
|
|
5
|
+
[](https://github.com/sergeypugin/hammerdown/releases)
|
|
6
|
+
[](https://www.python.org/)
|
|
7
|
+
[](https://github.com/sergeypugin/hammerdown/releases)
|
|
8
|
+
[](CONTRIBUTING.md)
|
|
9
|
+
|
|
10
|
+
`hammerdown` converts PDF, Word, Excel, PowerPoint, and plain-text documents to Markdown. PDF conversion uses PyMuPDF and PyMuPDF4LLM; source images are extracted into the output folder and Markdown image references are generated.
|
|
11
|
+
|
|
12
|
+
## Table of Contents
|
|
13
|
+
|
|
14
|
+
- [Installation](#installation)
|
|
15
|
+
- [Recommended: Python package (pip)](#recommended-python-package-pip)
|
|
16
|
+
- [Direct Download (Standalone binaries)](#direct-download-standalone-binaries)
|
|
17
|
+
- [Context menu integration](#context-menu-integration)
|
|
18
|
+
- [Usage](#usage)
|
|
19
|
+
- [Supported Formats](#supported-formats)
|
|
20
|
+
- [Output Structure](#output-structure)
|
|
21
|
+
- [Contributing](#contributing)
|
|
22
|
+
|
|
23
|
+
## Installation
|
|
24
|
+
|
|
25
|
+
### Recommended: Python package (pip)
|
|
26
|
+
|
|
27
|
+
If Python (3.10+) is installed on your system, installing via pip is the recommended method:
|
|
28
|
+
|
|
29
|
+
```sh
|
|
30
|
+
pip install hammerdown
|
|
31
|
+
```
|
|
32
|
+
|
|
33
|
+
Or after cloning the repository locally:
|
|
34
|
+
|
|
35
|
+
```sh
|
|
36
|
+
git clone https://github.com/sergeypugin/hammerdown.git
|
|
37
|
+
cd hammerdown
|
|
38
|
+
pip install .
|
|
39
|
+
```
|
|
40
|
+
|
|
41
|
+
### Direct Download (Standalone binaries)
|
|
42
|
+
|
|
43
|
+
If you do not have Python installed, precompiled standalone binaries are available on the [GitHub Releases page](https://github.com/sergeypugin/hammerdown/releases) or you can download directly from the links below:
|
|
44
|
+
|
|
45
|
+
| OS | Download |
|
|
46
|
+
| :--- | :--- |
|
|
47
|
+
| **Windows** | [](https://github.com/sergeypugin/hammerdown/releases/latest/download/hammerdown-windows-x64.exe) |
|
|
48
|
+
| **Linux** | [](https://github.com/sergeypugin/hammerdown/releases/latest/download/hammerdown-linux-x64) |
|
|
49
|
+
| **macOS** | [](https://github.com/sergeypugin/hammerdown/releases/latest/download/hammerdown-macos-x64) [](https://github.com/sergeypugin/hammerdown/releases/latest/download/hammerdown-macos-arm64) |
|
|
50
|
+
|
|
51
|
+
On Linux and macOS, make the downloaded binary executable before running:
|
|
52
|
+
|
|
53
|
+
```sh
|
|
54
|
+
chmod +x ./hammerdown-linux-x64
|
|
55
|
+
./hammerdown-linux-x64 --version
|
|
56
|
+
```
|
|
57
|
+
|
|
58
|
+
### Context menu integration
|
|
59
|
+
|
|
60
|
+
Run `--install` to register file manager integrations, or `--uninstall` to remove them:
|
|
61
|
+
|
|
62
|
+
```sh
|
|
63
|
+
hammerdown --install
|
|
64
|
+
hammerdown --uninstall
|
|
65
|
+
```
|
|
66
|
+
|
|
67
|
+
What happens on each platform:
|
|
68
|
+
- windows: moves binary or script into `%LOCALAPPDATA%\Programs\hammerdown\` and adds a "Convert to Markdown" context menu item for PDF files in Windows Explorer
|
|
69
|
+
- linux: creates launcher in `~/.local/bin/hammerdown` and Nautilus script in `~/.local/share/nautilus/scripts/Convert to Markdown`
|
|
70
|
+
- macOS: creates launcher in `~/.local/bin/hammerdown`
|
|
71
|
+
|
|
72
|
+
Update hammerdown to the latest version at any time:
|
|
73
|
+
|
|
74
|
+
```sh
|
|
75
|
+
hammerdown --update
|
|
76
|
+
```
|
|
77
|
+
|
|
78
|
+
This installer is local and does not require administrator privileges.
|
|
79
|
+
|
|
80
|
+
## Usage
|
|
81
|
+
|
|
82
|
+
Convert one or several supported files:
|
|
83
|
+
|
|
84
|
+
```sh
|
|
85
|
+
hammerdown document.pdf
|
|
86
|
+
hammerdown report.docx workbook.xlsx slides.pptx
|
|
87
|
+
```
|
|
88
|
+
|
|
89
|
+
Files can also be dragged onto the executable. Running `hammerdown` without file arguments opens a file-selection dialog when a graphical desktop is available. `--quiet` suppresses routine status messages, which is used by file-manager integrations:
|
|
90
|
+
|
|
91
|
+
```sh
|
|
92
|
+
hammerdown --quiet document.pdf
|
|
93
|
+
hammerdown --version
|
|
94
|
+
```
|
|
95
|
+
|
|
96
|
+
## Supported Formats
|
|
97
|
+
|
|
98
|
+
Supported document extensions:
|
|
99
|
+
- **PDF & E-books**: `.pdf`, `.epub`, `.mobi`, `.fb2`, `.xps`
|
|
100
|
+
- **Word**: `.docx`, `.doc` (legacy `.doc` via LibreOffice / MS Word)
|
|
101
|
+
- **Excel**: `.xlsx`, `.xls` (legacy `.xls` via LibreOffice / MS Excel)
|
|
102
|
+
- **PowerPoint**: `.pptx`, `.ppt` (legacy `.ppt` via LibreOffice / MS PowerPoint)
|
|
103
|
+
- **Plain text & tabular**: `.txt`, `.md`, `.log`, `.csv`
|
|
104
|
+
|
|
105
|
+
## Output Structure
|
|
106
|
+
|
|
107
|
+
For each input file, an output folder `MD_<name>_<ext>` is written next to the source document:
|
|
108
|
+
|
|
109
|
+
```text
|
|
110
|
+
document.pdf
|
|
111
|
+
MD_document_pdf/
|
|
112
|
+
├── document.md
|
|
113
|
+
└── images/
|
|
114
|
+
└── extracted illustration and embedded document images
|
|
115
|
+
```
|
|
116
|
+
|
|
117
|
+
Output folder names consistently include the file extension suffix (for example, `MD_report_doc/` and `MD_report_docx/`) to prevent conflicts between different document formats with the same base name.
|
|
118
|
+
|
|
119
|
+
Images embedded in PDF, Word, or PowerPoint documents are extracted into `MD_<name>_<ext>/images/`, and the generated Markdown contains relative image links. Conversion continues through a batch if an individual file fails; the process exits with status `1` if any input fails.
|
|
120
|
+
|
|
121
|
+
## Contributing
|
|
122
|
+
|
|
123
|
+
Contributions are welcome. Please refer to [CONTRIBUTING.md](CONTRIBUTING.md) for contribution guidelines. For architectural details, processing pipelines, and internal specifications, see [TECHNICAL.md](TECHNICAL.md).
|
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=68", "wheel"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "hammerdown"
|
|
7
|
+
version = "1.1.0"
|
|
8
|
+
description = "Convert documents to Markdown with extracted images"
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.10"
|
|
11
|
+
dependencies = [
|
|
12
|
+
"pymupdf",
|
|
13
|
+
"pymupdf4llm",
|
|
14
|
+
"python-docx",
|
|
15
|
+
"openpyxl",
|
|
16
|
+
"python-pptx",
|
|
17
|
+
"onnxruntime",
|
|
18
|
+
]
|
|
19
|
+
|
|
20
|
+
[project.scripts]
|
|
21
|
+
hammerdown = "converter:main"
|
|
22
|
+
|
|
23
|
+
[tool.setuptools]
|
|
24
|
+
package-dir = {"" = "src"}
|
|
25
|
+
py-modules = ["converter"]
|
|
26
|
+
|
|
27
|
+
[tool.pytest.ini_options]
|
|
28
|
+
testpaths = ["tests"]
|
|
29
|
+
pythonpath = ["src"]
|
|
@@ -0,0 +1,666 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import argparse
|
|
4
|
+
import csv
|
|
5
|
+
import io
|
|
6
|
+
import logging
|
|
7
|
+
import os
|
|
8
|
+
import re
|
|
9
|
+
import shutil
|
|
10
|
+
import stat
|
|
11
|
+
import subprocess
|
|
12
|
+
import sys
|
|
13
|
+
import tempfile
|
|
14
|
+
import time
|
|
15
|
+
import zipfile
|
|
16
|
+
from pathlib import Path
|
|
17
|
+
from typing import Any, Sequence
|
|
18
|
+
|
|
19
|
+
__version__ = "1.1.0"
|
|
20
|
+
|
|
21
|
+
logging.basicConfig(
|
|
22
|
+
level=logging.INFO,
|
|
23
|
+
format="%(asctime)s [%(levelname)s] %(message)s",
|
|
24
|
+
datefmt="%H:%M:%S",
|
|
25
|
+
)
|
|
26
|
+
logger = logging.getLogger("hammerdown")
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def normalize_path(path_str: str | os.PathLike[str]) -> str:
|
|
30
|
+
value = os.fspath(path_str).strip().strip("'\"")
|
|
31
|
+
if not value:
|
|
32
|
+
return ""
|
|
33
|
+
|
|
34
|
+
if os.name == "nt" and value.lower().startswith("/mnt/"):
|
|
35
|
+
parts = value.split("/")
|
|
36
|
+
if len(parts) >= 3 and len(parts[2]) == 1:
|
|
37
|
+
value = f"{parts[2].upper()}:\\" + "\\".join(parts[3:])
|
|
38
|
+
elif os.name != "nt":
|
|
39
|
+
match = re.match(r"^([A-Za-z]):[\\/](.*)$", value)
|
|
40
|
+
if match:
|
|
41
|
+
value = f"/mnt/{match.group(1).lower()}/{match.group(2).replace(chr(92), '/') }"
|
|
42
|
+
|
|
43
|
+
return str(Path(value).expanduser().resolve())
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def convert_pdf_or_ebook(file_path: str, out_dir: str, stem: str) -> tuple[str | None, int]:
|
|
47
|
+
try:
|
|
48
|
+
import pymupdf # type: ignore
|
|
49
|
+
import pymupdf4llm # type: ignore
|
|
50
|
+
except ImportError:
|
|
51
|
+
logger.error("PDF support requires pymupdf and pymupdf4llm")
|
|
52
|
+
return None, 0
|
|
53
|
+
|
|
54
|
+
img_dir = Path(out_dir) / "images"
|
|
55
|
+
img_dir.mkdir(parents=True, exist_ok=True)
|
|
56
|
+
|
|
57
|
+
saved_imgs = 0
|
|
58
|
+
extracted_xrefs: set[int] = set()
|
|
59
|
+
with pymupdf.open(file_path) as doc:
|
|
60
|
+
for page_number in range(len(doc)):
|
|
61
|
+
page = doc[page_number]
|
|
62
|
+
for img_info in page.get_images():
|
|
63
|
+
xref = img_info[0]
|
|
64
|
+
if xref in extracted_xrefs:
|
|
65
|
+
continue
|
|
66
|
+
|
|
67
|
+
img_data = doc.extract_image(xref)
|
|
68
|
+
if img_data["width"] < 50 or img_data["height"] < 50:
|
|
69
|
+
continue
|
|
70
|
+
|
|
71
|
+
img_name = f"{stem}_p{page_number:03d}_xref{xref}.{img_data['ext']}"
|
|
72
|
+
(img_dir / img_name).write_bytes(img_data["image"])
|
|
73
|
+
extracted_xrefs.add(xref)
|
|
74
|
+
saved_imgs += 1
|
|
75
|
+
|
|
76
|
+
md_text = pymupdf4llm.to_markdown(file_path, write_images=False)
|
|
77
|
+
return str(md_text), saved_imgs
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def convert_docx(docx_path: str, out_dir: str) -> tuple[str | None, int]:
|
|
81
|
+
try:
|
|
82
|
+
import docx # type: ignore
|
|
83
|
+
except ImportError:
|
|
84
|
+
logger.error("DOCX support requires python-docx")
|
|
85
|
+
return None, 0
|
|
86
|
+
|
|
87
|
+
doc = docx.Document(docx_path)
|
|
88
|
+
lines: list[str] = []
|
|
89
|
+
for paragraph in doc.paragraphs:
|
|
90
|
+
if paragraph.text.strip():
|
|
91
|
+
lines.append(paragraph.text)
|
|
92
|
+
|
|
93
|
+
for table in doc.tables:
|
|
94
|
+
if not table.rows:
|
|
95
|
+
continue
|
|
96
|
+
header_cells = [cell.text.strip() for cell in table.rows[0].cells]
|
|
97
|
+
lines.append(f"| {' | '.join(header_cells)} |")
|
|
98
|
+
lines.append(f"| {' | '.join(['---'] * len(header_cells))} |")
|
|
99
|
+
for row in table.rows[1:]:
|
|
100
|
+
row_text = " | ".join(cell.text.strip() for cell in row.cells)
|
|
101
|
+
lines.append(f"| {row_text} |")
|
|
102
|
+
|
|
103
|
+
return "\n\n".join(lines), 0
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def convert_xlsx(xlsx_path: str, out_dir: str) -> tuple[str | None, int]:
|
|
107
|
+
try:
|
|
108
|
+
import openpyxl # type: ignore
|
|
109
|
+
except ImportError:
|
|
110
|
+
logger.error("XLSX support requires openpyxl")
|
|
111
|
+
return None, 0
|
|
112
|
+
|
|
113
|
+
workbook = openpyxl.load_workbook(xlsx_path, data_only=True, read_only=True)
|
|
114
|
+
lines: list[str] = []
|
|
115
|
+
for sheet in workbook.sheetnames:
|
|
116
|
+
lines.append(f"# Sheet: {sheet}\n")
|
|
117
|
+
rows = workbook[sheet].iter_rows(values_only=True)
|
|
118
|
+
header = next(rows, None)
|
|
119
|
+
if header is None:
|
|
120
|
+
continue
|
|
121
|
+
header_cells = [str(cell) if cell is not None else "" for cell in header]
|
|
122
|
+
lines.append(f"| {' | '.join(header_cells)} |")
|
|
123
|
+
lines.append(f"| {' | '.join(['---'] * len(header_cells))} |")
|
|
124
|
+
for row in rows:
|
|
125
|
+
if any(row):
|
|
126
|
+
row_text = " | ".join(str(cell) if cell is not None else "" for cell in row)
|
|
127
|
+
lines.append(f"| {row_text} |")
|
|
128
|
+
lines.append("\n")
|
|
129
|
+
workbook.close()
|
|
130
|
+
return "\n".join(lines), 0
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
def convert_pptx(pptx_path: str, out_dir: str) -> tuple[str | None, int]:
|
|
134
|
+
try:
|
|
135
|
+
from pptx import Presentation # type: ignore
|
|
136
|
+
except ImportError:
|
|
137
|
+
logger.error("PPTX support requires python-pptx")
|
|
138
|
+
return None, 0
|
|
139
|
+
|
|
140
|
+
presentation = Presentation(pptx_path)
|
|
141
|
+
lines: list[str] = []
|
|
142
|
+
for slide_idx, slide in enumerate(list(presentation.slides), start=1):
|
|
143
|
+
lines.append(f"## Slide {slide_idx}\n")
|
|
144
|
+
for shape in slide.shapes:
|
|
145
|
+
if getattr(shape, "has_table", False):
|
|
146
|
+
table = getattr(shape, "table", None)
|
|
147
|
+
if table is None:
|
|
148
|
+
continue
|
|
149
|
+
rows = list(table.rows)
|
|
150
|
+
if rows:
|
|
151
|
+
header = [cell.text.strip() for cell in rows[0].cells]
|
|
152
|
+
lines.append(f"| {' | '.join(header)} |")
|
|
153
|
+
lines.append(f"| {' | '.join(['---'] * len(header))} |")
|
|
154
|
+
for row in rows[1:]:
|
|
155
|
+
lines.append(f"| {' | '.join(cell.text.strip() for cell in row.cells)} |")
|
|
156
|
+
elif getattr(shape, "has_text_frame", False):
|
|
157
|
+
text_frame = getattr(shape, "text_frame", None)
|
|
158
|
+
if text_frame is None:
|
|
159
|
+
continue
|
|
160
|
+
for paragraph in text_frame.paragraphs:
|
|
161
|
+
text = paragraph.text.strip()
|
|
162
|
+
if text:
|
|
163
|
+
lines.append(text)
|
|
164
|
+
lines.append("")
|
|
165
|
+
|
|
166
|
+
return "\n".join(lines), 0
|
|
167
|
+
|
|
168
|
+
|
|
169
|
+
def _convert_via_soffice(file_path: str, out_ext: str) -> str | None:
|
|
170
|
+
soffice_bin = shutil.which("soffice") or shutil.which("libreoffice")
|
|
171
|
+
if not soffice_bin and os.name == "nt":
|
|
172
|
+
common_paths = [
|
|
173
|
+
Path(os.environ.get("PROGRAMFILES", "C:\\Program Files")) / "LibreOffice" / "program" / "soffice.exe",
|
|
174
|
+
Path(os.environ.get("PROGRAMFILES(X86)", "C:\\Program Files (x86)")) / "LibreOffice" / "program" / "soffice.exe",
|
|
175
|
+
]
|
|
176
|
+
for p in common_paths:
|
|
177
|
+
if p.is_file():
|
|
178
|
+
soffice_bin = str(p)
|
|
179
|
+
break
|
|
180
|
+
|
|
181
|
+
if not soffice_bin:
|
|
182
|
+
return None
|
|
183
|
+
|
|
184
|
+
temp_dir = tempfile.mkdtemp()
|
|
185
|
+
cmd = [
|
|
186
|
+
soffice_bin,
|
|
187
|
+
"--headless",
|
|
188
|
+
"--convert-to",
|
|
189
|
+
out_ext,
|
|
190
|
+
"--outdir",
|
|
191
|
+
temp_dir,
|
|
192
|
+
file_path,
|
|
193
|
+
]
|
|
194
|
+
try:
|
|
195
|
+
subprocess.run(cmd, check=True, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL)
|
|
196
|
+
converted_file = Path(temp_dir) / f"{Path(file_path).stem}.{out_ext}"
|
|
197
|
+
if converted_file.is_file():
|
|
198
|
+
return str(converted_file)
|
|
199
|
+
except Exception:
|
|
200
|
+
pass
|
|
201
|
+
return None
|
|
202
|
+
|
|
203
|
+
|
|
204
|
+
def convert_doc(doc_path: str, out_dir: str) -> tuple[str | None, int]:
|
|
205
|
+
if zipfile.is_zipfile(doc_path):
|
|
206
|
+
return convert_docx(doc_path, out_dir)
|
|
207
|
+
|
|
208
|
+
converted = _convert_via_soffice(doc_path, "docx")
|
|
209
|
+
if converted:
|
|
210
|
+
return convert_docx(converted, out_dir)
|
|
211
|
+
|
|
212
|
+
if os.name == "nt":
|
|
213
|
+
try:
|
|
214
|
+
import win32com.client # type: ignore
|
|
215
|
+
word = win32com.client.Dispatch("Word.Application")
|
|
216
|
+
word.Visible = False
|
|
217
|
+
temp_file = Path(tempfile.mkdtemp()) / f"{Path(doc_path).stem}.docx"
|
|
218
|
+
doc = word.Documents.Open(str(Path(doc_path).resolve()))
|
|
219
|
+
doc.SaveAs2(str(temp_file.resolve()), FileFormat=16)
|
|
220
|
+
doc.Close()
|
|
221
|
+
word.Quit()
|
|
222
|
+
if temp_file.is_file():
|
|
223
|
+
return convert_docx(str(temp_file), out_dir)
|
|
224
|
+
except Exception:
|
|
225
|
+
pass
|
|
226
|
+
|
|
227
|
+
logger.error("Legacy .doc format requires LibreOffice or Microsoft Word to be installed")
|
|
228
|
+
return None, 0
|
|
229
|
+
|
|
230
|
+
|
|
231
|
+
def convert_xls(xls_path: str, out_dir: str) -> tuple[str | None, int]:
|
|
232
|
+
if zipfile.is_zipfile(xls_path):
|
|
233
|
+
return convert_xlsx(xls_path, out_dir)
|
|
234
|
+
|
|
235
|
+
converted = _convert_via_soffice(xls_path, "xlsx")
|
|
236
|
+
if converted:
|
|
237
|
+
return convert_xlsx(converted, out_dir)
|
|
238
|
+
|
|
239
|
+
if os.name == "nt":
|
|
240
|
+
try:
|
|
241
|
+
import win32com.client # type: ignore
|
|
242
|
+
excel = win32com.client.Dispatch("Excel.Application")
|
|
243
|
+
excel.Visible = False
|
|
244
|
+
temp_file = Path(tempfile.mkdtemp()) / f"{Path(xls_path).stem}.xlsx"
|
|
245
|
+
wb = excel.Workbooks.Open(str(Path(xls_path).resolve()))
|
|
246
|
+
wb.SaveAs(str(temp_file.resolve()), FileFormat=51)
|
|
247
|
+
wb.Close()
|
|
248
|
+
excel.Quit()
|
|
249
|
+
if temp_file.is_file():
|
|
250
|
+
return convert_xlsx(str(temp_file), out_dir)
|
|
251
|
+
except Exception:
|
|
252
|
+
pass
|
|
253
|
+
|
|
254
|
+
logger.error("Legacy .xls format requires LibreOffice or Microsoft Excel to be installed")
|
|
255
|
+
return None, 0
|
|
256
|
+
|
|
257
|
+
|
|
258
|
+
def convert_ppt(ppt_path: str, out_dir: str) -> tuple[str | None, int]:
|
|
259
|
+
if zipfile.is_zipfile(ppt_path):
|
|
260
|
+
return convert_pptx(ppt_path, out_dir)
|
|
261
|
+
|
|
262
|
+
converted = _convert_via_soffice(ppt_path, "pptx")
|
|
263
|
+
if converted:
|
|
264
|
+
return convert_pptx(converted, out_dir)
|
|
265
|
+
|
|
266
|
+
if os.name == "nt":
|
|
267
|
+
try:
|
|
268
|
+
import win32com.client # type: ignore
|
|
269
|
+
powerpoint = win32com.client.Dispatch("PowerPoint.Application")
|
|
270
|
+
temp_file = Path(tempfile.mkdtemp()) / f"{Path(ppt_path).stem}.pptx"
|
|
271
|
+
ppt = powerpoint.Presentations.Open(str(Path(ppt_path).resolve()), WithWindow=False)
|
|
272
|
+
ppt.SaveAs(str(temp_file.resolve()), FileFormat=24)
|
|
273
|
+
ppt.Close()
|
|
274
|
+
powerpoint.Quit()
|
|
275
|
+
if temp_file.is_file():
|
|
276
|
+
return convert_pptx(str(temp_file), out_dir)
|
|
277
|
+
except Exception:
|
|
278
|
+
pass
|
|
279
|
+
|
|
280
|
+
logger.error("Legacy .ppt format requires LibreOffice or Microsoft PowerPoint to be installed")
|
|
281
|
+
return None, 0
|
|
282
|
+
|
|
283
|
+
|
|
284
|
+
def convert_txt_or_md(file_path: str, out_dir: str) -> tuple[str | None, int]:
|
|
285
|
+
raw = Path(file_path).read_bytes()
|
|
286
|
+
for enc in ("utf-8-sig", "utf-8", "cp1251", "cp1252", "latin-1"):
|
|
287
|
+
try:
|
|
288
|
+
return raw.decode(enc), 0
|
|
289
|
+
except UnicodeDecodeError:
|
|
290
|
+
continue
|
|
291
|
+
return raw.decode("utf-8", errors="replace"), 0
|
|
292
|
+
|
|
293
|
+
|
|
294
|
+
def convert_csv(csv_path: str, out_dir: str) -> tuple[str | None, int]:
|
|
295
|
+
if zipfile.is_zipfile(csv_path):
|
|
296
|
+
try:
|
|
297
|
+
import openpyxl # type: ignore
|
|
298
|
+
data = Path(csv_path).read_bytes()
|
|
299
|
+
workbook = openpyxl.load_workbook(io.BytesIO(data), data_only=True, read_only=True)
|
|
300
|
+
lines: list[str] = []
|
|
301
|
+
for sheet in workbook.sheetnames:
|
|
302
|
+
lines.append(f"# Sheet: {sheet}\n")
|
|
303
|
+
rows = workbook[sheet].iter_rows(values_only=True)
|
|
304
|
+
header = next(rows, None)
|
|
305
|
+
if header is None:
|
|
306
|
+
continue
|
|
307
|
+
header_cells = [str(cell) if cell is not None else "" for cell in header]
|
|
308
|
+
lines.append(f"| {' | '.join(header_cells)} |")
|
|
309
|
+
lines.append(f"| {' | '.join(['---'] * len(header_cells))} |")
|
|
310
|
+
for row in rows:
|
|
311
|
+
if any(row):
|
|
312
|
+
row_text = " | ".join(str(cell) if cell is not None else "" for cell in row)
|
|
313
|
+
lines.append(f"| {row_text} |")
|
|
314
|
+
lines.append("\n")
|
|
315
|
+
workbook.close()
|
|
316
|
+
return "\n".join(lines), 0
|
|
317
|
+
except Exception:
|
|
318
|
+
pass
|
|
319
|
+
|
|
320
|
+
raw = Path(csv_path).read_bytes()
|
|
321
|
+
text = None
|
|
322
|
+
for enc in ("utf-8-sig", "utf-8", "cp1251", "cp1252", "latin-1"):
|
|
323
|
+
try:
|
|
324
|
+
text = raw.decode(enc)
|
|
325
|
+
break
|
|
326
|
+
except UnicodeDecodeError:
|
|
327
|
+
continue
|
|
328
|
+
|
|
329
|
+
if text is None:
|
|
330
|
+
text = raw.decode("utf-8", errors="replace")
|
|
331
|
+
|
|
332
|
+
sample = text[:2048]
|
|
333
|
+
delimiter = ","
|
|
334
|
+
try:
|
|
335
|
+
dialect = csv.Sniffer().sniff(sample, delimiters=",;\t|")
|
|
336
|
+
delimiter = dialect.delimiter
|
|
337
|
+
except Exception:
|
|
338
|
+
if ";" in sample and "," not in sample:
|
|
339
|
+
delimiter = ";"
|
|
340
|
+
elif "\t" in sample and "," not in sample:
|
|
341
|
+
delimiter = "\t"
|
|
342
|
+
|
|
343
|
+
reader = csv.reader(io.StringIO(text), delimiter=delimiter)
|
|
344
|
+
rows = list(reader)
|
|
345
|
+
if not rows:
|
|
346
|
+
return "", 0
|
|
347
|
+
|
|
348
|
+
lines = []
|
|
349
|
+
header = [cell.strip() for cell in rows[0]]
|
|
350
|
+
lines.append(f"| {' | '.join(header)} |")
|
|
351
|
+
lines.append(f"| {' | '.join(['---'] * len(header))} |")
|
|
352
|
+
for row in rows[1:]:
|
|
353
|
+
if any(cell.strip() for cell in row):
|
|
354
|
+
cells = [cell.strip() for cell in row]
|
|
355
|
+
lines.append(f"| {' | '.join(cells)} |")
|
|
356
|
+
|
|
357
|
+
return "\n".join(lines), 0
|
|
358
|
+
|
|
359
|
+
|
|
360
|
+
def convert_file(file_path: str | os.PathLike[str]) -> bool:
|
|
361
|
+
normalized_path = normalize_path(file_path)
|
|
362
|
+
source = Path(normalized_path)
|
|
363
|
+
if not source.is_file():
|
|
364
|
+
logger.error("File not found: %s", normalized_path)
|
|
365
|
+
return False
|
|
366
|
+
|
|
367
|
+
stem = source.stem
|
|
368
|
+
extension = source.suffix.lower()
|
|
369
|
+
ext_clean = extension.lstrip(".")
|
|
370
|
+
|
|
371
|
+
folder_name = f"MD_{stem}_{ext_clean}" if ext_clean else f"MD_{stem}"
|
|
372
|
+
out_dir = source.parent / folder_name
|
|
373
|
+
out_dir.mkdir(parents=True, exist_ok=True)
|
|
374
|
+
out_md = out_dir / f"{stem}.md"
|
|
375
|
+
|
|
376
|
+
converters = {
|
|
377
|
+
".pdf": convert_pdf_or_ebook,
|
|
378
|
+
".epub": convert_pdf_or_ebook,
|
|
379
|
+
".mobi": convert_pdf_or_ebook,
|
|
380
|
+
".fb2": convert_pdf_or_ebook,
|
|
381
|
+
".xps": convert_pdf_or_ebook,
|
|
382
|
+
".docx": convert_docx,
|
|
383
|
+
".doc": convert_doc,
|
|
384
|
+
".xlsx": convert_xlsx,
|
|
385
|
+
".xls": convert_xls,
|
|
386
|
+
".pptx": convert_pptx,
|
|
387
|
+
".ppt": convert_ppt,
|
|
388
|
+
".csv": convert_csv,
|
|
389
|
+
".txt": convert_txt_or_md,
|
|
390
|
+
".md": convert_txt_or_md,
|
|
391
|
+
".log": convert_txt_or_md,
|
|
392
|
+
}
|
|
393
|
+
converter = converters.get(extension)
|
|
394
|
+
if converter is None:
|
|
395
|
+
logger.error("Unsupported format: %s", extension or "(no extension)")
|
|
396
|
+
return False
|
|
397
|
+
|
|
398
|
+
logger.info("Processing %s...", source.name)
|
|
399
|
+
started_at = time.monotonic()
|
|
400
|
+
try:
|
|
401
|
+
if extension in {".pdf", ".epub", ".mobi", ".fb2", ".xps"}:
|
|
402
|
+
md_text, saved_images = converter(str(source), str(out_dir), stem)
|
|
403
|
+
else:
|
|
404
|
+
md_text, saved_images = converter(str(source), str(out_dir))
|
|
405
|
+
if md_text is None:
|
|
406
|
+
return False
|
|
407
|
+
|
|
408
|
+
# Trim trailing whitespace on each line and ensure a single final newline
|
|
409
|
+
normalized_text = md_text.replace("\r\n", "\n").replace("\r", "\n")
|
|
410
|
+
trimmed_lines = [line.rstrip() for line in normalized_text.split("\n")]
|
|
411
|
+
cleaned_md = "\n".join(trimmed_lines).rstrip() + "\n"
|
|
412
|
+
out_md.write_text(cleaned_md, encoding="utf-8", newline="\n")
|
|
413
|
+
logger.info(
|
|
414
|
+
"Completed %s in %.2fs (saved %d images)",
|
|
415
|
+
source.name,
|
|
416
|
+
time.monotonic() - started_at,
|
|
417
|
+
saved_images,
|
|
418
|
+
)
|
|
419
|
+
return True
|
|
420
|
+
except Exception:
|
|
421
|
+
logger.exception("Failed to convert %s", source.name)
|
|
422
|
+
return False
|
|
423
|
+
|
|
424
|
+
|
|
425
|
+
def _installed_command() -> tuple[Path, str]:
|
|
426
|
+
if os.name == "nt":
|
|
427
|
+
app_dir = Path(os.environ.get("LOCALAPPDATA", Path.home() / "AppData/Local")) / "Programs" / "hammerdown"
|
|
428
|
+
else:
|
|
429
|
+
app_dir = Path.home() / ".local" / "share" / "hammerdown"
|
|
430
|
+
app_dir.mkdir(parents=True, exist_ok=True)
|
|
431
|
+
|
|
432
|
+
if getattr(sys, "frozen", False):
|
|
433
|
+
suffix = ".exe" if os.name == "nt" else ""
|
|
434
|
+
target = app_dir / f"hammerdown{suffix}"
|
|
435
|
+
if Path(sys.executable).resolve() != target.resolve():
|
|
436
|
+
shutil.move(sys.executable, target)
|
|
437
|
+
return app_dir, str(target)
|
|
438
|
+
|
|
439
|
+
target = app_dir / "converter.py"
|
|
440
|
+
shutil.copy2(Path(__file__).resolve(), target)
|
|
441
|
+
return app_dir, f'"{sys.executable}" "{target}"'
|
|
442
|
+
|
|
443
|
+
|
|
444
|
+
def _install_windows(command: str) -> None:
|
|
445
|
+
import winreg
|
|
446
|
+
|
|
447
|
+
key_path = r"Software\Classes\SystemFileAssociations\.pdf\shell\Convert to Markdown"
|
|
448
|
+
with winreg.CreateKey(winreg.HKEY_CURRENT_USER, key_path) as key:
|
|
449
|
+
winreg.SetValueEx(key, "", 0, winreg.REG_SZ, "Convert to Markdown")
|
|
450
|
+
winreg.SetValueEx(key, "Icon", 0, winreg.REG_SZ, command.split('"')[1])
|
|
451
|
+
with winreg.CreateKey(winreg.HKEY_CURRENT_USER, key_path + r"\command") as key:
|
|
452
|
+
winreg.SetValueEx(key, "", 0, winreg.REG_SZ, f'{command} --quiet "%1"')
|
|
453
|
+
|
|
454
|
+
|
|
455
|
+
def _install_unix(command: str) -> None:
|
|
456
|
+
bin_dir = Path.home() / ".local" / "bin"
|
|
457
|
+
bin_dir.mkdir(parents=True, exist_ok=True)
|
|
458
|
+
launcher = bin_dir / "hammerdown"
|
|
459
|
+
launcher.write_text(f"#!/bin/sh\nexec {command} \"$@\"\n", encoding="utf-8")
|
|
460
|
+
launcher.chmod(launcher.stat().st_mode | stat.S_IXUSR | stat.S_IXGRP | stat.S_IXOTH)
|
|
461
|
+
|
|
462
|
+
if sys.platform.startswith("linux"):
|
|
463
|
+
nautilus_dir = Path.home() / ".local" / "share" / "nautilus" / "scripts"
|
|
464
|
+
nautilus_dir.mkdir(parents=True, exist_ok=True)
|
|
465
|
+
script = nautilus_dir / "Convert to Markdown"
|
|
466
|
+
script.write_text('exec "$HOME/.local/bin/hammerdown" --quiet "$@"\n', encoding="utf-8")
|
|
467
|
+
script.chmod(script.stat().st_mode | stat.S_IXUSR | stat.S_IXGRP | stat.S_IXOTH)
|
|
468
|
+
|
|
469
|
+
|
|
470
|
+
def install() -> bool:
|
|
471
|
+
try:
|
|
472
|
+
_, command = _installed_command()
|
|
473
|
+
if os.name == "nt":
|
|
474
|
+
_install_windows(command)
|
|
475
|
+
else:
|
|
476
|
+
_install_unix(command)
|
|
477
|
+
logger.info("hammerdown installed. Restart the file manager if its menu does not update.")
|
|
478
|
+
return True
|
|
479
|
+
except (OSError, ImportError) as exc:
|
|
480
|
+
logger.error("Installation failed: %s", exc)
|
|
481
|
+
return False
|
|
482
|
+
|
|
483
|
+
|
|
484
|
+
def _remove_registry_tree(root: Any, path: str) -> None:
|
|
485
|
+
import winreg
|
|
486
|
+
|
|
487
|
+
try:
|
|
488
|
+
with winreg.OpenKey(root, path, 0, winreg.KEY_READ | winreg.KEY_WRITE) as key:
|
|
489
|
+
while True:
|
|
490
|
+
try:
|
|
491
|
+
child = winreg.EnumKey(key, 0)
|
|
492
|
+
except OSError:
|
|
493
|
+
break
|
|
494
|
+
_remove_registry_tree(root, f"{path}\\{child}")
|
|
495
|
+
winreg.DeleteKey(root, path)
|
|
496
|
+
except FileNotFoundError:
|
|
497
|
+
pass
|
|
498
|
+
|
|
499
|
+
|
|
500
|
+
def uninstall() -> bool:
|
|
501
|
+
try:
|
|
502
|
+
if os.name == "nt":
|
|
503
|
+
import winreg
|
|
504
|
+
|
|
505
|
+
_remove_registry_tree(
|
|
506
|
+
winreg.HKEY_CURRENT_USER,
|
|
507
|
+
r"Software\Classes\SystemFileAssociations\.pdf\shell\Convert to Markdown",
|
|
508
|
+
)
|
|
509
|
+
install_dir = Path(os.environ.get("LOCALAPPDATA", Path.home() / "AppData/Local")) / "Programs" / "hammerdown"
|
|
510
|
+
else:
|
|
511
|
+
(Path.home() / ".local" / "bin" / "hammerdown").unlink(missing_ok=True)
|
|
512
|
+
(Path.home() / ".local" / "share" / "nautilus" / "scripts" / "Convert to Markdown").unlink(missing_ok=True)
|
|
513
|
+
install_dir = Path.home() / ".local" / "share" / "hammerdown"
|
|
514
|
+
|
|
515
|
+
try:
|
|
516
|
+
if install_dir.exists():
|
|
517
|
+
shutil.rmtree(install_dir)
|
|
518
|
+
except OSError:
|
|
519
|
+
logger.warning("Could not remove installed program files at %s", install_dir)
|
|
520
|
+
logger.info("hammerdown integration uninstalled")
|
|
521
|
+
return True
|
|
522
|
+
except (OSError, ImportError) as exc:
|
|
523
|
+
logger.error("Uninstallation failed: %s", exc)
|
|
524
|
+
return False
|
|
525
|
+
|
|
526
|
+
|
|
527
|
+
def update() -> bool:
|
|
528
|
+
if not getattr(sys, "frozen", False):
|
|
529
|
+
logger.info("Updating Python package via pip...")
|
|
530
|
+
cmd = [sys.executable, "-m", "pip", "install", "--upgrade", "git+https://github.com/sergeypugin/hammerdown.git"]
|
|
531
|
+
try:
|
|
532
|
+
subprocess.check_call(cmd)
|
|
533
|
+
logger.info("Successfully updated hammerdown package.")
|
|
534
|
+
return True
|
|
535
|
+
except subprocess.CalledProcessError as exc:
|
|
536
|
+
logger.error("Failed to update via pip: %s", exc)
|
|
537
|
+
return False
|
|
538
|
+
|
|
539
|
+
import json
|
|
540
|
+
import urllib.request
|
|
541
|
+
|
|
542
|
+
url = "https://api.github.com/repos/sergeypugin/hammerdown/releases/latest"
|
|
543
|
+
req = urllib.request.Request(url, headers={"User-Agent": "hammerdown"})
|
|
544
|
+
try:
|
|
545
|
+
with urllib.request.urlopen(req, timeout=15) as resp:
|
|
546
|
+
data = json.loads(resp.read().decode("utf-8"))
|
|
547
|
+
latest_tag = data.get("tag_name", "").lstrip("v")
|
|
548
|
+
if latest_tag and latest_tag <= __version__:
|
|
549
|
+
logger.info("hammerdown is already at the latest version (%s)", __version__)
|
|
550
|
+
return True
|
|
551
|
+
|
|
552
|
+
logger.info("New version available: %s (current: %s)", latest_tag, __version__)
|
|
553
|
+
asset_name = "hammerdown-windows-x64.exe" if os.name == "nt" else "hammerdown-linux-x64"
|
|
554
|
+
if sys.platform == "darwin":
|
|
555
|
+
asset_name = "hammerdown-macos-arm64" if "arm" in os.uname().machine.lower() else "hammerdown-macos-x64"
|
|
556
|
+
|
|
557
|
+
download_url = None
|
|
558
|
+
for asset in data.get("assets", []):
|
|
559
|
+
if asset.get("name") == asset_name:
|
|
560
|
+
download_url = asset.get("browser_download_url")
|
|
561
|
+
break
|
|
562
|
+
|
|
563
|
+
if not download_url:
|
|
564
|
+
logger.error("No release asset found matching %s", asset_name)
|
|
565
|
+
return False
|
|
566
|
+
|
|
567
|
+
logger.info("Downloading %s...", download_url)
|
|
568
|
+
temp_exe = Path(tempfile.gettempdir()) / f"hammerdown-update{'.exe' if os.name == 'nt' else ''}"
|
|
569
|
+
with urllib.request.urlopen(download_url, timeout=30) as resp, open(temp_exe, "wb") as f:
|
|
570
|
+
f.write(resp.read())
|
|
571
|
+
|
|
572
|
+
target = Path(sys.executable)
|
|
573
|
+
if os.name == "nt":
|
|
574
|
+
old_exe = target.with_suffix(".exe.old")
|
|
575
|
+
if old_exe.exists():
|
|
576
|
+
try:
|
|
577
|
+
old_exe.unlink()
|
|
578
|
+
except OSError:
|
|
579
|
+
pass
|
|
580
|
+
target.rename(old_exe)
|
|
581
|
+
shutil.copy2(temp_exe, target)
|
|
582
|
+
else:
|
|
583
|
+
shutil.copy2(temp_exe, target)
|
|
584
|
+
target.chmod(target.stat().st_mode | stat.S_IXUSR | stat.S_IXGRP | stat.S_IXOTH)
|
|
585
|
+
|
|
586
|
+
logger.info("hammerdown successfully updated to version %s!", latest_tag)
|
|
587
|
+
return True
|
|
588
|
+
except Exception as exc:
|
|
589
|
+
logger.error("Failed to update: %s", exc)
|
|
590
|
+
return False
|
|
591
|
+
|
|
592
|
+
|
|
593
|
+
def _select_files() -> Sequence[str]:
|
|
594
|
+
if os.name == "nt":
|
|
595
|
+
try:
|
|
596
|
+
ps_script = (
|
|
597
|
+
"Add-Type -AssemblyName PresentationFramework | Out-Null;"
|
|
598
|
+
"$dialog = New-Object Microsoft.Win32.OpenFileDialog;"
|
|
599
|
+
"$dialog.Filter = 'Supported Documents|*.pdf;*.docx;*.doc;*.xlsx;*.xls;*.pptx;*.ppt;*.txt;*.md;*.log;*.csv|All Files (*.*)|*.*';"
|
|
600
|
+
"$dialog.Multiselect = $true;"
|
|
601
|
+
"$dialog.Title = 'Select documents to convert';"
|
|
602
|
+
"if ($dialog.ShowDialog() -eq $true) { $dialog.FileNames }"
|
|
603
|
+
)
|
|
604
|
+
cmd = ["powershell", "-NoProfile", "-NonInteractive", "-Command", ps_script]
|
|
605
|
+
result = subprocess.run(cmd, capture_output=True, text=True)
|
|
606
|
+
if result.returncode == 0 and result.stdout.strip():
|
|
607
|
+
return [line.strip() for line in result.stdout.strip().splitlines() if line.strip()]
|
|
608
|
+
except Exception:
|
|
609
|
+
pass
|
|
610
|
+
|
|
611
|
+
try:
|
|
612
|
+
from tkinter import Tk, filedialog
|
|
613
|
+
|
|
614
|
+
root = Tk()
|
|
615
|
+
root.withdraw()
|
|
616
|
+
try:
|
|
617
|
+
return filedialog.askopenfilenames(title="Select documents to convert")
|
|
618
|
+
finally:
|
|
619
|
+
root.destroy()
|
|
620
|
+
except Exception as exc:
|
|
621
|
+
logger.error("No files were provided and the file dialog is unavailable: %s", exc)
|
|
622
|
+
return ()
|
|
623
|
+
|
|
624
|
+
|
|
625
|
+
def main(argv: Sequence[str] | None = None) -> int:
|
|
626
|
+
parser = argparse.ArgumentParser(
|
|
627
|
+
prog="hammerdown",
|
|
628
|
+
description="Convert PDF and office documents to Markdown",
|
|
629
|
+
)
|
|
630
|
+
parser.add_argument("--version", action="version", version=f"hammerdown {__version__}")
|
|
631
|
+
action = parser.add_mutually_exclusive_group()
|
|
632
|
+
action.add_argument("--install", action="store_true", help="Install the file-manager integration")
|
|
633
|
+
action.add_argument("--uninstall", action="store_true", help="Remove the file-manager integration")
|
|
634
|
+
action.add_argument("--update", action="store_true", help="Update md-maker to the latest version")
|
|
635
|
+
parser.add_argument("--quiet", action="store_true", help="Suppress routine conversion messages")
|
|
636
|
+
parser.add_argument("files", nargs="*", help="Files to convert")
|
|
637
|
+
args = parser.parse_args(argv)
|
|
638
|
+
|
|
639
|
+
if args.quiet:
|
|
640
|
+
logger.setLevel(logging.ERROR)
|
|
641
|
+
if args.install:
|
|
642
|
+
return 0 if install() else 1
|
|
643
|
+
if args.uninstall:
|
|
644
|
+
return 0 if uninstall() else 1
|
|
645
|
+
if args.update:
|
|
646
|
+
return 0 if update() else 1
|
|
647
|
+
|
|
648
|
+
files = args.files or _select_files()
|
|
649
|
+
if not files:
|
|
650
|
+
return 0
|
|
651
|
+
|
|
652
|
+
started_at = time.monotonic()
|
|
653
|
+
results = [convert_file(file_path) for file_path in files]
|
|
654
|
+
failed_count = results.count(False)
|
|
655
|
+
if len(files) > 1 and not args.quiet:
|
|
656
|
+
logger.info(
|
|
657
|
+
"Batch conversion finished in %.2fs: %d succeeded, %d failed",
|
|
658
|
+
time.monotonic() - started_at,
|
|
659
|
+
len(files) - failed_count,
|
|
660
|
+
failed_count,
|
|
661
|
+
)
|
|
662
|
+
return 1 if failed_count else 0
|
|
663
|
+
|
|
664
|
+
|
|
665
|
+
if __name__ == "__main__":
|
|
666
|
+
raise SystemExit(main())
|
|
@@ -0,0 +1,136 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: hammerdown
|
|
3
|
+
Version: 1.1.0
|
|
4
|
+
Summary: Convert documents to Markdown with extracted images
|
|
5
|
+
Requires-Python: >=3.10
|
|
6
|
+
Description-Content-Type: text/markdown
|
|
7
|
+
Requires-Dist: pymupdf
|
|
8
|
+
Requires-Dist: pymupdf4llm
|
|
9
|
+
Requires-Dist: python-docx
|
|
10
|
+
Requires-Dist: openpyxl
|
|
11
|
+
Requires-Dist: python-pptx
|
|
12
|
+
Requires-Dist: onnxruntime
|
|
13
|
+
|
|
14
|
+
# hammerdown
|
|
15
|
+
|
|
16
|
+
[](https://github.com/sergeypugin/hammerdown/actions)
|
|
17
|
+
[](https://pypi.org/project/hammerdown/)
|
|
18
|
+
[](https://github.com/sergeypugin/hammerdown/releases)
|
|
19
|
+
[](https://www.python.org/)
|
|
20
|
+
[](https://github.com/sergeypugin/hammerdown/releases)
|
|
21
|
+
[](CONTRIBUTING.md)
|
|
22
|
+
|
|
23
|
+
`hammerdown` converts PDF, Word, Excel, PowerPoint, and plain-text documents to Markdown. PDF conversion uses PyMuPDF and PyMuPDF4LLM; source images are extracted into the output folder and Markdown image references are generated.
|
|
24
|
+
|
|
25
|
+
## Table of Contents
|
|
26
|
+
|
|
27
|
+
- [Installation](#installation)
|
|
28
|
+
- [Recommended: Python package (pip)](#recommended-python-package-pip)
|
|
29
|
+
- [Direct Download (Standalone binaries)](#direct-download-standalone-binaries)
|
|
30
|
+
- [Context menu integration](#context-menu-integration)
|
|
31
|
+
- [Usage](#usage)
|
|
32
|
+
- [Supported Formats](#supported-formats)
|
|
33
|
+
- [Output Structure](#output-structure)
|
|
34
|
+
- [Contributing](#contributing)
|
|
35
|
+
|
|
36
|
+
## Installation
|
|
37
|
+
|
|
38
|
+
### Recommended: Python package (pip)
|
|
39
|
+
|
|
40
|
+
If Python (3.10+) is installed on your system, installing via pip is the recommended method:
|
|
41
|
+
|
|
42
|
+
```sh
|
|
43
|
+
pip install hammerdown
|
|
44
|
+
```
|
|
45
|
+
|
|
46
|
+
Or after cloning the repository locally:
|
|
47
|
+
|
|
48
|
+
```sh
|
|
49
|
+
git clone https://github.com/sergeypugin/hammerdown.git
|
|
50
|
+
cd hammerdown
|
|
51
|
+
pip install .
|
|
52
|
+
```
|
|
53
|
+
|
|
54
|
+
### Direct Download (Standalone binaries)
|
|
55
|
+
|
|
56
|
+
If you do not have Python installed, precompiled standalone binaries are available on the [GitHub Releases page](https://github.com/sergeypugin/hammerdown/releases) or you can download directly from the links below:
|
|
57
|
+
|
|
58
|
+
| OS | Download |
|
|
59
|
+
| :--- | :--- |
|
|
60
|
+
| **Windows** | [](https://github.com/sergeypugin/hammerdown/releases/latest/download/hammerdown-windows-x64.exe) |
|
|
61
|
+
| **Linux** | [](https://github.com/sergeypugin/hammerdown/releases/latest/download/hammerdown-linux-x64) |
|
|
62
|
+
| **macOS** | [](https://github.com/sergeypugin/hammerdown/releases/latest/download/hammerdown-macos-x64) [](https://github.com/sergeypugin/hammerdown/releases/latest/download/hammerdown-macos-arm64) |
|
|
63
|
+
|
|
64
|
+
On Linux and macOS, make the downloaded binary executable before running:
|
|
65
|
+
|
|
66
|
+
```sh
|
|
67
|
+
chmod +x ./hammerdown-linux-x64
|
|
68
|
+
./hammerdown-linux-x64 --version
|
|
69
|
+
```
|
|
70
|
+
|
|
71
|
+
### Context menu integration
|
|
72
|
+
|
|
73
|
+
Run `--install` to register file manager integrations, or `--uninstall` to remove them:
|
|
74
|
+
|
|
75
|
+
```sh
|
|
76
|
+
hammerdown --install
|
|
77
|
+
hammerdown --uninstall
|
|
78
|
+
```
|
|
79
|
+
|
|
80
|
+
What happens on each platform:
|
|
81
|
+
- windows: moves binary or script into `%LOCALAPPDATA%\Programs\hammerdown\` and adds a "Convert to Markdown" context menu item for PDF files in Windows Explorer
|
|
82
|
+
- linux: creates launcher in `~/.local/bin/hammerdown` and Nautilus script in `~/.local/share/nautilus/scripts/Convert to Markdown`
|
|
83
|
+
- macOS: creates launcher in `~/.local/bin/hammerdown`
|
|
84
|
+
|
|
85
|
+
Update hammerdown to the latest version at any time:
|
|
86
|
+
|
|
87
|
+
```sh
|
|
88
|
+
hammerdown --update
|
|
89
|
+
```
|
|
90
|
+
|
|
91
|
+
This installer is local and does not require administrator privileges.
|
|
92
|
+
|
|
93
|
+
## Usage
|
|
94
|
+
|
|
95
|
+
Convert one or several supported files:
|
|
96
|
+
|
|
97
|
+
```sh
|
|
98
|
+
hammerdown document.pdf
|
|
99
|
+
hammerdown report.docx workbook.xlsx slides.pptx
|
|
100
|
+
```
|
|
101
|
+
|
|
102
|
+
Files can also be dragged onto the executable. Running `hammerdown` without file arguments opens a file-selection dialog when a graphical desktop is available. `--quiet` suppresses routine status messages, which is used by file-manager integrations:
|
|
103
|
+
|
|
104
|
+
```sh
|
|
105
|
+
hammerdown --quiet document.pdf
|
|
106
|
+
hammerdown --version
|
|
107
|
+
```
|
|
108
|
+
|
|
109
|
+
## Supported Formats
|
|
110
|
+
|
|
111
|
+
Supported document extensions:
|
|
112
|
+
- **PDF & E-books**: `.pdf`, `.epub`, `.mobi`, `.fb2`, `.xps`
|
|
113
|
+
- **Word**: `.docx`, `.doc` (legacy `.doc` via LibreOffice / MS Word)
|
|
114
|
+
- **Excel**: `.xlsx`, `.xls` (legacy `.xls` via LibreOffice / MS Excel)
|
|
115
|
+
- **PowerPoint**: `.pptx`, `.ppt` (legacy `.ppt` via LibreOffice / MS PowerPoint)
|
|
116
|
+
- **Plain text & tabular**: `.txt`, `.md`, `.log`, `.csv`
|
|
117
|
+
|
|
118
|
+
## Output Structure
|
|
119
|
+
|
|
120
|
+
For each input file, an output folder `MD_<name>_<ext>` is written next to the source document:
|
|
121
|
+
|
|
122
|
+
```text
|
|
123
|
+
document.pdf
|
|
124
|
+
MD_document_pdf/
|
|
125
|
+
├── document.md
|
|
126
|
+
└── images/
|
|
127
|
+
└── extracted illustration and embedded document images
|
|
128
|
+
```
|
|
129
|
+
|
|
130
|
+
Output folder names consistently include the file extension suffix (for example, `MD_report_doc/` and `MD_report_docx/`) to prevent conflicts between different document formats with the same base name.
|
|
131
|
+
|
|
132
|
+
Images embedded in PDF, Word, or PowerPoint documents are extracted into `MD_<name>_<ext>/images/`, and the generated Markdown contains relative image links. Conversion continues through a batch if an individual file fails; the process exits with status `1` if any input fails.
|
|
133
|
+
|
|
134
|
+
## Contributing
|
|
135
|
+
|
|
136
|
+
Contributions are welcome. Please refer to [CONTRIBUTING.md](CONTRIBUTING.md) for contribution guidelines. For architectural details, processing pipelines, and internal specifications, see [TECHNICAL.md](TECHNICAL.md).
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
README.md
|
|
2
|
+
pyproject.toml
|
|
3
|
+
src/converter.py
|
|
4
|
+
src/hammerdown.egg-info/PKG-INFO
|
|
5
|
+
src/hammerdown.egg-info/SOURCES.txt
|
|
6
|
+
src/hammerdown.egg-info/dependency_links.txt
|
|
7
|
+
src/hammerdown.egg-info/entry_points.txt
|
|
8
|
+
src/hammerdown.egg-info/requires.txt
|
|
9
|
+
src/hammerdown.egg-info/top_level.txt
|
|
10
|
+
tests/test_cli.py
|
|
11
|
+
tests/test_golden.py
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
converter
|
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
from pathlib import Path
|
|
2
|
+
|
|
3
|
+
import converter
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
def test_normalize_wsl_windows_path():
|
|
7
|
+
normalized = converter.normalize_path(r"C:\Users\someone\document.pdf")
|
|
8
|
+
if converter.os.name == "nt":
|
|
9
|
+
assert normalized.endswith(r"C:\Users\someone\document.pdf")
|
|
10
|
+
else:
|
|
11
|
+
assert normalized.endswith("/mnt/c/Users/someone/document.pdf")
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def test_convert_plain_text_creates_markdown_output(tmp_path: Path):
|
|
15
|
+
source = tmp_path / "notes.txt"
|
|
16
|
+
source.write_text("line one\nline two\n", encoding="utf-8")
|
|
17
|
+
|
|
18
|
+
assert converter.convert_file(source)
|
|
19
|
+
assert (tmp_path / "MD_notes_txt" / "notes.md").read_text(encoding="utf-8") == "line one\nline two\n"
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def test_unsupported_file_returns_failure(tmp_path: Path):
|
|
23
|
+
source = tmp_path / "image.png"
|
|
24
|
+
source.write_bytes(b"image")
|
|
25
|
+
|
|
26
|
+
assert converter.main([str(source)]) == 1
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def test_missing_file_returns_failure(tmp_path: Path):
|
|
30
|
+
assert converter.main([str(tmp_path / "missing.pdf")]) == 1
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def test_version_option(capsys):
|
|
34
|
+
try:
|
|
35
|
+
converter.main(["--version"])
|
|
36
|
+
except SystemExit as error:
|
|
37
|
+
assert error.code == 0
|
|
38
|
+
assert f"hammerdown {converter.__version__}" in capsys.readouterr().out
|
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
import glob
|
|
2
|
+
import os
|
|
3
|
+
import subprocess
|
|
4
|
+
import sys
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
def test_golden_conversion():
|
|
9
|
+
inputs_dir = Path("tests/inputs")
|
|
10
|
+
golden_dir = Path("tests/golden")
|
|
11
|
+
|
|
12
|
+
test_files = [
|
|
13
|
+
f for f in inputs_dir.glob("*.*")
|
|
14
|
+
if f.suffix.lower() in [".pdf", ".docx", ".doc", ".xlsx", ".xls", ".pptx", ".ppt", ".csv", ".txt", ".epub", ".fb2"]
|
|
15
|
+
]
|
|
16
|
+
assert len(test_files) > 0, "No test input files found"
|
|
17
|
+
|
|
18
|
+
for file_path in test_files:
|
|
19
|
+
stem = file_path.stem
|
|
20
|
+
ext_clean = file_path.suffix.lower().lstrip(".")
|
|
21
|
+
print(f"Testing golden output for: {file_path.name}")
|
|
22
|
+
|
|
23
|
+
cmd = [sys.executable, "src/converter.py", str(file_path)]
|
|
24
|
+
result = subprocess.run(cmd, capture_output=True, text=True, stdin=subprocess.DEVNULL)
|
|
25
|
+
assert result.returncode == 0, f"Converter failed for {file_path.name}: {result.stderr}"
|
|
26
|
+
|
|
27
|
+
folder_name = f"MD_{stem}_{ext_clean}" if ext_clean else f"MD_{stem}"
|
|
28
|
+
out_md_path = inputs_dir / folder_name / f"{stem}.md"
|
|
29
|
+
assert out_md_path.is_file(), f"Expected output MD file not found: {out_md_path}"
|
|
30
|
+
|
|
31
|
+
generated_content = out_md_path.read_text(encoding="utf-8")
|
|
32
|
+
golden_file_path = golden_dir / f"{stem}.md"
|
|
33
|
+
|
|
34
|
+
if not golden_file_path.is_file():
|
|
35
|
+
golden_dir.mkdir(parents=True, exist_ok=True)
|
|
36
|
+
golden_file_path.write_text(generated_content, encoding="utf-8")
|
|
37
|
+
print(f"Created new golden reference for {stem}")
|
|
38
|
+
continue
|
|
39
|
+
|
|
40
|
+
golden_content = golden_file_path.read_text(encoding="utf-8")
|
|
41
|
+
assert (
|
|
42
|
+
generated_content == golden_content
|
|
43
|
+
), f"Golden test failed for {file_path.name}! Output does not match reference."
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
if __name__ == "__main__":
|
|
47
|
+
try:
|
|
48
|
+
test_golden_conversion()
|
|
49
|
+
print("\nSUCCESS: All golden tests passed!")
|
|
50
|
+
except AssertionError as e:
|
|
51
|
+
print(f"\nFAILURE: {e}")
|
|
52
|
+
sys.exit(1)
|