bruce-doc-converter 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- bruce_doc_converter-0.1.0/MANIFEST.in +4 -0
- bruce_doc_converter-0.1.0/PKG-INFO +189 -0
- bruce_doc_converter-0.1.0/README.md +176 -0
- bruce_doc_converter-0.1.0/bruce_doc_converter/__init__.py +3 -0
- bruce_doc_converter-0.1.0/bruce_doc_converter/cli.py +206 -0
- bruce_doc_converter-0.1.0/bruce_doc_converter/converter.py +2853 -0
- bruce_doc_converter-0.1.0/bruce_doc_converter/md_to_docx/html-converter.js +882 -0
- bruce_doc_converter-0.1.0/bruce_doc_converter/md_to_docx/index.js +113 -0
- bruce_doc_converter-0.1.0/bruce_doc_converter/md_to_docx/markdown-converter.js +620 -0
- bruce_doc_converter-0.1.0/bruce_doc_converter/md_to_docx/mermaid-renderer.js +288 -0
- bruce_doc_converter-0.1.0/bruce_doc_converter/md_to_docx/package-lock.json +5033 -0
- bruce_doc_converter-0.1.0/bruce_doc_converter/md_to_docx/package.json +15 -0
- bruce_doc_converter-0.1.0/bruce_doc_converter/md_to_docx/styles.js +447 -0
- bruce_doc_converter-0.1.0/bruce_doc_converter.egg-info/PKG-INFO +189 -0
- bruce_doc_converter-0.1.0/bruce_doc_converter.egg-info/SOURCES.txt +21 -0
- bruce_doc_converter-0.1.0/bruce_doc_converter.egg-info/dependency_links.txt +1 -0
- bruce_doc_converter-0.1.0/bruce_doc_converter.egg-info/entry_points.txt +3 -0
- bruce_doc_converter-0.1.0/bruce_doc_converter.egg-info/requires.txt +4 -0
- bruce_doc_converter-0.1.0/bruce_doc_converter.egg-info/top_level.txt +1 -0
- bruce_doc_converter-0.1.0/pyproject.toml +32 -0
- bruce_doc_converter-0.1.0/setup.cfg +4 -0
- bruce_doc_converter-0.1.0/tests/test_cli.py +206 -0
- bruce_doc_converter-0.1.0/tests/test_convert_document.py +716 -0
|
@@ -0,0 +1,189 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: bruce-doc-converter
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Agent-facing document converter CLI for Office/PDF/Markdown workflows
|
|
5
|
+
Author: Bruce
|
|
6
|
+
License: MIT
|
|
7
|
+
Requires-Python: >=3.8
|
|
8
|
+
Description-Content-Type: text/markdown
|
|
9
|
+
Requires-Dist: python-docx
|
|
10
|
+
Requires-Dist: openpyxl
|
|
11
|
+
Requires-Dist: python-pptx
|
|
12
|
+
Requires-Dist: pdfplumber
|
|
13
|
+
|
|
14
|
+
# Bruce Doc Converter
|
|
15
|
+
|
|
16
|
+
> 为 Claude Code / OpenClaw 添加双向文档转换能力
|
|
17
|
+
|
|
18
|
+
[](https://github.com/anthropics/claude-code)
|
|
19
|
+
[](https://www.python.org/downloads/)
|
|
20
|
+
[](LICENSE)
|
|
21
|
+
|
|
22
|
+
**Bruce Doc Converter** 是一个面向 Agent 的文档转换 CLI,为 **Claude Code / OpenClaw** 添加**双向文档转换**能力:
|
|
23
|
+
|
|
24
|
+
- **Office/PDF → Markdown**:将 Word、Excel、PowerPoint、PDF 转换为 AI 友好的 Markdown 格式
|
|
25
|
+
- **Markdown → Word**:将 Markdown 导出为排版精美的 Word 文档,自动渲染 Mermaid 图表
|
|
26
|
+
|
|
27
|
+
## 安装
|
|
28
|
+
|
|
29
|
+
```bash
|
|
30
|
+
pipx install bruce-doc-converter
|
|
31
|
+
```
|
|
32
|
+
|
|
33
|
+
如果 `pipx` 不可用:
|
|
34
|
+
|
|
35
|
+
```bash
|
|
36
|
+
python3 -m pip install bruce-doc-converter
|
|
37
|
+
# macOS Homebrew Python 需要加 --break-system-packages 或使用 venv:
|
|
38
|
+
# python3 -m venv .venv && .venv/bin/pip install bruce-doc-converter
|
|
39
|
+
```
|
|
40
|
+
|
|
41
|
+
## Agent CLI 用法
|
|
42
|
+
|
|
43
|
+
```bash
|
|
44
|
+
bdc convert /path/to/document.docx
|
|
45
|
+
bdc convert /path/to/notes.md
|
|
46
|
+
bdc batch /path/to/documents
|
|
47
|
+
```
|
|
48
|
+
|
|
49
|
+
CLI 默认向 stdout 输出 JSON,stderr 仅用于进度和依赖安装日志。
|
|
50
|
+
|
|
51
|
+
查看帮助:
|
|
52
|
+
|
|
53
|
+
```bash
|
|
54
|
+
bdc --help-json
|
|
55
|
+
```
|
|
56
|
+
|
|
57
|
+
### 输出示例(单文件成功)
|
|
58
|
+
|
|
59
|
+
```json
|
|
60
|
+
{
|
|
61
|
+
"schema_version": "1.0",
|
|
62
|
+
"success": true,
|
|
63
|
+
"input_path": "/absolute/input.docx",
|
|
64
|
+
"input_format": "docx",
|
|
65
|
+
"output_format": "markdown",
|
|
66
|
+
"output_path": "/absolute/Markdown/input.md",
|
|
67
|
+
"markdown_content": "# 内容...",
|
|
68
|
+
"extracted_images": [],
|
|
69
|
+
"warnings": []
|
|
70
|
+
}
|
|
71
|
+
```
|
|
72
|
+
|
|
73
|
+
### 输出示例(失败)
|
|
74
|
+
|
|
75
|
+
```json
|
|
76
|
+
{
|
|
77
|
+
"schema_version": "1.0",
|
|
78
|
+
"success": false,
|
|
79
|
+
"input_path": "/absolute/input.doc",
|
|
80
|
+
"input_format": "doc",
|
|
81
|
+
"error_code": "UNSUPPORTED_FORMAT",
|
|
82
|
+
"error": "不支持的文件格式: .doc。支持的格式: .docx, .xlsx, .pptx, .pdf, .md",
|
|
83
|
+
"suggestion": "请先转换为 .docx/.xlsx/.pptx 后再重试。"
|
|
84
|
+
}
|
|
85
|
+
```
|
|
86
|
+
|
|
87
|
+
### 输出示例(批量转换)
|
|
88
|
+
|
|
89
|
+
批量转换的 `success` 表示是否所有文件都转换成功;部分失败时 `success` 为 `false`,但 `succeeded`、`failed` 和 `results` 会保留每个文件的明细。
|
|
90
|
+
|
|
91
|
+
```json
|
|
92
|
+
{
|
|
93
|
+
"schema_version": "1.0",
|
|
94
|
+
"success": true,
|
|
95
|
+
"total": 1,
|
|
96
|
+
"succeeded": 1,
|
|
97
|
+
"failed": 0,
|
|
98
|
+
"results": [
|
|
99
|
+
{
|
|
100
|
+
"input_path": "/absolute/input.docx",
|
|
101
|
+
"result": {
|
|
102
|
+
"schema_version": "1.0",
|
|
103
|
+
"success": true,
|
|
104
|
+
"input_path": "/absolute/input.docx",
|
|
105
|
+
"input_format": "docx",
|
|
106
|
+
"output_format": "markdown",
|
|
107
|
+
"output_path": "/absolute/Markdown/input.md",
|
|
108
|
+
"markdown_content": "# 内容...",
|
|
109
|
+
"extracted_images": [],
|
|
110
|
+
"warnings": []
|
|
111
|
+
}
|
|
112
|
+
}
|
|
113
|
+
]
|
|
114
|
+
}
|
|
115
|
+
```
|
|
116
|
+
|
|
117
|
+
## 功能特性
|
|
118
|
+
|
|
119
|
+
- **标题识别**:自动识别 Word 标题层级(Heading 1-6)及中文标题样式
|
|
120
|
+
- **格式保留**:保留粗体、斜体等文本格式
|
|
121
|
+
- **表格转换**:智能转换表格为 Markdown 格式
|
|
122
|
+
- **列表支持**:有序列表、无序列表及多级嵌套
|
|
123
|
+
- **Mermaid 图表**:支持通过 `mmdc` 渲染 Mermaid 代码块,嵌入 Word 为 PNG 图片
|
|
124
|
+
- **图片提取**:Office/PDF 转 Markdown 时可提取内嵌图片
|
|
125
|
+
|
|
126
|
+
## 支持的格式
|
|
127
|
+
|
|
128
|
+
| 格式 | 输入 | 输出 | 质量 |
|
|
129
|
+
| ------------------ | ---- | ---- | ---------- |
|
|
130
|
+
| Word (.docx) | ✅ | ✅ | 优秀 |
|
|
131
|
+
| Excel (.xlsx) | ✅ | ❌ | 优秀 |
|
|
132
|
+
| PowerPoint (.pptx) | ✅ | ❌ | 良好 |
|
|
133
|
+
| PDF (.pdf) | ✅ | ❌ | 取决于类型 |
|
|
134
|
+
| Markdown (.md) | ✅ | ✅ | 优秀 |
|
|
135
|
+
|
|
136
|
+
> **注意**:不支持旧版格式(.doc, .xls, .ppt),请先转换为新格式。
|
|
137
|
+
|
|
138
|
+
## 环境要求
|
|
139
|
+
|
|
140
|
+
- **Python 3.8+**(必需)
|
|
141
|
+
- **Node.js 14+**(可选,仅 Markdown → Word 需要)
|
|
142
|
+
|
|
143
|
+
## 常见问题
|
|
144
|
+
|
|
145
|
+
### 文件过大怎么办?
|
|
146
|
+
|
|
147
|
+
当前限制为 100MB,建议分割文件或压缩内容。
|
|
148
|
+
|
|
149
|
+
### Markdown 转 Word 失败?
|
|
150
|
+
|
|
151
|
+
需要安装 Node.js。如果 Node.js 已安装但仍报错,检查依赖:
|
|
152
|
+
|
|
153
|
+
```bash
|
|
154
|
+
npm --prefix bruce_doc_converter/md_to_docx install
|
|
155
|
+
```
|
|
156
|
+
|
|
157
|
+
### PDF 提取不到内容?
|
|
158
|
+
|
|
159
|
+
扫描型 PDF 需先执行 OCR,或解除 PDF 保护后重试。
|
|
160
|
+
|
|
161
|
+
## 最佳实践
|
|
162
|
+
|
|
163
|
+
1. **使用新版 Office 格式**(.docx, .xlsx, .pptx)
|
|
164
|
+
2. **PDF 优先使用文本型**,扫描型建议先 OCR
|
|
165
|
+
3. **文件大小建议 < 50MB**
|
|
166
|
+
|
|
167
|
+
## 项目结构
|
|
168
|
+
|
|
169
|
+
```
|
|
170
|
+
bruce-doc-converter/
|
|
171
|
+
├── SKILL.md # Agent Skill 定义
|
|
172
|
+
├── pyproject.toml # Python 包元数据
|
|
173
|
+
├── requirements.txt # 本地开发依赖
|
|
174
|
+
├── bruce_doc_converter/
|
|
175
|
+
│ ├── __init__.py
|
|
176
|
+
│ ├── cli.py # bdc CLI 入口
|
|
177
|
+
│ ├── converter.py # 转换核心逻辑
|
|
178
|
+
│ └── md_to_docx/ # Markdown → Word 的 Node.js 模块
|
|
179
|
+
├── references/
|
|
180
|
+
│ └── supported-formats.md
|
|
181
|
+
└── tests/
|
|
182
|
+
├── test_cli.py
|
|
183
|
+
├── test_convert_document.py
|
|
184
|
+
└── md_to_docx.test.js
|
|
185
|
+
```
|
|
186
|
+
|
|
187
|
+
## 许可证
|
|
188
|
+
|
|
189
|
+
MIT License
|
|
@@ -0,0 +1,176 @@
|
|
|
1
|
+
# Bruce Doc Converter
|
|
2
|
+
|
|
3
|
+
> 为 Claude Code / OpenClaw 添加双向文档转换能力
|
|
4
|
+
|
|
5
|
+
[](https://github.com/anthropics/claude-code)
|
|
6
|
+
[](https://www.python.org/downloads/)
|
|
7
|
+
[](LICENSE)
|
|
8
|
+
|
|
9
|
+
**Bruce Doc Converter** 是一个面向 Agent 的文档转换 CLI,为 **Claude Code / OpenClaw** 添加**双向文档转换**能力:
|
|
10
|
+
|
|
11
|
+
- **Office/PDF → Markdown**:将 Word、Excel、PowerPoint、PDF 转换为 AI 友好的 Markdown 格式
|
|
12
|
+
- **Markdown → Word**:将 Markdown 导出为排版精美的 Word 文档,自动渲染 Mermaid 图表
|
|
13
|
+
|
|
14
|
+
## 安装
|
|
15
|
+
|
|
16
|
+
```bash
|
|
17
|
+
pipx install bruce-doc-converter
|
|
18
|
+
```
|
|
19
|
+
|
|
20
|
+
如果 `pipx` 不可用:
|
|
21
|
+
|
|
22
|
+
```bash
|
|
23
|
+
python3 -m pip install bruce-doc-converter
|
|
24
|
+
# macOS Homebrew Python 需要加 --break-system-packages 或使用 venv:
|
|
25
|
+
# python3 -m venv .venv && .venv/bin/pip install bruce-doc-converter
|
|
26
|
+
```
|
|
27
|
+
|
|
28
|
+
## Agent CLI 用法
|
|
29
|
+
|
|
30
|
+
```bash
|
|
31
|
+
bdc convert /path/to/document.docx
|
|
32
|
+
bdc convert /path/to/notes.md
|
|
33
|
+
bdc batch /path/to/documents
|
|
34
|
+
```
|
|
35
|
+
|
|
36
|
+
CLI 默认向 stdout 输出 JSON,stderr 仅用于进度和依赖安装日志。
|
|
37
|
+
|
|
38
|
+
查看帮助:
|
|
39
|
+
|
|
40
|
+
```bash
|
|
41
|
+
bdc --help-json
|
|
42
|
+
```
|
|
43
|
+
|
|
44
|
+
### 输出示例(单文件成功)
|
|
45
|
+
|
|
46
|
+
```json
|
|
47
|
+
{
|
|
48
|
+
"schema_version": "1.0",
|
|
49
|
+
"success": true,
|
|
50
|
+
"input_path": "/absolute/input.docx",
|
|
51
|
+
"input_format": "docx",
|
|
52
|
+
"output_format": "markdown",
|
|
53
|
+
"output_path": "/absolute/Markdown/input.md",
|
|
54
|
+
"markdown_content": "# 内容...",
|
|
55
|
+
"extracted_images": [],
|
|
56
|
+
"warnings": []
|
|
57
|
+
}
|
|
58
|
+
```
|
|
59
|
+
|
|
60
|
+
### 输出示例(失败)
|
|
61
|
+
|
|
62
|
+
```json
|
|
63
|
+
{
|
|
64
|
+
"schema_version": "1.0",
|
|
65
|
+
"success": false,
|
|
66
|
+
"input_path": "/absolute/input.doc",
|
|
67
|
+
"input_format": "doc",
|
|
68
|
+
"error_code": "UNSUPPORTED_FORMAT",
|
|
69
|
+
"error": "不支持的文件格式: .doc。支持的格式: .docx, .xlsx, .pptx, .pdf, .md",
|
|
70
|
+
"suggestion": "请先转换为 .docx/.xlsx/.pptx 后再重试。"
|
|
71
|
+
}
|
|
72
|
+
```
|
|
73
|
+
|
|
74
|
+
### 输出示例(批量转换)
|
|
75
|
+
|
|
76
|
+
批量转换的 `success` 表示是否所有文件都转换成功;部分失败时 `success` 为 `false`,但 `succeeded`、`failed` 和 `results` 会保留每个文件的明细。
|
|
77
|
+
|
|
78
|
+
```json
|
|
79
|
+
{
|
|
80
|
+
"schema_version": "1.0",
|
|
81
|
+
"success": true,
|
|
82
|
+
"total": 1,
|
|
83
|
+
"succeeded": 1,
|
|
84
|
+
"failed": 0,
|
|
85
|
+
"results": [
|
|
86
|
+
{
|
|
87
|
+
"input_path": "/absolute/input.docx",
|
|
88
|
+
"result": {
|
|
89
|
+
"schema_version": "1.0",
|
|
90
|
+
"success": true,
|
|
91
|
+
"input_path": "/absolute/input.docx",
|
|
92
|
+
"input_format": "docx",
|
|
93
|
+
"output_format": "markdown",
|
|
94
|
+
"output_path": "/absolute/Markdown/input.md",
|
|
95
|
+
"markdown_content": "# 内容...",
|
|
96
|
+
"extracted_images": [],
|
|
97
|
+
"warnings": []
|
|
98
|
+
}
|
|
99
|
+
}
|
|
100
|
+
]
|
|
101
|
+
}
|
|
102
|
+
```
|
|
103
|
+
|
|
104
|
+
## 功能特性
|
|
105
|
+
|
|
106
|
+
- **标题识别**:自动识别 Word 标题层级(Heading 1-6)及中文标题样式
|
|
107
|
+
- **格式保留**:保留粗体、斜体等文本格式
|
|
108
|
+
- **表格转换**:智能转换表格为 Markdown 格式
|
|
109
|
+
- **列表支持**:有序列表、无序列表及多级嵌套
|
|
110
|
+
- **Mermaid 图表**:支持通过 `mmdc` 渲染 Mermaid 代码块,嵌入 Word 为 PNG 图片
|
|
111
|
+
- **图片提取**:Office/PDF 转 Markdown 时可提取内嵌图片
|
|
112
|
+
|
|
113
|
+
## 支持的格式
|
|
114
|
+
|
|
115
|
+
| 格式 | 输入 | 输出 | 质量 |
|
|
116
|
+
| ------------------ | ---- | ---- | ---------- |
|
|
117
|
+
| Word (.docx) | ✅ | ✅ | 优秀 |
|
|
118
|
+
| Excel (.xlsx) | ✅ | ❌ | 优秀 |
|
|
119
|
+
| PowerPoint (.pptx) | ✅ | ❌ | 良好 |
|
|
120
|
+
| PDF (.pdf) | ✅ | ❌ | 取决于类型 |
|
|
121
|
+
| Markdown (.md) | ✅ | ✅ | 优秀 |
|
|
122
|
+
|
|
123
|
+
> **注意**:不支持旧版格式(.doc, .xls, .ppt),请先转换为新格式。
|
|
124
|
+
|
|
125
|
+
## 环境要求
|
|
126
|
+
|
|
127
|
+
- **Python 3.8+**(必需)
|
|
128
|
+
- **Node.js 14+**(可选,仅 Markdown → Word 需要)
|
|
129
|
+
|
|
130
|
+
## 常见问题
|
|
131
|
+
|
|
132
|
+
### 文件过大怎么办?
|
|
133
|
+
|
|
134
|
+
当前限制为 100MB,建议分割文件或压缩内容。
|
|
135
|
+
|
|
136
|
+
### Markdown 转 Word 失败?
|
|
137
|
+
|
|
138
|
+
需要安装 Node.js。如果 Node.js 已安装但仍报错,检查依赖:
|
|
139
|
+
|
|
140
|
+
```bash
|
|
141
|
+
npm --prefix bruce_doc_converter/md_to_docx install
|
|
142
|
+
```
|
|
143
|
+
|
|
144
|
+
### PDF 提取不到内容?
|
|
145
|
+
|
|
146
|
+
扫描型 PDF 需先执行 OCR,或解除 PDF 保护后重试。
|
|
147
|
+
|
|
148
|
+
## 最佳实践
|
|
149
|
+
|
|
150
|
+
1. **使用新版 Office 格式**(.docx, .xlsx, .pptx)
|
|
151
|
+
2. **PDF 优先使用文本型**,扫描型建议先 OCR
|
|
152
|
+
3. **文件大小建议 < 50MB**
|
|
153
|
+
|
|
154
|
+
## 项目结构
|
|
155
|
+
|
|
156
|
+
```
|
|
157
|
+
bruce-doc-converter/
|
|
158
|
+
├── SKILL.md # Agent Skill 定义
|
|
159
|
+
├── pyproject.toml # Python 包元数据
|
|
160
|
+
├── requirements.txt # 本地开发依赖
|
|
161
|
+
├── bruce_doc_converter/
|
|
162
|
+
│ ├── __init__.py
|
|
163
|
+
│ ├── cli.py # bdc CLI 入口
|
|
164
|
+
│ ├── converter.py # 转换核心逻辑
|
|
165
|
+
│ └── md_to_docx/ # Markdown → Word 的 Node.js 模块
|
|
166
|
+
├── references/
|
|
167
|
+
│ └── supported-formats.md
|
|
168
|
+
└── tests/
|
|
169
|
+
├── test_cli.py
|
|
170
|
+
├── test_convert_document.py
|
|
171
|
+
└── md_to_docx.test.js
|
|
172
|
+
```
|
|
173
|
+
|
|
174
|
+
## 许可证
|
|
175
|
+
|
|
176
|
+
MIT License
|
|
@@ -0,0 +1,206 @@
|
|
|
1
|
+
import argparse
|
|
2
|
+
import json
|
|
3
|
+
import os
|
|
4
|
+
import sys
|
|
5
|
+
|
|
6
|
+
from bruce_doc_converter.converter import SUPPORTED_EXTENSIONS, batch_convert, convert_document
|
|
7
|
+
|
|
8
|
+
SCHEMA_VERSION = "1.0"
|
|
9
|
+
|
|
10
|
+
SUGGESTIONS = {
|
|
11
|
+
"UNSUPPORTED_FORMAT": "请先转换为 .docx/.xlsx/.pptx 后再重试。",
|
|
12
|
+
"NODE_NOT_FOUND": "请安装 Node.js 后重试 Markdown 到 Word 转换。",
|
|
13
|
+
"EMPTY_PDF_CONTENT": "请先对扫描件执行 OCR,或解除 PDF 保护后重试。",
|
|
14
|
+
}
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def _emit(payload, exit_code):
|
|
18
|
+
print(json.dumps(payload, ensure_ascii=False, indent=2))
|
|
19
|
+
return exit_code
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def _usage_error(message):
|
|
23
|
+
return {
|
|
24
|
+
"schema_version": SCHEMA_VERSION,
|
|
25
|
+
"success": False,
|
|
26
|
+
"error_code": "USAGE_ERROR",
|
|
27
|
+
"error": message,
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
class _JsonArgumentParser(argparse.ArgumentParser):
|
|
32
|
+
"""ArgumentParser that emits JSON on error instead of human-readable text."""
|
|
33
|
+
|
|
34
|
+
def error(self, message):
|
|
35
|
+
_emit(_usage_error(message), 1)
|
|
36
|
+
sys.exit(1)
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def _format_of(path):
|
|
40
|
+
ext = os.path.splitext(str(path))[1].lower()
|
|
41
|
+
if not ext:
|
|
42
|
+
return "unknown"
|
|
43
|
+
return ext[1:] if ext.startswith(".") else ext
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def _output_format(input_format):
|
|
47
|
+
return "docx" if input_format == "md" else "markdown"
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def _classify_error(error):
|
|
51
|
+
text = error or ""
|
|
52
|
+
if "文件不存在" in text or "目录不存在" in text or "文件未找到" in text:
|
|
53
|
+
return "FILE_NOT_FOUND"
|
|
54
|
+
if "输入路径不是文件" in text:
|
|
55
|
+
return "NOT_A_FILE"
|
|
56
|
+
if "输入路径不是目录" in text or "输出路径不是目录" in text:
|
|
57
|
+
return "NOT_A_DIRECTORY"
|
|
58
|
+
if "文件过大" in text:
|
|
59
|
+
return "FILE_TOO_LARGE"
|
|
60
|
+
if "不支持的文件格式" in text or "不支持的文件类型" in text:
|
|
61
|
+
return "UNSUPPORTED_FORMAT"
|
|
62
|
+
if "未找到 Node.js" in text:
|
|
63
|
+
return "NODE_NOT_FOUND"
|
|
64
|
+
if "Node.js 依赖安装失败" in text or "依赖安装失败" in text:
|
|
65
|
+
return "DEPENDENCY_INSTALL_FAILED"
|
|
66
|
+
if "Node.js 脚本输出解析失败" in text or "调用 Node.js 脚本失败" in text:
|
|
67
|
+
return "NODE_CONVERSION_FAILED"
|
|
68
|
+
if "转换超时" in text:
|
|
69
|
+
return "CONVERSION_TIMEOUT"
|
|
70
|
+
if "PDF 未提取到任何文本或表格" in text:
|
|
71
|
+
return "EMPTY_PDF_CONTENT"
|
|
72
|
+
if "权限不足" in text:
|
|
73
|
+
return "PERMISSION_DENIED"
|
|
74
|
+
if "内存不足" in text:
|
|
75
|
+
return "OUT_OF_MEMORY"
|
|
76
|
+
if "系统错误" in text:
|
|
77
|
+
return "OS_ERROR"
|
|
78
|
+
return "CONVERSION_ERROR"
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def _normalize_single_result(input_path, result):
|
|
82
|
+
normalized_input = os.path.realpath(os.path.expanduser(str(input_path)))
|
|
83
|
+
input_format = _format_of(normalized_input)
|
|
84
|
+
|
|
85
|
+
if result.get("success"):
|
|
86
|
+
raw_output = result.get("output_path")
|
|
87
|
+
output_path = os.path.realpath(raw_output) if raw_output else None
|
|
88
|
+
payload = {
|
|
89
|
+
"schema_version": SCHEMA_VERSION,
|
|
90
|
+
"success": True,
|
|
91
|
+
"input_path": normalized_input,
|
|
92
|
+
"input_format": input_format,
|
|
93
|
+
"output_format": _output_format(input_format),
|
|
94
|
+
"output_path": output_path,
|
|
95
|
+
"warnings": [],
|
|
96
|
+
}
|
|
97
|
+
if input_format == "md":
|
|
98
|
+
payload["message"] = result.get("message", "")
|
|
99
|
+
else:
|
|
100
|
+
payload["markdown_content"] = result.get("markdown_content", "")
|
|
101
|
+
payload["extracted_images"] = result.get("extracted_images", [])
|
|
102
|
+
if result.get("warning"):
|
|
103
|
+
payload["warnings"].append(result["warning"])
|
|
104
|
+
return payload
|
|
105
|
+
|
|
106
|
+
error = result.get("error", "转换失败")
|
|
107
|
+
error_code = result.get("error_code") or _classify_error(error)
|
|
108
|
+
payload = {
|
|
109
|
+
"schema_version": SCHEMA_VERSION,
|
|
110
|
+
"success": False,
|
|
111
|
+
"input_path": normalized_input,
|
|
112
|
+
"input_format": input_format,
|
|
113
|
+
"error_code": error_code,
|
|
114
|
+
"error": error,
|
|
115
|
+
}
|
|
116
|
+
if error_code in SUGGESTIONS:
|
|
117
|
+
payload["suggestion"] = SUGGESTIONS[error_code]
|
|
118
|
+
return payload
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
def _help_payload():
|
|
122
|
+
return {
|
|
123
|
+
"schema_version": SCHEMA_VERSION,
|
|
124
|
+
"success": True,
|
|
125
|
+
"commands": {
|
|
126
|
+
"convert": "Convert one .docx/.xlsx/.pptx/.pdf file to Markdown, or one .md file to DOCX.",
|
|
127
|
+
"batch": "Convert supported files in a directory.",
|
|
128
|
+
},
|
|
129
|
+
"supported_extensions": SUPPORTED_EXTENSIONS,
|
|
130
|
+
}
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
def _build_parser():
|
|
134
|
+
# add_help=False: -h would print human text, breaking the JSON-only stdout contract.
|
|
135
|
+
# Use --help-json for machine-readable help instead.
|
|
136
|
+
parser = _JsonArgumentParser(prog="bdc", add_help=False)
|
|
137
|
+
parser.add_argument("--help-json", action="store_true")
|
|
138
|
+
subparsers = parser.add_subparsers(dest="command")
|
|
139
|
+
|
|
140
|
+
# Subparsers inherit _JsonArgumentParser because type(parser) is _JsonArgumentParser
|
|
141
|
+
convert_parser = subparsers.add_parser("convert", add_help=False)
|
|
142
|
+
convert_parser.add_argument("file")
|
|
143
|
+
convert_parser.add_argument("--output-dir")
|
|
144
|
+
convert_parser.add_argument("--extract-images", choices=["true", "false"], default="true")
|
|
145
|
+
|
|
146
|
+
batch_parser = subparsers.add_parser("batch", add_help=False)
|
|
147
|
+
batch_parser.add_argument("directory")
|
|
148
|
+
batch_parser.add_argument("--output-dir")
|
|
149
|
+
batch_parser.add_argument("--recursive", choices=["true", "false"], default="true")
|
|
150
|
+
batch_parser.add_argument("--extract-images", choices=["true", "false"], default="true")
|
|
151
|
+
return parser
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
def main(argv=None):
|
|
155
|
+
args = list(sys.argv[1:] if argv is None else argv)
|
|
156
|
+
if not args:
|
|
157
|
+
return _emit(_usage_error("缺少命令。可用命令: convert, batch"), 1)
|
|
158
|
+
|
|
159
|
+
parser = _build_parser()
|
|
160
|
+
namespace = parser.parse_args(args)
|
|
161
|
+
|
|
162
|
+
if namespace.help_json:
|
|
163
|
+
return _emit(_help_payload(), 0)
|
|
164
|
+
|
|
165
|
+
if namespace.command == "convert":
|
|
166
|
+
output_dir = os.path.realpath(os.path.expanduser(namespace.output_dir)) if namespace.output_dir else None
|
|
167
|
+
result = convert_document(
|
|
168
|
+
namespace.file,
|
|
169
|
+
extract_images=namespace.extract_images == "true",
|
|
170
|
+
output_dir=output_dir,
|
|
171
|
+
)
|
|
172
|
+
payload = _normalize_single_result(namespace.file, result)
|
|
173
|
+
return _emit(payload, 0 if payload["success"] else 1)
|
|
174
|
+
|
|
175
|
+
if namespace.command == "batch":
|
|
176
|
+
output_dir = os.path.realpath(os.path.expanduser(namespace.output_dir)) if namespace.output_dir else None
|
|
177
|
+
raw_results = batch_convert(
|
|
178
|
+
namespace.directory,
|
|
179
|
+
recursive=namespace.recursive == "true",
|
|
180
|
+
extract_images=namespace.extract_images == "true",
|
|
181
|
+
output_dir=output_dir,
|
|
182
|
+
)
|
|
183
|
+
results = []
|
|
184
|
+
for item in raw_results:
|
|
185
|
+
result_payload = _normalize_single_result(item["file"], item["result"])
|
|
186
|
+
results.append({
|
|
187
|
+
"input_path": result_payload["input_path"],
|
|
188
|
+
"result": result_payload,
|
|
189
|
+
})
|
|
190
|
+
succeeded = sum(1 for item in results if item["result"]["success"])
|
|
191
|
+
total = len(results)
|
|
192
|
+
payload = {
|
|
193
|
+
"schema_version": SCHEMA_VERSION,
|
|
194
|
+
"success": succeeded == total,
|
|
195
|
+
"total": total,
|
|
196
|
+
"succeeded": succeeded,
|
|
197
|
+
"failed": total - succeeded,
|
|
198
|
+
"results": results,
|
|
199
|
+
}
|
|
200
|
+
return _emit(payload, 0 if payload["success"] else 1)
|
|
201
|
+
|
|
202
|
+
return _emit(_usage_error("缺少命令。可用命令: convert, batch"), 1)
|
|
203
|
+
|
|
204
|
+
|
|
205
|
+
if __name__ == "__main__":
|
|
206
|
+
sys.exit(main())
|