docpilot-ai 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- docpilot_ai-0.2.0.dist-info/METADATA +287 -0
- docpilot_ai-0.2.0.dist-info/RECORD +53 -0
- docpilot_ai-0.2.0.dist-info/WHEEL +4 -0
- docpilot_ai-0.2.0.dist-info/entry_points.txt +2 -0
- parsedoc/__init__.py +3 -0
- parsedoc/__main__.py +6 -0
- parsedoc/ai/__init__.py +21 -0
- parsedoc/ai/base.py +59 -0
- parsedoc/ai/cache.py +47 -0
- parsedoc/ai/compatible.py +58 -0
- parsedoc/ai/gemini.py +38 -0
- parsedoc/ai/ollama.py +59 -0
- parsedoc/ai/openai.py +41 -0
- parsedoc/ai/prompts.py +48 -0
- parsedoc/cli.py +226 -0
- parsedoc/core/__init__.py +13 -0
- parsedoc/core/chunking.py +52 -0
- parsedoc/core/config.py +188 -0
- parsedoc/core/detection.py +47 -0
- parsedoc/core/document.py +10 -0
- parsedoc/core/pipeline.py +307 -0
- parsedoc/extraction/__init__.py +25 -0
- parsedoc/extraction/images.py +100 -0
- parsedoc/extraction/layout.py +12 -0
- parsedoc/extraction/quality.py +29 -0
- parsedoc/extraction/tables.py +25 -0
- parsedoc/extraction/text.py +83 -0
- parsedoc/ocr/__init__.py +6 -0
- parsedoc/ocr/base.py +27 -0
- parsedoc/ocr/tesseract.py +44 -0
- parsedoc/parsers/__init__.py +40 -0
- parsedoc/parsers/base.py +56 -0
- parsedoc/parsers/docx.py +71 -0
- parsedoc/parsers/html.py +104 -0
- parsedoc/parsers/image.py +44 -0
- parsedoc/parsers/pdf.py +48 -0
- parsedoc/parsers/pptx.py +39 -0
- parsedoc/parsers/text.py +54 -0
- parsedoc/renderers/__init__.py +33 -0
- parsedoc/renderers/html.py +59 -0
- parsedoc/renderers/json.py +10 -0
- parsedoc/renderers/markdown.py +66 -0
- parsedoc/renderers/text.py +21 -0
- parsedoc/schema/__init__.py +12 -0
- parsedoc/schema/document.py +46 -0
- parsedoc/schema/validation.py +35 -0
- parsedoc/tests/AYANTRA TECHNOLOGIES - Official Commercial Pricing & Service Engagement Framework.docx +0 -0
- parsedoc/tests/Ayantra - Quotation Quotation-AY-QT-2026-001.docx +0 -0
- parsedoc/tests/Presentation-1.pptx +0 -0
- parsedoc/utils/__init__.py +23 -0
- parsedoc/utils/filesystem.py +35 -0
- parsedoc/utils/logging.py +31 -0
- parsedoc/utils/timing.py +39 -0
|
@@ -0,0 +1,287 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: docpilot-ai
|
|
3
|
+
Version: 0.2.0
|
|
4
|
+
Summary: AI-powered local-first CLI that converts documents into clean, structured, AI-ready Markdown
|
|
5
|
+
Author-email: Anuj Paroha <dev.77anuj77@gmail.com>
|
|
6
|
+
License: MIT
|
|
7
|
+
Keywords: ai,cli,document-conversion,docx,llm,markdown,pdf,pptx
|
|
8
|
+
Classifier: Development Status :: 4 - Beta
|
|
9
|
+
Classifier: Environment :: Console
|
|
10
|
+
Classifier: Intended Audience :: Developers
|
|
11
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
12
|
+
Classifier: Operating System :: OS Independent
|
|
13
|
+
Classifier: Programming Language :: Python :: 3
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
17
|
+
Classifier: Topic :: Text Processing :: Markup :: Markdown
|
|
18
|
+
Classifier: Topic :: Utilities
|
|
19
|
+
Requires-Python: >=3.10
|
|
20
|
+
Requires-Dist: beautifulsoup4>=4.12
|
|
21
|
+
Requires-Dist: httpx>=0.27
|
|
22
|
+
Requires-Dist: openai>=1.0
|
|
23
|
+
Requires-Dist: pillow>=10.0
|
|
24
|
+
Requires-Dist: pydantic>=2.5
|
|
25
|
+
Requires-Dist: pymupdf>=1.23
|
|
26
|
+
Requires-Dist: pytesseract>=0.3.10
|
|
27
|
+
Requires-Dist: python-docx>=1.1
|
|
28
|
+
Requires-Dist: python-pptx>=0.6
|
|
29
|
+
Requires-Dist: rich>=13.0
|
|
30
|
+
Requires-Dist: tomli>=2.0; python_version < '3.11'
|
|
31
|
+
Requires-Dist: typer>=0.12
|
|
32
|
+
Provides-Extra: dev
|
|
33
|
+
Requires-Dist: mypy>=1.10; extra == 'dev'
|
|
34
|
+
Requires-Dist: pytest>=8.0; extra == 'dev'
|
|
35
|
+
Requires-Dist: ruff>=0.5; extra == 'dev'
|
|
36
|
+
Description-Content-Type: text/markdown
|
|
37
|
+
|
|
38
|
+
# ParseDoc
|
|
39
|
+
|
|
40
|
+
> Document → AI → Markdown. A local-first CLI that converts documents into clean, structured, AI-ready Markdown.
|
|
41
|
+
|
|
42
|
+
ParseDoc turns PDFs, Word docs, PowerPoint decks, HTML pages, plain text, and images into well-structured Markdown. Libraries extract the raw facts, an optional AI model understands the structure, and a deterministic renderer produces the final output. It works **fully offline** (local rule-based structuring) and can optionally use an AI provider for smarter structuring.
|
|
43
|
+
|
|
44
|
+
---
|
|
45
|
+
|
|
46
|
+
## Features
|
|
47
|
+
|
|
48
|
+
- **Local-first**: conversion works with no AI provider configured (deterministic fallback).
|
|
49
|
+
- **Multi-format input**: PDF, DOCX, PPTX, HTML, TXT/MD, and images (PNG/JPG) via OCR.
|
|
50
|
+
- **AI structuring** (optional): `local`, `ai`, or `hybrid` modes.
|
|
51
|
+
- **Pluggable AI providers**: OpenAI-compatible (vLLM/LM Studio), OpenAI, Google Gemini, and Ollama (native, no SDK needed).
|
|
52
|
+
- **Image extraction**: pull embedded images from DOCX into an assets folder.
|
|
53
|
+
- **OCR**: Tesseract-backed text extraction for scanned PDFs and images.
|
|
54
|
+
- **Multiple output formats**: `markdown` (default), `json`, `html`, `text`.
|
|
55
|
+
|
|
56
|
+
---
|
|
57
|
+
|
|
58
|
+
## Installation
|
|
59
|
+
|
|
60
|
+
```bash
|
|
61
|
+
git clone <repo-url>
|
|
62
|
+
cd parsedoc
|
|
63
|
+
pip install -e .
|
|
64
|
+
```
|
|
65
|
+
|
|
66
|
+
This installs the `parsedoc` command. (Requires Python 3.10+.)
|
|
67
|
+
|
|
68
|
+
### Optional dependencies
|
|
69
|
+
|
|
70
|
+
| Capability | What you need |
|
|
71
|
+
|------------|---------------|
|
|
72
|
+
| PDF parsing | `pymupdf` (fitz) |
|
|
73
|
+
| DOCX / PPTX | `python-docx`, `python-pptx` |
|
|
74
|
+
| HTML | `beautifulsoup4`, `lxml` |
|
|
75
|
+
| OCR | [Tesseract](https://github.com/tesseract-ocr/tesseract) installed + `pytesseract` |
|
|
76
|
+
|
|
77
|
+
---
|
|
78
|
+
|
|
79
|
+
## CLI Usage
|
|
80
|
+
|
|
81
|
+
The CLI is built with Typer. Run `parsedoc --help` for the full list.
|
|
82
|
+
|
|
83
|
+
### Commands
|
|
84
|
+
|
|
85
|
+
| Command | Purpose |
|
|
86
|
+
|---------|---------|
|
|
87
|
+
| `parsedoc convert` | Convert a single document to Markdown. |
|
|
88
|
+
| `parsedoc batch` | Convert every matching file in a directory. |
|
|
89
|
+
| `parsedoc inspect` | Inspect a document's extracted structure (JSON or summary). |
|
|
90
|
+
| `parsedoc config` | View or edit the TOML configuration (`list`, `set`, `reset`). |
|
|
91
|
+
| `parsedoc version` | Print version + banner. |
|
|
92
|
+
|
|
93
|
+
### `convert`
|
|
94
|
+
|
|
95
|
+
```bash
|
|
96
|
+
parsedoc convert INPUT_FILE \
|
|
97
|
+
--output out.md \
|
|
98
|
+
--format markdown \
|
|
99
|
+
--mode hybrid \
|
|
100
|
+
--ai-provider ollama \
|
|
101
|
+
--model qwen3 \
|
|
102
|
+
--extract-images
|
|
103
|
+
```
|
|
104
|
+
|
|
105
|
+
**Options**
|
|
106
|
+
|
|
107
|
+
| Flag | Default | Description |
|
|
108
|
+
|------|---------|-------------|
|
|
109
|
+
| `--output`, `-o` | stdout | Write output to a file. |
|
|
110
|
+
| `--format`, `-f` | `markdown` | `markdown`, `json`, `html`, `text`. |
|
|
111
|
+
| `--mode`, `-m` | `hybrid` | `local`, `ai`, or `hybrid`. |
|
|
112
|
+
| `--ai-provider` | config | Override the AI provider. |
|
|
113
|
+
| `--model` | config | Override the model name. |
|
|
114
|
+
| `--temperature` | `0.2` | AI sampling temperature. |
|
|
115
|
+
| `--max-tokens` | `2048` | Max AI tokens. |
|
|
116
|
+
| `--ocr` | off | Force OCR processing. |
|
|
117
|
+
| `--extract-images` | off | Extract embedded images (DOCX) into `<stem>_assets/`. |
|
|
118
|
+
| `--quiet` / `--verbose` | off | Logging verbosity. |
|
|
119
|
+
|
|
120
|
+
### `batch`
|
|
121
|
+
|
|
122
|
+
```bash
|
|
123
|
+
parsedoc batch ./docs --pattern "*.docx" --format markdown --extract-images
|
|
124
|
+
```
|
|
125
|
+
|
|
126
|
+
### `inspect`
|
|
127
|
+
|
|
128
|
+
```bash
|
|
129
|
+
parsedoc inspect INPUT_FILE --format json # full extracted structure
|
|
130
|
+
parsedoc inspect INPUT_FILE --format summary # blocks / title / format
|
|
131
|
+
```
|
|
132
|
+
|
|
133
|
+
---
|
|
134
|
+
|
|
135
|
+
## Configuration
|
|
136
|
+
|
|
137
|
+
ParseDoc stores config as TOML (default location: `~/.config/parsedoc/config.toml`, or via `default_config_path()`). You can edit it directly or use the CLI.
|
|
138
|
+
|
|
139
|
+
```bash
|
|
140
|
+
parsedoc config list
|
|
141
|
+
parsedoc config set --key ai_provider --value ollama
|
|
142
|
+
parsedoc config set --key model --value qwen3
|
|
143
|
+
parsedoc config reset
|
|
144
|
+
```
|
|
145
|
+
|
|
146
|
+
### Environment variables
|
|
147
|
+
|
|
148
|
+
These override the TOML/config values at runtime:
|
|
149
|
+
|
|
150
|
+
| Variable | Maps to |
|
|
151
|
+
|----------|---------|
|
|
152
|
+
| `PARSEDOC_AI_PROVIDER` | `ai_provider` |
|
|
153
|
+
| `PARSEDOC_BASE_URL` | `base_url` |
|
|
154
|
+
| `PARSEDOC_MODEL` | `model` |
|
|
155
|
+
| `PARSEDOC_API_KEY` | `api_key` |
|
|
156
|
+
| `PARSEDOC_OCR_LANGUAGE` | `ocr_language` |
|
|
157
|
+
|
|
158
|
+
---
|
|
159
|
+
|
|
160
|
+
## Integrating AI Providers
|
|
161
|
+
|
|
162
|
+
ParseDoc supports four provider types, selected via `--ai-provider` (CLI), the `ai.provider` config key, or `PARSEDOC_AI_PROVIDER`.
|
|
163
|
+
|
|
164
|
+
| Provider value | Use case |
|
|
165
|
+
|----------------|----------|
|
|
166
|
+
| `openai-compatible` | Any OpenAI-compatible endpoint (vLLM, LM Studio, local servers, OpenRouter, etc.) |
|
|
167
|
+
| `openai` | OpenAI's hosted API |
|
|
168
|
+
| `gemini` | Google Gemini API |
|
|
169
|
+
| `ollama` | Ollama running locally (native HTTP, no SDK required) |
|
|
170
|
+
|
|
171
|
+
Provider is chosen by the factory in `parsedoc/ai/base.py:build_provider`.
|
|
172
|
+
|
|
173
|
+
### 1. OpenAI-compatible (default)
|
|
174
|
+
|
|
175
|
+
This is the default and works with most self-hosted / drop-in OpenAI servers.
|
|
176
|
+
|
|
177
|
+
```bash
|
|
178
|
+
export PARSEDOC_BASE_URL="http://localhost:11434/v1" # e.g. Ollama's OpenAI shim
|
|
179
|
+
export PARSEDOC_API_KEY="local" # or your real key
|
|
180
|
+
export PARSEDOC_MODEL="qwen3"
|
|
181
|
+
parsedoc config set --key ai_provider --value openai-compatible
|
|
182
|
+
```
|
|
183
|
+
|
|
184
|
+
TOML equivalent (`~/.config/parsedoc/config.toml`):
|
|
185
|
+
|
|
186
|
+
```toml
|
|
187
|
+
[ai]
|
|
188
|
+
enabled = true
|
|
189
|
+
provider = "openai-compatible"
|
|
190
|
+
base_url = "http://localhost:11434/v1"
|
|
191
|
+
model = "qwen3"
|
|
192
|
+
api_key = "local"
|
|
193
|
+
temperature = 0.2
|
|
194
|
+
max_tokens = 4096
|
|
195
|
+
```
|
|
196
|
+
|
|
197
|
+
### 2. OpenAI (hosted)
|
|
198
|
+
|
|
199
|
+
```bash
|
|
200
|
+
export PARSEDOC_BASE_URL="https://api.openai.com/v1"
|
|
201
|
+
export PARSEDOC_API_KEY="sk-..."
|
|
202
|
+
export PARSEDOC_MODEL="gpt-4o-mini"
|
|
203
|
+
parsedoc config set --key ai_provider --value openai
|
|
204
|
+
```
|
|
205
|
+
|
|
206
|
+
### 3. Google Gemini
|
|
207
|
+
|
|
208
|
+
```bash
|
|
209
|
+
export PARSEDOC_API_KEY="AIza..."
|
|
210
|
+
export PARSEDOC_MODEL="gemini-1.5-flash"
|
|
211
|
+
parsedoc config set --key ai_provider --value gemini
|
|
212
|
+
```
|
|
213
|
+
|
|
214
|
+
> Note: the Gemini provider uses its own endpoint; `base_url` is optional and falls back to the Google Generative Language API.
|
|
215
|
+
|
|
216
|
+
### 4. Ollama (native)
|
|
217
|
+
|
|
218
|
+
No Python SDK required — ParseDoc talks to Ollama over HTTP using the standard library.
|
|
219
|
+
|
|
220
|
+
```bash
|
|
221
|
+
export PARSEDOC_BASE_URL="http://localhost:11434" # Ollama root, not the /v1 shim
|
|
222
|
+
export PARSEDOC_MODEL="qwen3"
|
|
223
|
+
parsedoc config set --key ai_provider --value ollama
|
|
224
|
+
```
|
|
225
|
+
|
|
226
|
+
Then make sure the model is pulled:
|
|
227
|
+
|
|
228
|
+
```bash
|
|
229
|
+
ollama pull qwen3
|
|
230
|
+
```
|
|
231
|
+
|
|
232
|
+
---
|
|
233
|
+
|
|
234
|
+
## Processing Modes
|
|
235
|
+
|
|
236
|
+
| Mode | Behavior |
|
|
237
|
+
|------|----------|
|
|
238
|
+
| `local` | Pure rule-based structuring. No network calls. Fast and private. |
|
|
239
|
+
| `ai` | Structures content using the configured AI provider. |
|
|
240
|
+
| `hybrid` | Uses AI when available, falls back to local structuring on failure. |
|
|
241
|
+
|
|
242
|
+
Set via `--mode` on `convert`/`batch`, or `output`/`ai` config sections (the `provider`/`enabled` keys).
|
|
243
|
+
|
|
244
|
+
---
|
|
245
|
+
|
|
246
|
+
## Supported Input Formats
|
|
247
|
+
|
|
248
|
+
| Format | Extensions |
|
|
249
|
+
|--------|-----------|
|
|
250
|
+
| PDF | `.pdf` |
|
|
251
|
+
| Word | `.docx` |
|
|
252
|
+
| PowerPoint | `.pptx` |
|
|
253
|
+
| HTML | `.html`, `.htm` |
|
|
254
|
+
| Text / Markdown | `.txt`, `.md`, `.markdown` |
|
|
255
|
+
| Images | `.png`, `.jpg`, `.jpeg` (OCR) |
|
|
256
|
+
|
|
257
|
+
Check support before converting:
|
|
258
|
+
|
|
259
|
+
```bash
|
|
260
|
+
python -c "from parsedoc.core.detection import is_supported; print(is_supported('file.docx'))"
|
|
261
|
+
```
|
|
262
|
+
|
|
263
|
+
---
|
|
264
|
+
|
|
265
|
+
## Using ParseDoc as a Library
|
|
266
|
+
|
|
267
|
+
```python
|
|
268
|
+
from parsedoc.core.config import Config
|
|
269
|
+
from parsedoc.core.pipeline import Pipeline
|
|
270
|
+
|
|
271
|
+
config = Config().load_from_file()
|
|
272
|
+
pipeline = Pipeline(config)
|
|
273
|
+
|
|
274
|
+
markdown = pipeline.process(
|
|
275
|
+
"report.docx",
|
|
276
|
+
output_format="markdown",
|
|
277
|
+
mode="hybrid",
|
|
278
|
+
extract_images=True,
|
|
279
|
+
output_path="report.md",
|
|
280
|
+
)
|
|
281
|
+
```
|
|
282
|
+
|
|
283
|
+
---
|
|
284
|
+
|
|
285
|
+
## License
|
|
286
|
+
|
|
287
|
+
See repository for license details.
|
|
@@ -0,0 +1,53 @@
|
|
|
1
|
+
parsedoc/__init__.py,sha256=MfqF421g7syQuYnaELGJxPDAbvnRN-z5x8Th_5FzjRk,87
|
|
2
|
+
parsedoc/__main__.py,sha256=vANrez1SjD08QTogjLaEkc2A_28Z0Pg9ngtsMZKrVes,123
|
|
3
|
+
parsedoc/cli.py,sha256=0DILrkFbnAxX50_DjAb21Vfb4KQqTR5pRfWLXrdg49A,7648
|
|
4
|
+
parsedoc/ai/__init__.py,sha256=7onRegAvnO5fXo57g7X1K2AvdOF-bKjVg-FeXxt5NrU,500
|
|
5
|
+
parsedoc/ai/base.py,sha256=T4VORY1-DD03A3xcGNl8x0QZ-To_DoeDK25_k3_HYZ8,1784
|
|
6
|
+
parsedoc/ai/cache.py,sha256=82VUFzpEN3ZdhvduEyW5khugZ8zuc21_2swhbjDYVD4,1390
|
|
7
|
+
parsedoc/ai/compatible.py,sha256=S5TuXkTiOpG3OFKj98OHL_gdWsxUbcyAk3VVrL_8QbE,1706
|
|
8
|
+
parsedoc/ai/gemini.py,sha256=I1X2fRqYxCHW49SlQ2Xco6Dwd-gBxrsIqqVOwfz5PqM,1141
|
|
9
|
+
parsedoc/ai/ollama.py,sha256=7Fb-B8EtjWXsRZNAiGm1Z0XP6XQ6XHNDUpO1g-Oxm4k,1901
|
|
10
|
+
parsedoc/ai/openai.py,sha256=14cMwn2Y4WFfg6TmpPlN2EiY7K4eGGHqrdj1yPdeePc,1234
|
|
11
|
+
parsedoc/ai/prompts.py,sha256=Xeq1xtjK4Z6ZR-k59OmGOTNZCgv4pw388R1_GIOIZWo,1826
|
|
12
|
+
parsedoc/core/__init__.py,sha256=cjPiNahyuXFC12_CSN8vwPZIe-zxH8mjaXuFczDnX4E,272
|
|
13
|
+
parsedoc/core/chunking.py,sha256=GjoVrV6_TSQARGjZy8vrGoSZ7koS61gx8ZjCluivLkE,1548
|
|
14
|
+
parsedoc/core/config.py,sha256=YkK2nJuCxZOjDy4PPXaulNJIuth7UektgecCwO6e9l0,6642
|
|
15
|
+
parsedoc/core/detection.py,sha256=74QZc5pdS6E9jMcaqT87G296EIbSG3d6BAOhGE68A9U,1218
|
|
16
|
+
parsedoc/core/document.py,sha256=NJR6mFoEWDF8CuaBT8OFBhufa2rgaLqWngQsgLkj-Y8,375
|
|
17
|
+
parsedoc/core/pipeline.py,sha256=z8XBpniP0qdERhUlbVvNDUcfsskm5nfqSmNClRLPBjE,11438
|
|
18
|
+
parsedoc/extraction/__init__.py,sha256=1Q2qStA4zt6FiK-GP6P7ciczJcrGqL9vEESR4DwDPs0,576
|
|
19
|
+
parsedoc/extraction/images.py,sha256=j8uw9XOmeLFAJw1h1d6omp8HYcV9LHc55X3s3F349f4,3118
|
|
20
|
+
parsedoc/extraction/layout.py,sha256=mcig8FcfAd-h1O8k7IgazQWSznRU4qF2ywYf8E1mXfE,497
|
|
21
|
+
parsedoc/extraction/quality.py,sha256=8Bo3im90rR7dhumbQzxgD5LFcIoAQVMC0TVUnnHx-yo,944
|
|
22
|
+
parsedoc/extraction/tables.py,sha256=V9V34Kkv_c_BAAA50e9q0woSHL6_fPKo6XIXHf4cZLg,835
|
|
23
|
+
parsedoc/extraction/text.py,sha256=2fJHSRT_VxduUfv_PPP9kEFFqM5cNGOMOaT_AnVp7DI,2971
|
|
24
|
+
parsedoc/ocr/__init__.py,sha256=ponvt0umORudlLDnFyeVWpuMH9spmDgmAU144IzUaS0,189
|
|
25
|
+
parsedoc/ocr/base.py,sha256=s3srGEdhbTJEqFU6Ve4jHdw8XUshTXt4NefGcRdnhDA,696
|
|
26
|
+
parsedoc/ocr/tesseract.py,sha256=5vHrk6_whhfqlgD9A0A1-yxwHxNqxASDaMM_Hkg0Ai8,1258
|
|
27
|
+
parsedoc/parsers/__init__.py,sha256=9aUo8RyiwN5Wwn7f89F3DkyppE7VbatPq6_UZcI9Qz0,895
|
|
28
|
+
parsedoc/parsers/base.py,sha256=lSmN3cIu6tIl_Pvx0_U9vF0kOQ-Zs62C8WyWyLrSntw,1870
|
|
29
|
+
parsedoc/parsers/docx.py,sha256=yhqjY7Ln3l9pgNpyM9_NyALoNtTj96TNMtErHkukDjE,2645
|
|
30
|
+
parsedoc/parsers/html.py,sha256=DtyBDTEhEfRfXhbcRjnMHH_1YkjmSYaeurRKJEl_T_E,4180
|
|
31
|
+
parsedoc/parsers/image.py,sha256=q_qngZ0NjnIfddxS_upAbm8_sX3rcSEwUfIny54w0hU,1355
|
|
32
|
+
parsedoc/parsers/pdf.py,sha256=_NvQF54s0V_9fpY4s09P9OXEJ3-LcX2HdMndS48zbtA,1605
|
|
33
|
+
parsedoc/parsers/pptx.py,sha256=g_mPh70xEQMoPun_cUzsQOQYVtFzqHPkNz3tZPhdA7M,1202
|
|
34
|
+
parsedoc/parsers/text.py,sha256=O0Lqri09U4wAVV1TsT9ucD9mmV-V-Szp4Y3wM0_kRFs,1892
|
|
35
|
+
parsedoc/renderers/__init__.py,sha256=bptHtTqsVgTsBTAD6M86fYJUAfBFGNG1Ys4DAfVRNo4,726
|
|
36
|
+
parsedoc/renderers/html.py,sha256=7qoK6X7v1tnT6YN5G5k_94VUKABWu2Ewpg5HSTKczu8,2058
|
|
37
|
+
parsedoc/renderers/json.py,sha256=Yg80uRMjCj05fX5xSx1cWUTozC5_Pz2EsazrhS888Pc,245
|
|
38
|
+
parsedoc/renderers/markdown.py,sha256=xzz7LlloXeFf68Fs3jpDXg35YcciEGqNhzeyrCssoQY,2217
|
|
39
|
+
parsedoc/renderers/text.py,sha256=QSnCxTEb8Z3znHVq5IbEpAZKKWreJ0NFELgu-X4tS0Q,726
|
|
40
|
+
parsedoc/schema/__init__.py,sha256=W-YCULJUbiT9jnySL3jVPd3HRKOeC0ABBhl94ETIvu0,270
|
|
41
|
+
parsedoc/schema/document.py,sha256=ObpE6M6q4SDP4bbat9WJ0BhdMOPgKSZqb32prW0miMk,1142
|
|
42
|
+
parsedoc/schema/validation.py,sha256=3XReGUt1gdwqPwcXgscT4LpwGWJ051mbOxKp34xlzaU,896
|
|
43
|
+
parsedoc/tests/AYANTRA TECHNOLOGIES - Official Commercial Pricing & Service Engagement Framework.docx,sha256=g5ndZKLggpbk8NS7wterCgSqOkoyA4YPc5fFa8tNyig,1995630
|
|
44
|
+
parsedoc/tests/Ayantra - Quotation Quotation-AY-QT-2026-001.docx,sha256=O1l-D2AMvFowrVqNANUi0qPXZHfZJDbNO5HAg9YbJJE,422551
|
|
45
|
+
parsedoc/tests/Presentation-1.pptx,sha256=KlMpG4dk9117FEJQzdY1CfO91NhYjzQJLPg9YNKP5Mo,1753060
|
|
46
|
+
parsedoc/utils/__init__.py,sha256=ZCvDs_noYPDmsoVYFhmJrMF7sulbzOPmtJuUGd38J9U,450
|
|
47
|
+
parsedoc/utils/filesystem.py,sha256=kNtvxcNL9Z2Qv2thorfNKRblllA418HTpEjDLrFAGic,930
|
|
48
|
+
parsedoc/utils/logging.py,sha256=31WnTaL6CEAr00vf93CktoTDk105wiuOJ3w6nxoGNco,795
|
|
49
|
+
parsedoc/utils/timing.py,sha256=goESFFeTKCWjsgkhvGVVLWWNXcqmNvWovUOjP89aPmA,954
|
|
50
|
+
docpilot_ai-0.2.0.dist-info/METADATA,sha256=r915tskRhaYS-1dhbCbcpk0QZLUZPGtjG0K9M7u9o3I,8455
|
|
51
|
+
docpilot_ai-0.2.0.dist-info/WHEEL,sha256=zOwg4jB6zX2kU910N-cMawjivD6tO8NEWvE12je1bVk,87
|
|
52
|
+
docpilot_ai-0.2.0.dist-info/entry_points.txt,sha256=oz_b-Rlrxb8eBVa6EoYmxUWcFcujf7onbcJGMzn-wEU,46
|
|
53
|
+
docpilot_ai-0.2.0.dist-info/RECORD,,
|
parsedoc/__init__.py
ADDED
parsedoc/__main__.py
ADDED
parsedoc/ai/__init__.py
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
"""ParseDoc AI package"""
|
|
2
|
+
|
|
3
|
+
from .base import AIProvider, build_provider
|
|
4
|
+
from .compatible import OpenAICompatibleProvider
|
|
5
|
+
from .openai import OpenAIProvider
|
|
6
|
+
from .gemini import GeminiProvider
|
|
7
|
+
from .ollama import OllamaProvider
|
|
8
|
+
from .prompts import PROMPT_VERSION, get_prompt
|
|
9
|
+
from . import cache
|
|
10
|
+
|
|
11
|
+
__all__ = [
|
|
12
|
+
"AIProvider",
|
|
13
|
+
"build_provider",
|
|
14
|
+
"OpenAICompatibleProvider",
|
|
15
|
+
"OpenAIProvider",
|
|
16
|
+
"GeminiProvider",
|
|
17
|
+
"OllamaProvider",
|
|
18
|
+
"PROMPT_VERSION",
|
|
19
|
+
"get_prompt",
|
|
20
|
+
"cache",
|
|
21
|
+
]
|
parsedoc/ai/base.py
ADDED
|
@@ -0,0 +1,59 @@
|
|
|
1
|
+
"""ParseDoc AI Provider abstraction (PRD #12)"""
|
|
2
|
+
|
|
3
|
+
import abc
|
|
4
|
+
from typing import Optional
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
class AIProvider(abc.ABC):
|
|
8
|
+
"""Abstract interface for AI providers.
|
|
9
|
+
|
|
10
|
+
Every provider must implement :meth:`generate`, which takes a prompt and
|
|
11
|
+
content and returns the model's raw text response.
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
name: str = "base"
|
|
15
|
+
model: str = ""
|
|
16
|
+
|
|
17
|
+
@abc.abstractmethod
|
|
18
|
+
def generate(self, prompt: str, content: str) -> str:
|
|
19
|
+
"""Generate a response from the model for the given prompt + content."""
|
|
20
|
+
raise NotImplementedError
|
|
21
|
+
|
|
22
|
+
def is_available(self) -> bool:
|
|
23
|
+
"""Return True if the provider can be used (dependencies/config present)."""
|
|
24
|
+
return True
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def build_provider(
|
|
28
|
+
provider: Optional[str] = None,
|
|
29
|
+
base_url: Optional[str] = None,
|
|
30
|
+
model: Optional[str] = None,
|
|
31
|
+
api_key: Optional[str] = None,
|
|
32
|
+
) -> AIProvider:
|
|
33
|
+
"""Factory that returns an AIProvider instance by name.
|
|
34
|
+
|
|
35
|
+
Supported names: openai-compatible, openai, gemini, ollama.
|
|
36
|
+
"""
|
|
37
|
+
provider = (provider or "openai-compatible").lower()
|
|
38
|
+
|
|
39
|
+
if provider in ("openai-compatible", "vllm", "lm-studio"):
|
|
40
|
+
from .compatible import OpenAICompatibleProvider
|
|
41
|
+
|
|
42
|
+
return OpenAICompatibleProvider(base_url=base_url, model=model, api_key=api_key)
|
|
43
|
+
|
|
44
|
+
if provider == "ollama":
|
|
45
|
+
from .ollama import OllamaProvider
|
|
46
|
+
|
|
47
|
+
return OllamaProvider(base_url=base_url, model=model, api_key=api_key)
|
|
48
|
+
|
|
49
|
+
if provider == "openai":
|
|
50
|
+
from .openai import OpenAIProvider
|
|
51
|
+
|
|
52
|
+
return OpenAIProvider(base_url=base_url, model=model, api_key=api_key)
|
|
53
|
+
|
|
54
|
+
if provider == "gemini":
|
|
55
|
+
from .gemini import GeminiProvider
|
|
56
|
+
|
|
57
|
+
return GeminiProvider(base_url=base_url, model=model, api_key=api_key)
|
|
58
|
+
|
|
59
|
+
raise ValueError(f"Unknown AI provider: {provider}")
|
parsedoc/ai/cache.py
ADDED
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
"""ParseDoc AI Cache (PRD #22, #46)"""
|
|
2
|
+
|
|
3
|
+
import hashlib
|
|
4
|
+
import json
|
|
5
|
+
import os
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
from typing import Dict, Optional
|
|
8
|
+
|
|
9
|
+
from ..utils.filesystem import default_cache_dir
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def _hash_inputs(content: str, model: str, prompt_version: str, config_hash: str) -> str:
|
|
13
|
+
payload = f"{content}|{model}|{prompt_version}|{config_hash}"
|
|
14
|
+
return hashlib.sha256(payload.encode("utf-8")).hexdigest()
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def _config_hash(config: Dict) -> str:
|
|
18
|
+
return hashlib.sha256(json.dumps(config, sort_keys=True).encode("utf-8")).hexdigest()
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def cache_path() -> Path:
|
|
22
|
+
path = Path(default_cache_dir())
|
|
23
|
+
path.mkdir(parents=True, exist_ok=True)
|
|
24
|
+
return path / "ai_cache.json"
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def get_cache() -> Dict:
|
|
28
|
+
p = cache_path()
|
|
29
|
+
if p.exists():
|
|
30
|
+
try:
|
|
31
|
+
return json.loads(p.read_text(encoding="utf-8"))
|
|
32
|
+
except Exception:
|
|
33
|
+
return {}
|
|
34
|
+
return {}
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def lookup(content: str, model: str, prompt_version: str, config: Dict) -> Optional[str]:
|
|
38
|
+
cache = get_cache()
|
|
39
|
+
key = _hash_inputs(content, model, prompt_version, _config_hash(config))
|
|
40
|
+
return cache.get(key)
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def store(content: str, model: str, prompt_version: str, config: Dict, result: str):
|
|
44
|
+
cache = get_cache()
|
|
45
|
+
key = _hash_inputs(content, model, prompt_version, _config_hash(config))
|
|
46
|
+
cache[key] = result
|
|
47
|
+
cache_path().write_text(json.dumps(cache, indent=2), encoding="utf-8")
|
|
@@ -0,0 +1,58 @@
|
|
|
1
|
+
"""ParseDoc OpenAI-compatible AI provider (PRD #13, P0)"""
|
|
2
|
+
|
|
3
|
+
from typing import Optional
|
|
4
|
+
|
|
5
|
+
from .base import AIProvider
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
class OpenAICompatibleProvider(AIProvider):
|
|
9
|
+
"""Provider that talks to any OpenAI-compatible chat completions API.
|
|
10
|
+
|
|
11
|
+
Works with OpenAI, Ollama, LM Studio, vLLM, etc. by pointing ``base_url``
|
|
12
|
+
at the server's ``/v1`` endpoint.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
name = "openai-compatible"
|
|
16
|
+
|
|
17
|
+
def __init__(
|
|
18
|
+
self,
|
|
19
|
+
base_url: Optional[str] = "http://localhost:11434/v1",
|
|
20
|
+
model: Optional[str] = "qwen3",
|
|
21
|
+
api_key: Optional[str] = "local",
|
|
22
|
+
):
|
|
23
|
+
self.base_url = base_url or "http://localhost:11434/v1"
|
|
24
|
+
self.model = model or "qwen3"
|
|
25
|
+
self.api_key = api_key or "local"
|
|
26
|
+
|
|
27
|
+
def generate(self, prompt: str, content: str) -> str:
|
|
28
|
+
try:
|
|
29
|
+
from openai import OpenAI
|
|
30
|
+
except ImportError as e:
|
|
31
|
+
raise RuntimeError(
|
|
32
|
+
"The 'openai' package is required for OpenAI-compatible providers. "
|
|
33
|
+
"Install with: pip install openai"
|
|
34
|
+
) from e
|
|
35
|
+
|
|
36
|
+
client = OpenAI(
|
|
37
|
+
base_url=self.base_url,
|
|
38
|
+
api_key=self.api_key,
|
|
39
|
+
timeout=120,
|
|
40
|
+
max_retries=1,
|
|
41
|
+
)
|
|
42
|
+
response = client.chat.completions.create(
|
|
43
|
+
model=self.model,
|
|
44
|
+
messages=[
|
|
45
|
+
{"role": "system", "content": prompt},
|
|
46
|
+
{"role": "user", "content": content},
|
|
47
|
+
],
|
|
48
|
+
temperature=0.2,
|
|
49
|
+
)
|
|
50
|
+
return response.choices[0].message.content or ""
|
|
51
|
+
|
|
52
|
+
def is_available(self) -> bool:
|
|
53
|
+
try:
|
|
54
|
+
import openai # noqa: F401
|
|
55
|
+
|
|
56
|
+
return True
|
|
57
|
+
except ImportError:
|
|
58
|
+
return False
|
parsedoc/ai/gemini.py
ADDED
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
"""ParseDoc Gemini provider (PRD #14, P1)"""
|
|
2
|
+
|
|
3
|
+
from typing import Optional
|
|
4
|
+
|
|
5
|
+
from .base import AIProvider
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
class GeminiProvider(AIProvider):
|
|
9
|
+
name = "gemini"
|
|
10
|
+
|
|
11
|
+
def __init__(
|
|
12
|
+
self,
|
|
13
|
+
base_url: Optional[str] = None,
|
|
14
|
+
model: Optional[str] = "gemini-1.5-flash",
|
|
15
|
+
api_key: Optional[str] = None,
|
|
16
|
+
):
|
|
17
|
+
self.model = model or "gemini-1.5-flash"
|
|
18
|
+
self.api_key = api_key
|
|
19
|
+
|
|
20
|
+
def generate(self, prompt: str, content: str) -> str:
|
|
21
|
+
try:
|
|
22
|
+
import google.generativeai as genai
|
|
23
|
+
except ImportError as e:
|
|
24
|
+
raise RuntimeError(
|
|
25
|
+
"The 'google-generativeai' package is required. "
|
|
26
|
+
"Install with: pip install google-generativeai"
|
|
27
|
+
) from e
|
|
28
|
+
|
|
29
|
+
if not self.api_key:
|
|
30
|
+
raise RuntimeError("Gemini requires an API key (GEMINI_API_KEY/GOOGLE_API_KEY).")
|
|
31
|
+
|
|
32
|
+
genai.configure(api_key=self.api_key)
|
|
33
|
+
model = genai.GenerativeModel(self.model)
|
|
34
|
+
response = model.generate_content(f"{prompt}\n\n{content}")
|
|
35
|
+
return response.text or ""
|
|
36
|
+
|
|
37
|
+
def is_available(self) -> bool:
|
|
38
|
+
return bool(self.api_key)
|
parsedoc/ai/ollama.py
ADDED
|
@@ -0,0 +1,59 @@
|
|
|
1
|
+
"""ParseDoc Ollama provider (PRD #14, P1) - native Ollama API.
|
|
2
|
+
|
|
3
|
+
Uses the local Ollama REST API via the standard library so no extra SDK is
|
|
4
|
+
required. Ollama also exposes an OpenAI-compatible ``/v1`` endpoint, but this
|
|
5
|
+
implementation talks to ``/api/chat`` directly and requests JSON output.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from typing import Optional
|
|
9
|
+
|
|
10
|
+
from .base import AIProvider
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
class OllamaProvider(AIProvider):
|
|
14
|
+
name = "ollama"
|
|
15
|
+
|
|
16
|
+
def __init__(
|
|
17
|
+
self,
|
|
18
|
+
base_url: Optional[str] = "http://localhost:11434",
|
|
19
|
+
model: Optional[str] = "qwen3",
|
|
20
|
+
api_key: Optional[str] = "ollama",
|
|
21
|
+
):
|
|
22
|
+
self.base_url = (base_url or "http://localhost:11434").rstrip("/")
|
|
23
|
+
self.model = model or "qwen3"
|
|
24
|
+
self.api_key = api_key
|
|
25
|
+
|
|
26
|
+
def generate(self, prompt: str, content: str) -> str:
|
|
27
|
+
import json
|
|
28
|
+
import urllib.request
|
|
29
|
+
|
|
30
|
+
url = f"{self.base_url}/api/chat"
|
|
31
|
+
payload = {
|
|
32
|
+
"model": self.model,
|
|
33
|
+
"messages": [
|
|
34
|
+
{"role": "system", "content": prompt},
|
|
35
|
+
{"role": "user", "content": content},
|
|
36
|
+
],
|
|
37
|
+
"stream": False,
|
|
38
|
+
"format": "json",
|
|
39
|
+
}
|
|
40
|
+
request = urllib.request.Request(
|
|
41
|
+
url,
|
|
42
|
+
data=json.dumps(payload).encode("utf-8"),
|
|
43
|
+
headers={"Content-Type": "application/json"},
|
|
44
|
+
)
|
|
45
|
+
try:
|
|
46
|
+
with urllib.request.urlopen(request, timeout=300) as response:
|
|
47
|
+
data = json.loads(response.read().decode("utf-8"))
|
|
48
|
+
return data["message"]["content"] or ""
|
|
49
|
+
except Exception as exc: # noqa: BLE001
|
|
50
|
+
raise RuntimeError(f"Ollama request failed: {exc}") from exc
|
|
51
|
+
|
|
52
|
+
def is_available(self) -> bool:
|
|
53
|
+
import urllib.request
|
|
54
|
+
|
|
55
|
+
try:
|
|
56
|
+
urllib.request.urlopen(f"{self.base_url}/api/tags", timeout=3)
|
|
57
|
+
return True
|
|
58
|
+
except Exception:
|
|
59
|
+
return False
|
parsedoc/ai/openai.py
ADDED
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
"""ParseDoc OpenAI provider (PRD #14, P2)"""
|
|
2
|
+
|
|
3
|
+
from typing import Optional
|
|
4
|
+
|
|
5
|
+
from .base import AIProvider
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
class OpenAIProvider(AIProvider):
|
|
9
|
+
name = "openai"
|
|
10
|
+
|
|
11
|
+
def __init__(
|
|
12
|
+
self,
|
|
13
|
+
base_url: Optional[str] = "https://api.openai.com/v1",
|
|
14
|
+
model: Optional[str] = "gpt-4o-mini",
|
|
15
|
+
api_key: Optional[str] = None,
|
|
16
|
+
):
|
|
17
|
+
self.base_url = base_url or "https://api.openai.com/v1"
|
|
18
|
+
self.model = model or "gpt-4o-mini"
|
|
19
|
+
self.api_key = api_key
|
|
20
|
+
|
|
21
|
+
def generate(self, prompt: str, content: str) -> str:
|
|
22
|
+
try:
|
|
23
|
+
from openai import OpenAI
|
|
24
|
+
except ImportError as e:
|
|
25
|
+
raise RuntimeError(
|
|
26
|
+
"The 'openai' package is required. Install with: pip install openai"
|
|
27
|
+
) from e
|
|
28
|
+
|
|
29
|
+
client = OpenAI(base_url=self.base_url, api_key=self.api_key)
|
|
30
|
+
response = client.chat.completions.create(
|
|
31
|
+
model=self.model,
|
|
32
|
+
messages=[
|
|
33
|
+
{"role": "system", "content": prompt},
|
|
34
|
+
{"role": "user", "content": content},
|
|
35
|
+
],
|
|
36
|
+
temperature=0.2,
|
|
37
|
+
)
|
|
38
|
+
return response.choices[0].message.content or ""
|
|
39
|
+
|
|
40
|
+
def is_available(self) -> bool:
|
|
41
|
+
return bool(self.api_key)
|