deepcode-hku 1.0.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- cli/__init__.py +18 -0
- cli/cli_app.py +296 -0
- cli/cli_interface.py +744 -0
- cli/cli_launcher.py +155 -0
- cli/main_cli.py +243 -0
- cli/workflows/__init__.py +11 -0
- cli/workflows/cli_workflow_adapter.py +336 -0
- deepcode.py +219 -0
- deepcode_hku-1.0.1.dist-info/METADATA +695 -0
- deepcode_hku-1.0.1.dist-info/RECORD +44 -0
- deepcode_hku-1.0.1.dist-info/WHEEL +5 -0
- deepcode_hku-1.0.1.dist-info/entry_points.txt +2 -0
- deepcode_hku-1.0.1.dist-info/licenses/LICENSE +21 -0
- deepcode_hku-1.0.1.dist-info/top_level.txt +6 -0
- tools/__init__.py +0 -0
- tools/code_implementation_server.py +1045 -0
- tools/code_indexer.py +1657 -0
- tools/code_reference_indexer.py +486 -0
- tools/command_executor.py +324 -0
- tools/git_command.py +356 -0
- tools/pdf_converter.py +640 -0
- tools/pdf_downloader.py +1370 -0
- tools/pdf_utils.py +52 -0
- ui/__init__.py +43 -0
- ui/app.py +13 -0
- ui/components.py +1450 -0
- ui/handlers.py +773 -0
- ui/layout.py +106 -0
- ui/streamlit_app.py +38 -0
- ui/styles.py +2116 -0
- utils/__init__.py +17 -0
- utils/cli_interface.py +459 -0
- utils/dialogue_logger.py +671 -0
- utils/file_processor.py +426 -0
- utils/simple_llm_logger.py +198 -0
- workflows/__init__.py +31 -0
- workflows/agent_orchestration_engine.py +1371 -0
- workflows/agents/__init__.py +13 -0
- workflows/agents/code_implementation_agent.py +1093 -0
- workflows/agents/memory_agent_concise.py +923 -0
- workflows/agents/memory_agent_concise_index.py +935 -0
- workflows/code_implementation_workflow.py +924 -0
- workflows/code_implementation_workflow_index.py +931 -0
- workflows/codebase_index_workflow.py +726 -0
tools/pdf_converter.py
ADDED
|
@@ -0,0 +1,640 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""
|
|
3
|
+
PDF Converter Utility
|
|
4
|
+
|
|
5
|
+
This module provides functionality for converting various document formats to PDF,
|
|
6
|
+
including Office documents (.doc, .docx, .ppt, .pptx, .xls, .xlsx) and text files (.txt, .md).
|
|
7
|
+
|
|
8
|
+
Requirements:
|
|
9
|
+
- LibreOffice for Office document conversion
|
|
10
|
+
- ReportLab for text-to-PDF conversion
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
import argparse
|
|
16
|
+
import logging
|
|
17
|
+
import subprocess
|
|
18
|
+
import tempfile
|
|
19
|
+
import shutil
|
|
20
|
+
import platform
|
|
21
|
+
from pathlib import Path
|
|
22
|
+
from typing import Union, Optional, Dict, Any
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
class PDFConverter:
|
|
26
|
+
"""
|
|
27
|
+
PDF conversion utility class.
|
|
28
|
+
|
|
29
|
+
Provides methods to convert Office documents and text files to PDF format.
|
|
30
|
+
"""
|
|
31
|
+
|
|
32
|
+
# Define supported file formats
|
|
33
|
+
OFFICE_FORMATS = {".doc", ".docx", ".ppt", ".pptx", ".xls", ".xlsx"}
|
|
34
|
+
TEXT_FORMATS = {".txt", ".md"}
|
|
35
|
+
|
|
36
|
+
# Class-level logger
|
|
37
|
+
logger = logging.getLogger(__name__)
|
|
38
|
+
|
|
39
|
+
def __init__(self) -> None:
|
|
40
|
+
"""Initialize the PDF converter."""
|
|
41
|
+
pass
|
|
42
|
+
|
|
43
|
+
@staticmethod
|
|
44
|
+
def convert_office_to_pdf(
|
|
45
|
+
doc_path: Union[str, Path], output_dir: Optional[str] = None
|
|
46
|
+
) -> Path:
|
|
47
|
+
"""
|
|
48
|
+
Convert Office document (.doc, .docx, .ppt, .pptx, .xls, .xlsx) to PDF.
|
|
49
|
+
Requires LibreOffice to be installed.
|
|
50
|
+
|
|
51
|
+
Args:
|
|
52
|
+
doc_path: Path to the Office document file
|
|
53
|
+
output_dir: Output directory for the PDF file
|
|
54
|
+
|
|
55
|
+
Returns:
|
|
56
|
+
Path to the generated PDF file
|
|
57
|
+
"""
|
|
58
|
+
try:
|
|
59
|
+
# Convert to Path object for easier handling
|
|
60
|
+
doc_path = Path(doc_path)
|
|
61
|
+
if not doc_path.exists():
|
|
62
|
+
raise FileNotFoundError(f"Office document does not exist: {doc_path}")
|
|
63
|
+
|
|
64
|
+
name_without_suff = doc_path.stem
|
|
65
|
+
|
|
66
|
+
# Prepare output directory
|
|
67
|
+
if output_dir:
|
|
68
|
+
base_output_dir = Path(output_dir)
|
|
69
|
+
else:
|
|
70
|
+
base_output_dir = doc_path.parent / "pdf_output"
|
|
71
|
+
|
|
72
|
+
base_output_dir.mkdir(parents=True, exist_ok=True)
|
|
73
|
+
|
|
74
|
+
# Check if LibreOffice is available
|
|
75
|
+
libreoffice_available = False
|
|
76
|
+
working_libreoffice_cmd: Optional[str] = None
|
|
77
|
+
|
|
78
|
+
# Prepare subprocess parameters to hide console window on Windows
|
|
79
|
+
subprocess_kwargs: Dict[str, Any] = {
|
|
80
|
+
"capture_output": True,
|
|
81
|
+
"check": True,
|
|
82
|
+
"timeout": 10,
|
|
83
|
+
"encoding": "utf-8",
|
|
84
|
+
"errors": "ignore",
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
# Hide console window on Windows
|
|
88
|
+
if platform.system() == "Windows":
|
|
89
|
+
subprocess_kwargs["creationflags"] = (
|
|
90
|
+
0x08000000 # subprocess.CREATE_NO_WINDOW
|
|
91
|
+
)
|
|
92
|
+
|
|
93
|
+
try:
|
|
94
|
+
result = subprocess.run(
|
|
95
|
+
["libreoffice", "--version"], **subprocess_kwargs
|
|
96
|
+
)
|
|
97
|
+
libreoffice_available = True
|
|
98
|
+
working_libreoffice_cmd = "libreoffice"
|
|
99
|
+
logging.info(f"LibreOffice detected: {result.stdout.strip()}") # type: ignore
|
|
100
|
+
except (
|
|
101
|
+
subprocess.CalledProcessError,
|
|
102
|
+
FileNotFoundError,
|
|
103
|
+
subprocess.TimeoutExpired,
|
|
104
|
+
):
|
|
105
|
+
pass
|
|
106
|
+
|
|
107
|
+
# Try alternative commands for LibreOffice
|
|
108
|
+
if not libreoffice_available:
|
|
109
|
+
for cmd in ["soffice", "libreoffice"]:
|
|
110
|
+
try:
|
|
111
|
+
result = subprocess.run([cmd, "--version"], **subprocess_kwargs)
|
|
112
|
+
libreoffice_available = True
|
|
113
|
+
working_libreoffice_cmd = cmd
|
|
114
|
+
logging.info(
|
|
115
|
+
f"LibreOffice detected with command '{cmd}': {result.stdout.strip()}" # type: ignore
|
|
116
|
+
)
|
|
117
|
+
break
|
|
118
|
+
except (
|
|
119
|
+
subprocess.CalledProcessError,
|
|
120
|
+
FileNotFoundError,
|
|
121
|
+
subprocess.TimeoutExpired,
|
|
122
|
+
):
|
|
123
|
+
continue
|
|
124
|
+
|
|
125
|
+
if not libreoffice_available:
|
|
126
|
+
raise RuntimeError(
|
|
127
|
+
"LibreOffice is required for Office document conversion but was not found.\n"
|
|
128
|
+
"Please install LibreOffice:\n"
|
|
129
|
+
"- Windows: Download from https://www.libreoffice.org/download/download/\n"
|
|
130
|
+
"- macOS: brew install --cask libreoffice\n"
|
|
131
|
+
"- Ubuntu/Debian: sudo apt-get install libreoffice\n"
|
|
132
|
+
"- CentOS/RHEL: sudo yum install libreoffice\n"
|
|
133
|
+
"Alternatively, convert the document to PDF manually."
|
|
134
|
+
)
|
|
135
|
+
|
|
136
|
+
# Create temporary directory for PDF conversion
|
|
137
|
+
with tempfile.TemporaryDirectory() as temp_dir:
|
|
138
|
+
temp_path = Path(temp_dir)
|
|
139
|
+
|
|
140
|
+
# Convert to PDF using LibreOffice
|
|
141
|
+
logging.info(f"Converting {doc_path.name} to PDF using LibreOffice...")
|
|
142
|
+
|
|
143
|
+
# Use the working LibreOffice command first, then try alternatives if it fails
|
|
144
|
+
commands_to_try = [working_libreoffice_cmd]
|
|
145
|
+
if working_libreoffice_cmd == "libreoffice":
|
|
146
|
+
commands_to_try.append("soffice")
|
|
147
|
+
else:
|
|
148
|
+
commands_to_try.append("libreoffice")
|
|
149
|
+
|
|
150
|
+
conversion_successful = False
|
|
151
|
+
for cmd in commands_to_try:
|
|
152
|
+
if cmd is None:
|
|
153
|
+
continue
|
|
154
|
+
try:
|
|
155
|
+
convert_cmd = [
|
|
156
|
+
cmd,
|
|
157
|
+
"--headless",
|
|
158
|
+
"--convert-to",
|
|
159
|
+
"pdf",
|
|
160
|
+
"--outdir",
|
|
161
|
+
str(temp_path),
|
|
162
|
+
str(doc_path),
|
|
163
|
+
]
|
|
164
|
+
|
|
165
|
+
# Prepare conversion subprocess parameters
|
|
166
|
+
convert_subprocess_kwargs: Dict[str, Any] = {
|
|
167
|
+
"capture_output": True,
|
|
168
|
+
"text": True,
|
|
169
|
+
"timeout": 60, # 60 second timeout
|
|
170
|
+
"encoding": "utf-8",
|
|
171
|
+
"errors": "ignore",
|
|
172
|
+
}
|
|
173
|
+
|
|
174
|
+
# Hide console window on Windows
|
|
175
|
+
if platform.system() == "Windows":
|
|
176
|
+
convert_subprocess_kwargs["creationflags"] = (
|
|
177
|
+
0x08000000 # subprocess.CREATE_NO_WINDOW
|
|
178
|
+
)
|
|
179
|
+
|
|
180
|
+
result = subprocess.run(
|
|
181
|
+
convert_cmd, **convert_subprocess_kwargs
|
|
182
|
+
)
|
|
183
|
+
|
|
184
|
+
if result.returncode == 0: # type: ignore
|
|
185
|
+
conversion_successful = True
|
|
186
|
+
logging.info(
|
|
187
|
+
f"Successfully converted {doc_path.name} to PDF"
|
|
188
|
+
)
|
|
189
|
+
break
|
|
190
|
+
else:
|
|
191
|
+
logging.warning(
|
|
192
|
+
f"LibreOffice command '{cmd}' failed: {result.stderr}" # type: ignore
|
|
193
|
+
)
|
|
194
|
+
except subprocess.TimeoutExpired:
|
|
195
|
+
logging.warning(f"LibreOffice command '{cmd}' timed out")
|
|
196
|
+
except Exception as e:
|
|
197
|
+
logging.error(
|
|
198
|
+
f"LibreOffice command '{cmd}' failed with exception: {e}"
|
|
199
|
+
)
|
|
200
|
+
|
|
201
|
+
if not conversion_successful:
|
|
202
|
+
raise RuntimeError(
|
|
203
|
+
f"LibreOffice conversion failed for {doc_path.name}. "
|
|
204
|
+
f"Please check if the file is corrupted or try converting manually."
|
|
205
|
+
)
|
|
206
|
+
|
|
207
|
+
# Find the generated PDF
|
|
208
|
+
pdf_files = list(temp_path.glob("*.pdf"))
|
|
209
|
+
if not pdf_files:
|
|
210
|
+
raise RuntimeError(
|
|
211
|
+
f"PDF conversion failed for {doc_path.name} - no PDF file generated. "
|
|
212
|
+
f"Please check LibreOffice installation or try manual conversion."
|
|
213
|
+
)
|
|
214
|
+
|
|
215
|
+
pdf_path = pdf_files[0]
|
|
216
|
+
logging.info(
|
|
217
|
+
f"Generated PDF: {pdf_path.name} ({pdf_path.stat().st_size} bytes)"
|
|
218
|
+
)
|
|
219
|
+
|
|
220
|
+
# Validate the generated PDF
|
|
221
|
+
if pdf_path.stat().st_size < 100: # Very small file, likely empty
|
|
222
|
+
raise RuntimeError(
|
|
223
|
+
"Generated PDF appears to be empty or corrupted. "
|
|
224
|
+
"Original file may have issues or LibreOffice conversion failed."
|
|
225
|
+
)
|
|
226
|
+
|
|
227
|
+
# Copy PDF to final output directory
|
|
228
|
+
final_pdf_path = base_output_dir / f"{name_without_suff}.pdf"
|
|
229
|
+
shutil.copy2(pdf_path, final_pdf_path)
|
|
230
|
+
|
|
231
|
+
return final_pdf_path
|
|
232
|
+
|
|
233
|
+
except Exception as e:
|
|
234
|
+
logging.error(f"Error in convert_office_to_pdf: {str(e)}")
|
|
235
|
+
raise
|
|
236
|
+
|
|
237
|
+
@staticmethod
|
|
238
|
+
def convert_text_to_pdf(
|
|
239
|
+
text_path: Union[str, Path], output_dir: Optional[str] = None
|
|
240
|
+
) -> Path:
|
|
241
|
+
"""
|
|
242
|
+
Convert text file (.txt, .md) to PDF using ReportLab with full markdown support.
|
|
243
|
+
|
|
244
|
+
Args:
|
|
245
|
+
text_path: Path to the text file
|
|
246
|
+
output_dir: Output directory for the PDF file
|
|
247
|
+
|
|
248
|
+
Returns:
|
|
249
|
+
Path to the generated PDF file
|
|
250
|
+
"""
|
|
251
|
+
try:
|
|
252
|
+
text_path = Path(text_path)
|
|
253
|
+
if not text_path.exists():
|
|
254
|
+
raise FileNotFoundError(f"Text file does not exist: {text_path}")
|
|
255
|
+
|
|
256
|
+
# Supported text formats
|
|
257
|
+
supported_text_formats = {".txt", ".md"}
|
|
258
|
+
if text_path.suffix.lower() not in supported_text_formats:
|
|
259
|
+
raise ValueError(f"Unsupported text format: {text_path.suffix}")
|
|
260
|
+
|
|
261
|
+
# Read the text content
|
|
262
|
+
try:
|
|
263
|
+
with open(text_path, "r", encoding="utf-8") as f:
|
|
264
|
+
text_content = f.read()
|
|
265
|
+
except UnicodeDecodeError:
|
|
266
|
+
# Try with different encodings
|
|
267
|
+
for encoding in ["gbk", "latin-1", "cp1252"]:
|
|
268
|
+
try:
|
|
269
|
+
with open(text_path, "r", encoding=encoding) as f:
|
|
270
|
+
text_content = f.read()
|
|
271
|
+
logging.info(f"Successfully read file with {encoding} encoding")
|
|
272
|
+
break
|
|
273
|
+
except UnicodeDecodeError:
|
|
274
|
+
continue
|
|
275
|
+
else:
|
|
276
|
+
raise RuntimeError(
|
|
277
|
+
f"Could not decode text file {text_path.name} with any supported encoding"
|
|
278
|
+
)
|
|
279
|
+
|
|
280
|
+
# Prepare output directory
|
|
281
|
+
if output_dir:
|
|
282
|
+
base_output_dir = Path(output_dir)
|
|
283
|
+
else:
|
|
284
|
+
base_output_dir = text_path.parent / "pdf_output"
|
|
285
|
+
|
|
286
|
+
base_output_dir.mkdir(parents=True, exist_ok=True)
|
|
287
|
+
pdf_path = base_output_dir / f"{text_path.stem}.pdf"
|
|
288
|
+
|
|
289
|
+
# Convert text to PDF
|
|
290
|
+
logging.info(f"Converting {text_path.name} to PDF...")
|
|
291
|
+
|
|
292
|
+
try:
|
|
293
|
+
from reportlab.lib.pagesizes import A4
|
|
294
|
+
from reportlab.platypus import SimpleDocTemplate, Paragraph, Spacer
|
|
295
|
+
from reportlab.lib.styles import getSampleStyleSheet, ParagraphStyle
|
|
296
|
+
from reportlab.lib.units import inch
|
|
297
|
+
from reportlab.pdfbase import pdfmetrics
|
|
298
|
+
|
|
299
|
+
# Create PDF document
|
|
300
|
+
doc = SimpleDocTemplate(
|
|
301
|
+
str(pdf_path),
|
|
302
|
+
pagesize=A4,
|
|
303
|
+
leftMargin=inch,
|
|
304
|
+
rightMargin=inch,
|
|
305
|
+
topMargin=inch,
|
|
306
|
+
bottomMargin=inch,
|
|
307
|
+
)
|
|
308
|
+
|
|
309
|
+
# Get styles
|
|
310
|
+
styles = getSampleStyleSheet()
|
|
311
|
+
normal_style = styles["Normal"]
|
|
312
|
+
heading_style = styles["Heading1"]
|
|
313
|
+
|
|
314
|
+
# Try to register a font that supports Chinese characters
|
|
315
|
+
try:
|
|
316
|
+
# Try to use system fonts that support Chinese
|
|
317
|
+
system = platform.system()
|
|
318
|
+
if system == "Windows":
|
|
319
|
+
# Try common Windows fonts
|
|
320
|
+
for font_name in ["SimSun", "SimHei", "Microsoft YaHei"]:
|
|
321
|
+
try:
|
|
322
|
+
from reportlab.pdfbase.cidfonts import (
|
|
323
|
+
UnicodeCIDFont,
|
|
324
|
+
)
|
|
325
|
+
|
|
326
|
+
pdfmetrics.registerFont(UnicodeCIDFont(font_name)) # type: ignore
|
|
327
|
+
normal_style.fontName = font_name
|
|
328
|
+
heading_style.fontName = font_name
|
|
329
|
+
break
|
|
330
|
+
except Exception:
|
|
331
|
+
continue
|
|
332
|
+
elif system == "Darwin": # macOS
|
|
333
|
+
for font_name in ["STSong-Light", "STHeiti"]:
|
|
334
|
+
try:
|
|
335
|
+
from reportlab.pdfbase.cidfonts import (
|
|
336
|
+
UnicodeCIDFont,
|
|
337
|
+
)
|
|
338
|
+
|
|
339
|
+
pdfmetrics.registerFont(UnicodeCIDFont(font_name)) # type: ignore
|
|
340
|
+
normal_style.fontName = font_name
|
|
341
|
+
heading_style.fontName = font_name
|
|
342
|
+
break
|
|
343
|
+
except Exception:
|
|
344
|
+
continue
|
|
345
|
+
except Exception:
|
|
346
|
+
pass # Use default fonts if Chinese font setup fails
|
|
347
|
+
|
|
348
|
+
# Build content
|
|
349
|
+
story = []
|
|
350
|
+
|
|
351
|
+
# Handle markdown or plain text
|
|
352
|
+
if text_path.suffix.lower() == ".md":
|
|
353
|
+
# Handle markdown content - simplified implementation
|
|
354
|
+
lines = text_content.split("\n")
|
|
355
|
+
for line in lines:
|
|
356
|
+
line = line.strip()
|
|
357
|
+
if not line:
|
|
358
|
+
story.append(Spacer(1, 12))
|
|
359
|
+
continue
|
|
360
|
+
|
|
361
|
+
# Headers
|
|
362
|
+
if line.startswith("#"):
|
|
363
|
+
level = len(line) - len(line.lstrip("#"))
|
|
364
|
+
header_text = line.lstrip("#").strip()
|
|
365
|
+
if header_text:
|
|
366
|
+
header_style = ParagraphStyle(
|
|
367
|
+
name=f"Heading{level}",
|
|
368
|
+
parent=heading_style,
|
|
369
|
+
fontSize=max(16 - level, 10),
|
|
370
|
+
spaceAfter=8,
|
|
371
|
+
spaceBefore=16 if level <= 2 else 12,
|
|
372
|
+
)
|
|
373
|
+
story.append(Paragraph(header_text, header_style))
|
|
374
|
+
else:
|
|
375
|
+
# Regular text
|
|
376
|
+
processed_line = PDFConverter._process_inline_markdown(line)
|
|
377
|
+
story.append(Paragraph(processed_line, normal_style))
|
|
378
|
+
story.append(Spacer(1, 6))
|
|
379
|
+
else:
|
|
380
|
+
# Handle plain text files (.txt)
|
|
381
|
+
logging.info(
|
|
382
|
+
f"Processing plain text file with {len(text_content)} characters..."
|
|
383
|
+
)
|
|
384
|
+
|
|
385
|
+
# Split text into lines and process each line
|
|
386
|
+
lines = text_content.split("\n")
|
|
387
|
+
line_count = 0
|
|
388
|
+
|
|
389
|
+
for line in lines:
|
|
390
|
+
line = line.rstrip()
|
|
391
|
+
line_count += 1
|
|
392
|
+
|
|
393
|
+
# Empty lines
|
|
394
|
+
if not line.strip():
|
|
395
|
+
story.append(Spacer(1, 6))
|
|
396
|
+
continue
|
|
397
|
+
|
|
398
|
+
# Regular text lines
|
|
399
|
+
# Escape special characters for ReportLab
|
|
400
|
+
safe_line = (
|
|
401
|
+
line.replace("&", "&")
|
|
402
|
+
.replace("<", "<")
|
|
403
|
+
.replace(">", ">")
|
|
404
|
+
)
|
|
405
|
+
|
|
406
|
+
# Create paragraph
|
|
407
|
+
story.append(Paragraph(safe_line, normal_style))
|
|
408
|
+
story.append(Spacer(1, 3))
|
|
409
|
+
|
|
410
|
+
logging.info(f"Added {line_count} lines to PDF")
|
|
411
|
+
|
|
412
|
+
# If no content was added, add a placeholder
|
|
413
|
+
if not story:
|
|
414
|
+
story.append(Paragraph("(Empty text file)", normal_style))
|
|
415
|
+
|
|
416
|
+
# Build PDF
|
|
417
|
+
doc.build(story)
|
|
418
|
+
logging.info(
|
|
419
|
+
f"Successfully converted {text_path.name} to PDF ({pdf_path.stat().st_size / 1024:.1f} KB)"
|
|
420
|
+
)
|
|
421
|
+
|
|
422
|
+
except ImportError:
|
|
423
|
+
raise RuntimeError(
|
|
424
|
+
"reportlab is required for text-to-PDF conversion. "
|
|
425
|
+
"Please install it using: pip install reportlab"
|
|
426
|
+
)
|
|
427
|
+
except Exception as e:
|
|
428
|
+
raise RuntimeError(
|
|
429
|
+
f"Failed to convert text file {text_path.name} to PDF: {str(e)}"
|
|
430
|
+
)
|
|
431
|
+
|
|
432
|
+
# Validate the generated PDF
|
|
433
|
+
if not pdf_path.exists() or pdf_path.stat().st_size < 100:
|
|
434
|
+
raise RuntimeError(
|
|
435
|
+
f"PDF conversion failed for {text_path.name} - generated PDF is empty or corrupted."
|
|
436
|
+
)
|
|
437
|
+
|
|
438
|
+
return pdf_path
|
|
439
|
+
|
|
440
|
+
except Exception as e:
|
|
441
|
+
logging.error(f"Error in convert_text_to_pdf: {str(e)}")
|
|
442
|
+
raise
|
|
443
|
+
|
|
444
|
+
@staticmethod
|
|
445
|
+
def _process_inline_markdown(text: str) -> str:
|
|
446
|
+
"""
|
|
447
|
+
Process inline markdown formatting (bold, italic, code, links)
|
|
448
|
+
|
|
449
|
+
Args:
|
|
450
|
+
text: Raw text with markdown formatting
|
|
451
|
+
|
|
452
|
+
Returns:
|
|
453
|
+
Text with ReportLab markup
|
|
454
|
+
"""
|
|
455
|
+
import re
|
|
456
|
+
|
|
457
|
+
# Escape special characters for ReportLab
|
|
458
|
+
text = text.replace("&", "&").replace("<", "<").replace(">", ">")
|
|
459
|
+
|
|
460
|
+
# Bold text: **text** or __text__
|
|
461
|
+
text = re.sub(r"\*\*(.*?)\*\*", r"<b>\1</b>", text)
|
|
462
|
+
text = re.sub(r"__(.*?)__", r"<b>\1</b>", text)
|
|
463
|
+
|
|
464
|
+
# Italic text: *text* or _text_ (but not in the middle of words)
|
|
465
|
+
text = re.sub(r"(?<!\w)\*([^*\n]+?)\*(?!\w)", r"<i>\1</i>", text)
|
|
466
|
+
text = re.sub(r"(?<!\w)_([^_\n]+?)_(?!\w)", r"<i>\1</i>", text)
|
|
467
|
+
|
|
468
|
+
# Inline code: `code`
|
|
469
|
+
text = re.sub(
|
|
470
|
+
r"`([^`]+?)`",
|
|
471
|
+
r'<font name="Courier" size="9" color="darkred">\1</font>',
|
|
472
|
+
text,
|
|
473
|
+
)
|
|
474
|
+
|
|
475
|
+
# Links: [text](url) - convert to text with URL annotation
|
|
476
|
+
def link_replacer(match):
|
|
477
|
+
link_text = match.group(1)
|
|
478
|
+
url = match.group(2)
|
|
479
|
+
return f'<link href="{url}" color="blue"><u>{link_text}</u></link>'
|
|
480
|
+
|
|
481
|
+
text = re.sub(r"\[([^\]]+?)\]\(([^)]+?)\)", link_replacer, text)
|
|
482
|
+
|
|
483
|
+
# Strikethrough: ~~text~~
|
|
484
|
+
text = re.sub(r"~~(.*?)~~", r"<strike>\1</strike>", text)
|
|
485
|
+
|
|
486
|
+
return text
|
|
487
|
+
|
|
488
|
+
def convert_to_pdf(
|
|
489
|
+
self,
|
|
490
|
+
file_path: Union[str, Path],
|
|
491
|
+
output_dir: Optional[str] = None,
|
|
492
|
+
) -> Path:
|
|
493
|
+
"""
|
|
494
|
+
Convert document to PDF based on file extension
|
|
495
|
+
|
|
496
|
+
Args:
|
|
497
|
+
file_path: Path to the file to be converted
|
|
498
|
+
output_dir: Output directory path
|
|
499
|
+
|
|
500
|
+
Returns:
|
|
501
|
+
Path to the generated PDF file
|
|
502
|
+
"""
|
|
503
|
+
# Convert to Path object
|
|
504
|
+
file_path = Path(file_path)
|
|
505
|
+
if not file_path.exists():
|
|
506
|
+
raise FileNotFoundError(f"File does not exist: {file_path}")
|
|
507
|
+
|
|
508
|
+
# Get file extension
|
|
509
|
+
ext = file_path.suffix.lower()
|
|
510
|
+
|
|
511
|
+
# Choose appropriate conversion method based on file type
|
|
512
|
+
if ext in self.OFFICE_FORMATS:
|
|
513
|
+
return self.convert_office_to_pdf(file_path, output_dir)
|
|
514
|
+
elif ext in self.TEXT_FORMATS:
|
|
515
|
+
return self.convert_text_to_pdf(file_path, output_dir)
|
|
516
|
+
else:
|
|
517
|
+
raise ValueError(
|
|
518
|
+
f"Unsupported file format: {ext}. "
|
|
519
|
+
f"Supported formats: {', '.join(self.OFFICE_FORMATS | self.TEXT_FORMATS)}"
|
|
520
|
+
)
|
|
521
|
+
|
|
522
|
+
def check_dependencies(self) -> dict:
|
|
523
|
+
"""
|
|
524
|
+
Check if required dependencies are available
|
|
525
|
+
|
|
526
|
+
Returns:
|
|
527
|
+
dict: Dictionary with dependency check results
|
|
528
|
+
"""
|
|
529
|
+
results = {
|
|
530
|
+
"libreoffice": False,
|
|
531
|
+
"reportlab": False,
|
|
532
|
+
}
|
|
533
|
+
|
|
534
|
+
# Check LibreOffice
|
|
535
|
+
try:
|
|
536
|
+
subprocess_kwargs: Dict[str, Any] = {
|
|
537
|
+
"capture_output": True,
|
|
538
|
+
"text": True,
|
|
539
|
+
"check": True,
|
|
540
|
+
"encoding": "utf-8",
|
|
541
|
+
"errors": "ignore",
|
|
542
|
+
}
|
|
543
|
+
|
|
544
|
+
if platform.system() == "Windows":
|
|
545
|
+
subprocess_kwargs["creationflags"] = (
|
|
546
|
+
0x08000000 # subprocess.CREATE_NO_WINDOW
|
|
547
|
+
)
|
|
548
|
+
|
|
549
|
+
subprocess.run(["libreoffice", "--version"], **subprocess_kwargs)
|
|
550
|
+
results["libreoffice"] = True
|
|
551
|
+
except (subprocess.CalledProcessError, FileNotFoundError):
|
|
552
|
+
try:
|
|
553
|
+
subprocess.run(["soffice", "--version"], **subprocess_kwargs)
|
|
554
|
+
results["libreoffice"] = True
|
|
555
|
+
except (subprocess.CalledProcessError, FileNotFoundError):
|
|
556
|
+
pass
|
|
557
|
+
|
|
558
|
+
# Check ReportLab
|
|
559
|
+
import importlib.util
|
|
560
|
+
|
|
561
|
+
if importlib.util.find_spec("reportlab") is not None:
|
|
562
|
+
results["reportlab"] = True
|
|
563
|
+
|
|
564
|
+
return results
|
|
565
|
+
|
|
566
|
+
|
|
567
|
+
def main():
|
|
568
|
+
"""
|
|
569
|
+
Main function to run the PDF converter from command line
|
|
570
|
+
"""
|
|
571
|
+
parser = argparse.ArgumentParser(description="Convert documents to PDF format")
|
|
572
|
+
parser.add_argument("file_path", nargs="?", help="Path to the document to convert")
|
|
573
|
+
parser.add_argument("--output", "-o", help="Output directory path")
|
|
574
|
+
parser.add_argument(
|
|
575
|
+
"--check",
|
|
576
|
+
action="store_true",
|
|
577
|
+
help="Check dependencies installation",
|
|
578
|
+
)
|
|
579
|
+
parser.add_argument(
|
|
580
|
+
"--verbose", "-v", action="store_true", help="Enable verbose logging"
|
|
581
|
+
)
|
|
582
|
+
|
|
583
|
+
args = parser.parse_args()
|
|
584
|
+
|
|
585
|
+
# Configure logging
|
|
586
|
+
log_level = logging.INFO if args.verbose else logging.WARNING
|
|
587
|
+
logging.basicConfig(
|
|
588
|
+
level=log_level,
|
|
589
|
+
format="%(asctime)s - %(levelname)s - %(message)s",
|
|
590
|
+
datefmt="%Y-%m-%d %H:%M:%S",
|
|
591
|
+
)
|
|
592
|
+
|
|
593
|
+
# Initialize converter
|
|
594
|
+
converter = PDFConverter()
|
|
595
|
+
|
|
596
|
+
# Check dependencies if requested
|
|
597
|
+
if args.check:
|
|
598
|
+
print("š Checking dependencies...")
|
|
599
|
+
deps = converter.check_dependencies()
|
|
600
|
+
|
|
601
|
+
print(
|
|
602
|
+
f"LibreOffice: {'ā
Available' if deps['libreoffice'] else 'ā Not found'}"
|
|
603
|
+
)
|
|
604
|
+
print(f"ReportLab: {'ā
Available' if deps['reportlab'] else 'ā Not found'}")
|
|
605
|
+
|
|
606
|
+
if not deps["libreoffice"]:
|
|
607
|
+
print("\nš To install LibreOffice:")
|
|
608
|
+
print(" - Windows: Download from https://www.libreoffice.org/")
|
|
609
|
+
print(" - macOS: brew install --cask libreoffice")
|
|
610
|
+
print(" - Ubuntu/Debian: sudo apt-get install libreoffice")
|
|
611
|
+
|
|
612
|
+
if not deps["reportlab"]:
|
|
613
|
+
print("\nš To install ReportLab:")
|
|
614
|
+
print(" pip install reportlab")
|
|
615
|
+
|
|
616
|
+
return 0
|
|
617
|
+
|
|
618
|
+
# If not checking dependencies, file_path is required
|
|
619
|
+
if not args.file_path:
|
|
620
|
+
parser.error("file_path is required when not using --check")
|
|
621
|
+
|
|
622
|
+
try:
|
|
623
|
+
# Convert the file
|
|
624
|
+
output_pdf = converter.convert_to_pdf(
|
|
625
|
+
file_path=args.file_path,
|
|
626
|
+
output_dir=args.output,
|
|
627
|
+
)
|
|
628
|
+
|
|
629
|
+
print(f"ā
Successfully converted to PDF: {output_pdf}")
|
|
630
|
+
print(f"š File size: {output_pdf.stat().st_size / 1024:.1f} KB")
|
|
631
|
+
|
|
632
|
+
except Exception as e:
|
|
633
|
+
print(f"ā Error: {str(e)}")
|
|
634
|
+
return 1
|
|
635
|
+
|
|
636
|
+
return 0
|
|
637
|
+
|
|
638
|
+
|
|
639
|
+
if __name__ == "__main__":
|
|
640
|
+
exit(main())
|