md-to-docx 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,103 @@
1
+ Metadata-Version: 2.2
2
+ Name: md_to_docx
3
+ Version: 0.1.0
4
+ Summary: Convert Markdown to DOCX with support for Mermaid diagrams
5
+ Author-email: Your Name <your.email@example.com>
6
+ License: MIT
7
+ Project-URL: Homepage, https://github.com/yourusername/md_to_docx
8
+ Project-URL: Bug Tracker, https://github.com/yourusername/md_to_docx/issues
9
+ Keywords: markdown,docx,mermaid,converter
10
+ Classifier: Development Status :: 4 - Beta
11
+ Classifier: Intended Audience :: Developers
12
+ Classifier: License :: OSI Approved :: MIT License
13
+ Classifier: Programming Language :: Python :: 3
14
+ Classifier: Programming Language :: Python :: 3.8
15
+ Classifier: Programming Language :: Python :: 3.9
16
+ Classifier: Programming Language :: Python :: 3.10
17
+ Classifier: Programming Language :: Python :: 3.11
18
+ Classifier: Topic :: Text Processing :: Markup
19
+ Requires-Python: >=3.8
20
+ Description-Content-Type: text/markdown
21
+ Requires-Dist: httpx>=0.28.1
22
+ Requires-Dist: mcp[cli]>=1.3.0
23
+ Requires-Dist: markdown>=3.7
24
+ Requires-Dist: python-docx>=1.1.2
25
+ Requires-Dist: beautifulsoup4>=4.13.3
26
+ Requires-Dist: pillow>=11.1.0
27
+ Requires-Dist: requests>=2.32.3
28
+
29
+ # md_to_docx
30
+
31
+ A Python tool to convert Markdown documents to DOCX format with special support for Mermaid diagrams.
32
+
33
+ ## Features
34
+
35
+ - Convert Markdown to DOCX while preserving formatting
36
+ - Render Mermaid diagrams as images in the resulting DOCX file
37
+ - Support for common Markdown elements:
38
+ - Headers
39
+ - Lists (ordered and unordered)
40
+ - Tables
41
+ - Bold and italic text
42
+ - Code blocks
43
+ - And more!
44
+
45
+ ## Installation
46
+
47
+ ```bash
48
+ pip install md_to_docx
49
+ ```
50
+
51
+ ## Usage
52
+
53
+ ### Command Line
54
+
55
+ ```bash
56
+ md-to-docx input.md output.docx
57
+ ```
58
+
59
+ ### Python API
60
+
61
+ ```python
62
+ from md_to_docx import md_to_docx
63
+
64
+ # Convert markdown to docx
65
+ md_to_docx("path/to/input.md", "path/to/output.docx")
66
+
67
+ # Or with string content
68
+ markdown_content = "# Hello World\n\nThis is a test."
69
+ md_to_docx(markdown_content, "output.docx")
70
+ ```
71
+
72
+ ### MCP Server
73
+
74
+ This tool can also be used as an MCP server:
75
+
76
+ ```python
77
+ from mcp.client import Client
78
+
79
+ client = Client()
80
+ result = client.md_to_docx(md_content="# Hello World", output_file="output.docx")
81
+ ```
82
+
83
+ ## Mermaid Support
84
+
85
+ The tool automatically renders Mermaid diagrams found in the Markdown. Example:
86
+
87
+ ````markdown
88
+ ```mermaid
89
+ graph TD
90
+ A[Start] --> B{Decision}
91
+ B -->|Yes| C[Do Something]
92
+ B -->|No| D[Do Nothing]
93
+ ```
94
+ ````
95
+
96
+ ## Requirements
97
+
98
+ - Python 3.8+
99
+ - Dependencies are automatically installed with the package
100
+
101
+ ## License
102
+
103
+ MIT
@@ -0,0 +1,75 @@
1
+ # md_to_docx
2
+
3
+ A Python tool to convert Markdown documents to DOCX format with special support for Mermaid diagrams.
4
+
5
+ ## Features
6
+
7
+ - Convert Markdown to DOCX while preserving formatting
8
+ - Render Mermaid diagrams as images in the resulting DOCX file
9
+ - Support for common Markdown elements:
10
+ - Headers
11
+ - Lists (ordered and unordered)
12
+ - Tables
13
+ - Bold and italic text
14
+ - Code blocks
15
+ - And more!
16
+
17
+ ## Installation
18
+
19
+ ```bash
20
+ pip install md_to_docx
21
+ ```
22
+
23
+ ## Usage
24
+
25
+ ### Command Line
26
+
27
+ ```bash
28
+ md-to-docx input.md output.docx
29
+ ```
30
+
31
+ ### Python API
32
+
33
+ ```python
34
+ from md_to_docx import md_to_docx
35
+
36
+ # Convert markdown to docx
37
+ md_to_docx("path/to/input.md", "path/to/output.docx")
38
+
39
+ # Or with string content
40
+ markdown_content = "# Hello World\n\nThis is a test."
41
+ md_to_docx(markdown_content, "output.docx")
42
+ ```
43
+
44
+ ### MCP Server
45
+
46
+ This tool can also be used as an MCP server:
47
+
48
+ ```python
49
+ from mcp.client import Client
50
+
51
+ client = Client()
52
+ result = client.md_to_docx(md_content="# Hello World", output_file="output.docx")
53
+ ```
54
+
55
+ ## Mermaid Support
56
+
57
+ The tool automatically renders Mermaid diagrams found in the Markdown. Example:
58
+
59
+ ````markdown
60
+ ```mermaid
61
+ graph TD
62
+ A[Start] --> B{Decision}
63
+ B -->|Yes| C[Do Something]
64
+ B -->|No| D[Do Nothing]
65
+ ```
66
+ ````
67
+
68
+ ## Requirements
69
+
70
+ - Python 3.8+
71
+ - Dependencies are automatically installed with the package
72
+
73
+ ## License
74
+
75
+ MIT
@@ -0,0 +1,313 @@
1
+ from typing import Any
2
+ import httpx
3
+ from mcp.server.fastmcp import FastMCP
4
+ import os
5
+ import re
6
+ import subprocess
7
+ import tempfile
8
+ import uuid
9
+ from pathlib import Path
10
+ import argparse
11
+ import requests
12
+ import json
13
+ import base64
14
+ from io import BytesIO
15
+ from bs4 import BeautifulSoup
16
+
17
+ import markdown
18
+ from docx import Document
19
+ from docx.shared import Inches, Pt, RGBColor
20
+ from docx.enum.text import WD_ALIGN_PARAGRAPH
21
+ from PIL import Image
22
+
23
+ # Initialize FastMCP server
24
+ mcp = FastMCP("md_to_docx")
25
+
26
+
27
+ def render_mermaid_to_image(mermaid_code, output_path=None):
28
+ """
29
+ Render Mermaid diagram to an image using multiple methods.
30
+ Returns the path to the saved image.
31
+ """
32
+ # Create a temporary file to save the image if not provided
33
+ if not output_path:
34
+ temp_dir = tempfile.gettempdir()
35
+ output_path = os.path.join(temp_dir, f"mermaid_{uuid.uuid4()}.png")
36
+
37
+ # Method 1: Try using Mermaid.ink API
38
+ try:
39
+ # Encode the Mermaid code for the URL
40
+ encoded_data = {"code": mermaid_code}
41
+ json_str = json.dumps(encoded_data)
42
+ base64_str = base64.urlsafe_b64encode(json_str.encode('utf-8')).decode('utf-8')
43
+
44
+ # Use the Mermaid.ink API
45
+ api_url = f"https://mermaid.ink/img/{base64_str}"
46
+ response = requests.get(api_url)
47
+
48
+ if response.status_code == 200:
49
+ # Save the image
50
+ with open(output_path, 'wb') as f:
51
+ f.write(response.content)
52
+ return output_path
53
+ except Exception as e:
54
+ print(f"Error using Mermaid.ink API: {e}")
55
+
56
+ # Method 2: Try using mermaid-cli if available
57
+ try:
58
+ # Check if mmdc (mermaid-cli) is installed
59
+ subprocess.run(["mmdc", "--version"], capture_output=True, check=True)
60
+
61
+ # Create a temporary file for the Mermaid code
62
+ with tempfile.NamedTemporaryFile(mode='w', suffix='.mmd', delete=False) as temp_mmd:
63
+ temp_mmd.write(mermaid_code)
64
+ temp_mmd_path = temp_mmd.name
65
+
66
+ # Run mmdc to generate the image
67
+ subprocess.run([
68
+ "mmdc",
69
+ "-i", temp_mmd_path,
70
+ "-o", output_path,
71
+ "-b", "transparent"
72
+ ], check=True)
73
+
74
+ # Clean up the temporary Mermaid file
75
+ os.unlink(temp_mmd_path)
76
+
77
+ return output_path
78
+ except (subprocess.SubprocessError, FileNotFoundError) as e:
79
+ print(f"Error using mermaid-cli: {e}")
80
+
81
+ # Method 3: Try using the Kroki API
82
+ try:
83
+ payload = {
84
+ "diagram_source": mermaid_code,
85
+ "diagram_type": "mermaid",
86
+ "output_format": "png"
87
+ }
88
+
89
+ response = requests.post("https://kroki.io/", json=payload)
90
+
91
+ if response.status_code == 200:
92
+ # Save the image
93
+ with open(output_path, 'wb') as f:
94
+ f.write(response.content)
95
+ return output_path
96
+ except Exception as e:
97
+ print(f"Error using Kroki API: {e}")
98
+
99
+ # All methods failed
100
+ print("All rendering methods failed for Mermaid diagram")
101
+ return None
102
+
103
+
104
+ def extract_mermaid_blocks(md_content):
105
+ """Extract Mermaid code blocks from Markdown content."""
106
+ # Pattern to match ```mermaid ... ``` blocks
107
+ pattern = r'```mermaid\s+(.*?)\s+```'
108
+ # Find all matches using re.DOTALL to match across multiple lines
109
+ matches = re.findall(pattern, md_content, re.DOTALL)
110
+ return matches
111
+
112
+
113
+ def html_to_docx(html_content, doc):
114
+ """Convert HTML content to Word document elements."""
115
+ soup = BeautifulSoup(html_content, 'html.parser')
116
+
117
+ # Process elements in order
118
+ for element in soup.find_all(['p', 'h1', 'h2', 'h3', 'h4', 'h5', 'h6', 'ul', 'ol', 'blockquote', 'table']):
119
+ if element.name in ['h1', 'h2', 'h3', 'h4', 'h5', 'h6']:
120
+ paragraph = doc.add_paragraph(element.get_text(strip=True))
121
+ paragraph.style = f'Heading {element.name[1]}'
122
+
123
+ elif element.name == 'p':
124
+ paragraph = doc.add_paragraph(element.get_text(strip=True))
125
+ apply_style_to_paragraph(paragraph, element)
126
+
127
+ elif element.name == 'blockquote':
128
+ paragraph = doc.add_paragraph(element.get_text(strip=True))
129
+ paragraph.style = 'Quote'
130
+
131
+ elif element.name == 'ul':
132
+ for li in element.find_all('li', recursive=False):
133
+ process_list_item(doc, li, 'List Bullet')
134
+
135
+ elif element.name == 'ol':
136
+ for li in element.find_all('li', recursive=False):
137
+ process_list_item(doc, li, 'List Number')
138
+
139
+ elif element.name == 'table':
140
+ process_table(doc, element)
141
+
142
+ return doc
143
+
144
+
145
+ def apply_style_to_paragraph(paragraph, element):
146
+ """Apply HTML styles to a Word paragraph based on the element."""
147
+ if element.name in ['h1', 'h2', 'h3', 'h4', 'h5', 'h6']:
148
+ level = int(element.name[1])
149
+ paragraph.style = f'Heading {level}'
150
+
151
+ if element.name == 'strong' or element.find('strong'):
152
+ for run in paragraph.runs:
153
+ run.bold = True
154
+
155
+ if element.name == 'em' or element.find('em'):
156
+ for run in paragraph.runs:
157
+ run.italic = True
158
+
159
+ if element.name == 'u' or element.find('u'):
160
+ for run in paragraph.runs:
161
+ run.underline = True
162
+
163
+ if element.name == 'code' or element.find('code'):
164
+ for run in paragraph.runs:
165
+ run.font.name = 'Courier New'
166
+
167
+ if element.name == 'center' or element.get('align') == 'center':
168
+ paragraph.alignment = WD_ALIGN_PARAGRAPH.CENTER
169
+
170
+ def process_list_item(doc, li_element, list_style, level=0):
171
+ """Process a list item and its children with proper indentation."""
172
+ # Add the list item with proper style and level
173
+ text = li_element.get_text(strip=True)
174
+ paragraph = doc.add_paragraph(text)
175
+ paragraph.style = list_style
176
+ paragraph.paragraph_format.left_indent = Pt(18 * level) # Indent based on nesting level
177
+
178
+ # Process any nested lists
179
+ nested_ul = li_element.find('ul')
180
+ nested_ol = li_element.find('ol')
181
+
182
+ if nested_ul:
183
+ for nested_li in nested_ul.find_all('li', recursive=False):
184
+ process_list_item(doc, nested_li, 'List Bullet', level + 1)
185
+
186
+ if nested_ol:
187
+ for nested_li in nested_ol.find_all('li', recursive=False):
188
+ process_list_item(doc, nested_li, 'List Number', level + 1)
189
+
190
+ def process_table(doc, table_element):
191
+ """Process a table element and convert it to a Word table."""
192
+ # Find all rows in the table
193
+ rows = table_element.find_all('tr')
194
+ if not rows:
195
+ return
196
+
197
+ # Count the maximum number of cells in any row
198
+ max_cols = 0
199
+ for row in rows:
200
+ cells = row.find_all(['th', 'td'])
201
+ max_cols = max(max_cols, len(cells))
202
+
203
+ if max_cols == 0:
204
+ return
205
+
206
+ # Create the table in the document
207
+ table = doc.add_table(rows=len(rows), cols=max_cols)
208
+ table.style = 'Table Grid'
209
+
210
+ # Fill the table with data
211
+ for i, row in enumerate(rows):
212
+ cells = row.find_all(['th', 'td'])
213
+ for j, cell in enumerate(cells):
214
+ if j < max_cols: # Ensure we don't exceed the table dimensions
215
+ # Get cell text and apply basic formatting
216
+ text = cell.get_text(strip=True)
217
+ table.cell(i, j).text = text
218
+
219
+ # Apply header formatting if it's a header cell
220
+ if cell.name == 'th' or i == 0:
221
+ for paragraph in table.cell(i, j).paragraphs:
222
+ for run in paragraph.runs:
223
+ run.bold = True
224
+
225
+
226
+
227
+ @mcp.tool()
228
+ async def md_to_docx(md_content: str, output_file: str = None):
229
+ """Convert Markdown file to DOCX, rendering Mermaid diagrams as images.
230
+
231
+ Args:
232
+ md_content: Markdown content to convert
233
+ output_file: Optional output file path, defaults to 'output.docx'
234
+ """
235
+ # Read the Markdown file
236
+ # with open(md_file, 'r', encoding='utf-8') as f:
237
+ # md_content = f.read()
238
+
239
+ # Extract Mermaid blocks
240
+ mermaid_blocks = extract_mermaid_blocks(md_content)
241
+
242
+ # Create a new Word document
243
+ doc = Document()
244
+
245
+ # Replace Mermaid blocks with placeholders and keep track of them
246
+ placeholders = []
247
+ for i, block in enumerate(mermaid_blocks):
248
+ placeholder = f"MERMAID_DIAGRAM_{i}"
249
+ placeholders.append(placeholder)
250
+ md_content = md_content.replace(f"```mermaid\n{block}\n```", placeholder)
251
+
252
+ # Convert Markdown to HTML with extensions
253
+ html_content = markdown.markdown(
254
+ md_content,
255
+ extensions=[
256
+ 'markdown.extensions.extra',
257
+ 'markdown.extensions.codehilite',
258
+ 'markdown.extensions.tables',
259
+ 'markdown.extensions.toc'
260
+ ]
261
+ )
262
+
263
+ # Split HTML by placeholders
264
+ parts = []
265
+ for part in re.split(f"({'|'.join(placeholders)})", html_content):
266
+ if part in placeholders:
267
+ # This is a placeholder, mark it for later replacement
268
+ parts.append((True, placeholders.index(part)))
269
+ else:
270
+ # This is regular HTML content
271
+ parts.append((False, part))
272
+
273
+ # Process each part
274
+ for is_placeholder, content in parts:
275
+ if is_placeholder:
276
+ # Render the Mermaid diagram
277
+ mermaid_code = mermaid_blocks[content]
278
+ img_path = render_mermaid_to_image(mermaid_code)
279
+
280
+ if img_path:
281
+ # Add the image to the document
282
+ doc.add_picture(img_path, width=Inches(6))
283
+
284
+ # Clean up the temporary image file
285
+ try:
286
+ os.unlink(img_path)
287
+ except:
288
+ pass
289
+ else:
290
+ # If rendering failed, add the Mermaid code as text
291
+ doc.add_paragraph("Failed to render Mermaid diagram:", style='Intense Quote')
292
+ code_para = doc.add_paragraph(mermaid_code)
293
+ code_para.style = 'No Spacing'
294
+ for run in code_para.runs:
295
+ run.font.name = 'Courier New'
296
+ run.font.size = Pt(9)
297
+ else:
298
+ # Add regular content as paragraphs with proper formatting
299
+ if content.strip():
300
+ html_to_docx(content, doc)
301
+
302
+ # Determine output file name if not provided
303
+ if not output_file:
304
+ output_file = 'output.docx'
305
+
306
+ # Save the document
307
+ doc.save(output_file)
308
+ return output_file
309
+
310
+
311
+ if __name__ == "__main__":
312
+ # Initialize and run the server
313
+ mcp.run(transport='stdio')
@@ -0,0 +1,9 @@
1
+ """
2
+ md_to_docx - Convert Markdown to DOCX with support for Mermaid diagrams
3
+ """
4
+
5
+ __version__ = "0.1.0"
6
+
7
+ from .converter import md_to_docx
8
+
9
+ __all__ = ["md_to_docx"]
@@ -0,0 +1,8 @@
1
+ """
2
+ Entry point for running the package directly.
3
+ """
4
+
5
+ from .server import run_server
6
+
7
+ if __name__ == "__main__":
8
+ run_server()
@@ -0,0 +1,50 @@
1
+ """
2
+ Command-line interface for md_to_docx.
3
+ """
4
+
5
+ import argparse
6
+ import sys
7
+ from pathlib import Path
8
+ from .converter import md_to_docx
9
+
10
+
11
+ def main():
12
+ """Main entry point for the CLI."""
13
+ parser = argparse.ArgumentParser(
14
+ description="Convert Markdown to DOCX with Mermaid diagram support"
15
+ )
16
+ parser.add_argument(
17
+ "input_file",
18
+ help="Input Markdown file path"
19
+ )
20
+ parser.add_argument(
21
+ "output_file",
22
+ nargs="?",
23
+ default=None,
24
+ help="Output DOCX file path (default: input filename with .docx extension)"
25
+ )
26
+
27
+ args = parser.parse_args()
28
+
29
+ # Validate input file
30
+ input_path = Path(args.input_file)
31
+ if not input_path.exists():
32
+ print(f"Error: Input file '{input_path}' does not exist.", file=sys.stderr)
33
+ sys.exit(1)
34
+
35
+ # Set default output file if not provided
36
+ if args.output_file is None:
37
+ output_file = input_path.with_suffix(".docx")
38
+ else:
39
+ output_file = args.output_file
40
+
41
+ try:
42
+ result = md_to_docx(str(input_path), str(output_file))
43
+ print(f"Successfully converted '{input_path}' to '{result}'")
44
+ except Exception as e:
45
+ print(f"Error: {e}", file=sys.stderr)
46
+ sys.exit(1)
47
+
48
+
49
+ if __name__ == "__main__":
50
+ main()
@@ -0,0 +1,310 @@
1
+ """
2
+ Core functionality for converting Markdown to DOCX with Mermaid support.
3
+ """
4
+
5
+ import os
6
+ import re
7
+ import subprocess
8
+ import tempfile
9
+ import uuid
10
+ import json
11
+ import base64
12
+ from io import BytesIO
13
+ from pathlib import Path
14
+ import requests
15
+ from bs4 import BeautifulSoup
16
+
17
+ import markdown
18
+ from docx import Document
19
+ from docx.shared import Inches, Pt, RGBColor
20
+ from docx.enum.text import WD_ALIGN_PARAGRAPH
21
+ from PIL import Image
22
+
23
+
24
+ def render_mermaid_to_image(mermaid_code, output_path=None):
25
+ """
26
+ Render Mermaid diagram to an image using multiple methods.
27
+ Returns the path to the saved image.
28
+ """
29
+ # Create a temporary file to save the image if not provided
30
+ if not output_path:
31
+ temp_dir = tempfile.gettempdir()
32
+ output_path = os.path.join(temp_dir, f"mermaid_{uuid.uuid4()}.png")
33
+
34
+ # Method 1: Try using Mermaid.ink API
35
+ try:
36
+ # Encode the Mermaid code for the URL
37
+ encoded_data = {"code": mermaid_code}
38
+ json_str = json.dumps(encoded_data)
39
+ base64_str = base64.urlsafe_b64encode(json_str.encode('utf-8')).decode('utf-8')
40
+
41
+ # Use the Mermaid.ink API
42
+ api_url = f"https://mermaid.ink/img/{base64_str}"
43
+ response = requests.get(api_url)
44
+
45
+ if response.status_code == 200:
46
+ # Save the image
47
+ with open(output_path, 'wb') as f:
48
+ f.write(response.content)
49
+ return output_path
50
+ except Exception as e:
51
+ print(f"Error using Mermaid.ink API: {e}")
52
+
53
+ # Method 2: Try using mermaid-cli if available
54
+ try:
55
+ # Check if mmdc (mermaid-cli) is installed
56
+ subprocess.run(["mmdc", "--version"], capture_output=True, check=True)
57
+
58
+ # Create a temporary file for the Mermaid code
59
+ with tempfile.NamedTemporaryFile(mode='w', suffix='.mmd', delete=False) as temp_mmd:
60
+ temp_mmd.write(mermaid_code)
61
+ temp_mmd_path = temp_mmd.name
62
+
63
+ # Run mmdc to generate the image
64
+ subprocess.run([
65
+ "mmdc",
66
+ "-i", temp_mmd_path,
67
+ "-o", output_path,
68
+ "-b", "transparent"
69
+ ], check=True)
70
+
71
+ # Clean up the temporary Mermaid file
72
+ os.unlink(temp_mmd_path)
73
+
74
+ return output_path
75
+ except (subprocess.SubprocessError, FileNotFoundError) as e:
76
+ print(f"Error using mermaid-cli: {e}")
77
+
78
+ # Method 3: Try using the Kroki API
79
+ try:
80
+ payload = {
81
+ "diagram_source": mermaid_code,
82
+ "diagram_type": "mermaid",
83
+ "output_format": "png"
84
+ }
85
+
86
+ response = requests.post("https://kroki.io/", json=payload)
87
+
88
+ if response.status_code == 200:
89
+ # Save the image
90
+ with open(output_path, 'wb') as f:
91
+ f.write(response.content)
92
+ return output_path
93
+ except Exception as e:
94
+ print(f"Error using Kroki API: {e}")
95
+
96
+ # All methods failed
97
+ print("All rendering methods failed for Mermaid diagram")
98
+ return None
99
+
100
+
101
+ def extract_mermaid_blocks(md_content):
102
+ """Extract Mermaid code blocks from Markdown content."""
103
+ # Pattern to match ```mermaid ... ``` blocks
104
+ pattern = r'```mermaid\s+(.*?)\s+```'
105
+ # Find all matches using re.DOTALL to match across multiple lines
106
+ matches = re.findall(pattern, md_content, re.DOTALL)
107
+ return matches
108
+
109
+
110
+ def html_to_docx(html_content, doc):
111
+ """Convert HTML content to Word document elements."""
112
+ soup = BeautifulSoup(html_content, 'html.parser')
113
+
114
+ # Process elements in order
115
+ for element in soup.find_all(['p', 'h1', 'h2', 'h3', 'h4', 'h5', 'h6', 'ul', 'ol', 'blockquote', 'table']):
116
+ if element.name in ['h1', 'h2', 'h3', 'h4', 'h5', 'h6']:
117
+ paragraph = doc.add_paragraph(element.get_text(strip=True))
118
+ paragraph.style = f'Heading {element.name[1]}'
119
+
120
+ elif element.name == 'p':
121
+ paragraph = doc.add_paragraph(element.get_text(strip=True))
122
+ apply_style_to_paragraph(paragraph, element)
123
+
124
+ elif element.name == 'blockquote':
125
+ paragraph = doc.add_paragraph(element.get_text(strip=True))
126
+ paragraph.style = 'Quote'
127
+
128
+ elif element.name == 'ul':
129
+ for li in element.find_all('li', recursive=False):
130
+ process_list_item(doc, li, 'List Bullet')
131
+
132
+ elif element.name == 'ol':
133
+ for li in element.find_all('li', recursive=False):
134
+ process_list_item(doc, li, 'List Number')
135
+
136
+ elif element.name == 'table':
137
+ process_table(doc, element)
138
+
139
+ return doc
140
+
141
+
142
+ def apply_style_to_paragraph(paragraph, element):
143
+ """Apply HTML styles to a Word paragraph based on the element."""
144
+ if element.name in ['h1', 'h2', 'h3', 'h4', 'h5', 'h6']:
145
+ level = int(element.name[1])
146
+ paragraph.style = f'Heading {level}'
147
+
148
+ if element.name == 'strong' or element.find('strong'):
149
+ for run in paragraph.runs:
150
+ run.bold = True
151
+
152
+ if element.name == 'em' or element.find('em'):
153
+ for run in paragraph.runs:
154
+ run.italic = True
155
+
156
+ if element.name == 'u' or element.find('u'):
157
+ for run in paragraph.runs:
158
+ run.underline = True
159
+
160
+ if element.name == 'code' or element.find('code'):
161
+ for run in paragraph.runs:
162
+ run.font.name = 'Courier New'
163
+
164
+ if element.name == 'center' or element.get('align') == 'center':
165
+ paragraph.alignment = WD_ALIGN_PARAGRAPH.CENTER
166
+
167
+
168
+ def process_list_item(doc, li_element, list_style, level=0):
169
+ """Process a list item and its children with proper indentation."""
170
+ # Add the list item with proper style and level
171
+ text = li_element.get_text(strip=True)
172
+ paragraph = doc.add_paragraph(text)
173
+ paragraph.style = list_style
174
+ paragraph.paragraph_format.left_indent = Pt(18 * level) # Indent based on nesting level
175
+
176
+ # Process any nested lists
177
+ nested_ul = li_element.find('ul')
178
+ nested_ol = li_element.find('ol')
179
+
180
+ if nested_ul:
181
+ for nested_li in nested_ul.find_all('li', recursive=False):
182
+ process_list_item(doc, nested_li, 'List Bullet', level + 1)
183
+
184
+ if nested_ol:
185
+ for nested_li in nested_ol.find_all('li', recursive=False):
186
+ process_list_item(doc, nested_li, 'List Number', level + 1)
187
+
188
+
189
+ def process_table(doc, table_element):
190
+ """Process a table element and convert it to a Word table."""
191
+ # Find all rows in the table
192
+ rows = table_element.find_all('tr')
193
+ if not rows:
194
+ return
195
+
196
+ # Count the maximum number of cells in any row
197
+ max_cols = 0
198
+ for row in rows:
199
+ cells = row.find_all(['th', 'td'])
200
+ max_cols = max(max_cols, len(cells))
201
+
202
+ if max_cols == 0:
203
+ return
204
+
205
+ # Create the table in the document
206
+ table = doc.add_table(rows=len(rows), cols=max_cols)
207
+ table.style = 'Table Grid'
208
+
209
+ # Fill the table with data
210
+ for i, row in enumerate(rows):
211
+ cells = row.find_all(['th', 'td'])
212
+ for j, cell in enumerate(cells):
213
+ if j < max_cols: # Ensure we don't exceed the table dimensions
214
+ # Get cell text and apply basic formatting
215
+ text = cell.get_text(strip=True)
216
+ table.cell(i, j).text = text
217
+
218
+ # Apply header formatting if it's a header cell
219
+ if cell.name == 'th' or i == 0:
220
+ for paragraph in table.cell(i, j).paragraphs:
221
+ for run in paragraph.runs:
222
+ run.bold = True
223
+
224
+
225
+ def md_to_docx(md_content, output_file=None):
226
+ """
227
+ Convert Markdown content to DOCX, rendering Mermaid diagrams as images.
228
+
229
+ Args:
230
+ md_content: Markdown content or file path to convert
231
+ output_file: Optional output file path, defaults to 'output.docx'
232
+
233
+ Returns:
234
+ Path to the generated DOCX file
235
+ """
236
+ # Check if md_content is a file path
237
+ if os.path.isfile(md_content):
238
+ with open(md_content, 'r', encoding='utf-8') as f:
239
+ md_content = f.read()
240
+
241
+ # Extract Mermaid blocks
242
+ mermaid_blocks = extract_mermaid_blocks(md_content)
243
+
244
+ # Create a new Word document
245
+ doc = Document()
246
+
247
+ # Replace Mermaid blocks with placeholders and keep track of them
248
+ placeholders = []
249
+ for i, block in enumerate(mermaid_blocks):
250
+ placeholder = f"MERMAID_DIAGRAM_{i}"
251
+ placeholders.append(placeholder)
252
+ md_content = md_content.replace(f"```mermaid\n{block}\n```", placeholder)
253
+
254
+ # Convert Markdown to HTML with extensions
255
+ html_content = markdown.markdown(
256
+ md_content,
257
+ extensions=[
258
+ 'markdown.extensions.extra',
259
+ 'markdown.extensions.codehilite',
260
+ 'markdown.extensions.tables',
261
+ 'markdown.extensions.toc'
262
+ ]
263
+ )
264
+
265
+ # Split HTML by placeholders
266
+ parts = []
267
+ for part in re.split(f"({'|'.join(placeholders)})", html_content):
268
+ if part in placeholders:
269
+ # This is a placeholder, mark it for later replacement
270
+ parts.append((True, placeholders.index(part)))
271
+ else:
272
+ # This is regular HTML content
273
+ parts.append((False, part))
274
+
275
+ # Process each part
276
+ for is_placeholder, content in parts:
277
+ if is_placeholder:
278
+ # Render the Mermaid diagram
279
+ mermaid_code = mermaid_blocks[content]
280
+ img_path = render_mermaid_to_image(mermaid_code)
281
+
282
+ if img_path:
283
+ # Add the image to the document
284
+ doc.add_picture(img_path, width=Inches(6))
285
+
286
+ # Clean up the temporary image file
287
+ try:
288
+ os.unlink(img_path)
289
+ except:
290
+ pass
291
+ else:
292
+ # If rendering failed, add the Mermaid code as text
293
+ doc.add_paragraph("Failed to render Mermaid diagram:", style='Intense Quote')
294
+ code_para = doc.add_paragraph(mermaid_code)
295
+ code_para.style = 'No Spacing'
296
+ for run in code_para.runs:
297
+ run.font.name = 'Courier New'
298
+ run.font.size = Pt(9)
299
+ else:
300
+ # Add regular content as paragraphs with proper formatting
301
+ if content.strip():
302
+ html_to_docx(content, doc)
303
+
304
+ # Determine output file name if not provided
305
+ if not output_file:
306
+ output_file = 'output.docx'
307
+
308
+ # Save the document
309
+ doc.save(output_file)
310
+ return output_file
@@ -0,0 +1,32 @@
1
+ """
2
+ MCP server implementation for md_to_docx.
3
+ """
4
+
5
+ from mcp.server.fastmcp import FastMCP
6
+ from .converter import md_to_docx as convert_md_to_docx
7
+
8
+ # Initialize FastMCP server
9
+ mcp = FastMCP("md_to_docx")
10
+
11
+
12
+ @mcp.tool()
13
+ async def md_to_docx(md_content: str, output_file: str = None):
14
+ """Convert Markdown content to DOCX, rendering Mermaid diagrams as images.
15
+
16
+ Args:
17
+ md_content: Markdown content to convert
18
+ output_file: Optional output file path, defaults to 'output.docx'
19
+
20
+ Returns:
21
+ Path to the generated DOCX file
22
+ """
23
+ return convert_md_to_docx(md_content, output_file)
24
+
25
+
26
+ def run_server():
27
+ """Run the MCP server."""
28
+ mcp.run(transport='stdio')
29
+
30
+
31
+ if __name__ == "__main__":
32
+ run_server()
@@ -0,0 +1,103 @@
1
+ Metadata-Version: 2.2
2
+ Name: md_to_docx
3
+ Version: 0.1.0
4
+ Summary: Convert Markdown to DOCX with support for Mermaid diagrams
5
+ Author-email: Your Name <your.email@example.com>
6
+ License: MIT
7
+ Project-URL: Homepage, https://github.com/yourusername/md_to_docx
8
+ Project-URL: Bug Tracker, https://github.com/yourusername/md_to_docx/issues
9
+ Keywords: markdown,docx,mermaid,converter
10
+ Classifier: Development Status :: 4 - Beta
11
+ Classifier: Intended Audience :: Developers
12
+ Classifier: License :: OSI Approved :: MIT License
13
+ Classifier: Programming Language :: Python :: 3
14
+ Classifier: Programming Language :: Python :: 3.8
15
+ Classifier: Programming Language :: Python :: 3.9
16
+ Classifier: Programming Language :: Python :: 3.10
17
+ Classifier: Programming Language :: Python :: 3.11
18
+ Classifier: Topic :: Text Processing :: Markup
19
+ Requires-Python: >=3.8
20
+ Description-Content-Type: text/markdown
21
+ Requires-Dist: httpx>=0.28.1
22
+ Requires-Dist: mcp[cli]>=1.3.0
23
+ Requires-Dist: markdown>=3.7
24
+ Requires-Dist: python-docx>=1.1.2
25
+ Requires-Dist: beautifulsoup4>=4.13.3
26
+ Requires-Dist: pillow>=11.1.0
27
+ Requires-Dist: requests>=2.32.3
28
+
29
+ # md_to_docx
30
+
31
+ A Python tool to convert Markdown documents to DOCX format with special support for Mermaid diagrams.
32
+
33
+ ## Features
34
+
35
+ - Convert Markdown to DOCX while preserving formatting
36
+ - Render Mermaid diagrams as images in the resulting DOCX file
37
+ - Support for common Markdown elements:
38
+ - Headers
39
+ - Lists (ordered and unordered)
40
+ - Tables
41
+ - Bold and italic text
42
+ - Code blocks
43
+ - And more!
44
+
45
+ ## Installation
46
+
47
+ ```bash
48
+ pip install md_to_docx
49
+ ```
50
+
51
+ ## Usage
52
+
53
+ ### Command Line
54
+
55
+ ```bash
56
+ md-to-docx input.md output.docx
57
+ ```
58
+
59
+ ### Python API
60
+
61
+ ```python
62
+ from md_to_docx import md_to_docx
63
+
64
+ # Convert markdown to docx
65
+ md_to_docx("path/to/input.md", "path/to/output.docx")
66
+
67
+ # Or with string content
68
+ markdown_content = "# Hello World\n\nThis is a test."
69
+ md_to_docx(markdown_content, "output.docx")
70
+ ```
71
+
72
+ ### MCP Server
73
+
74
+ This tool can also be used as an MCP server:
75
+
76
+ ```python
77
+ from mcp.client import Client
78
+
79
+ client = Client()
80
+ result = client.md_to_docx(md_content="# Hello World", output_file="output.docx")
81
+ ```
82
+
83
+ ## Mermaid Support
84
+
85
+ The tool automatically renders Mermaid diagrams found in the Markdown. Example:
86
+
87
+ ````markdown
88
+ ```mermaid
89
+ graph TD
90
+ A[Start] --> B{Decision}
91
+ B -->|Yes| C[Do Something]
92
+ B -->|No| D[Do Nothing]
93
+ ```
94
+ ````
95
+
96
+ ## Requirements
97
+
98
+ - Python 3.8+
99
+ - Dependencies are automatically installed with the package
100
+
101
+ ## License
102
+
103
+ MIT
@@ -0,0 +1,14 @@
1
+ README.md
2
+ main.py
3
+ pyproject.toml
4
+ md_to_docx/__init__.py
5
+ md_to_docx/__main__.py
6
+ md_to_docx/cli.py
7
+ md_to_docx/converter.py
8
+ md_to_docx/server.py
9
+ md_to_docx.egg-info/PKG-INFO
10
+ md_to_docx.egg-info/SOURCES.txt
11
+ md_to_docx.egg-info/dependency_links.txt
12
+ md_to_docx.egg-info/entry_points.txt
13
+ md_to_docx.egg-info/requires.txt
14
+ md_to_docx.egg-info/top_level.txt
@@ -0,0 +1,2 @@
1
+ [console_scripts]
2
+ md-to-docx = md_to_docx.cli:main
@@ -0,0 +1,7 @@
1
+ httpx>=0.28.1
2
+ mcp[cli]>=1.3.0
3
+ markdown>=3.7
4
+ python-docx>=1.1.2
5
+ beautifulsoup4>=4.13.3
6
+ pillow>=11.1.0
7
+ requests>=2.32.3
@@ -0,0 +1 @@
1
+ md_to_docx
@@ -0,0 +1,42 @@
1
+ [build-system]
2
+ requires = ["setuptools>=61.0"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "md_to_docx"
7
+ version = "0.1.0"
8
+ description = "Convert Markdown to DOCX with support for Mermaid diagrams"
9
+ readme = "README.md"
10
+ requires-python = ">=3.8"
11
+ license = {text = "MIT"}
12
+ authors = [
13
+ {name = "Your Name", email = "your.email@example.com"}
14
+ ]
15
+ keywords = ["markdown", "docx", "mermaid", "converter"]
16
+ classifiers = [
17
+ "Development Status :: 4 - Beta",
18
+ "Intended Audience :: Developers",
19
+ "License :: OSI Approved :: MIT License",
20
+ "Programming Language :: Python :: 3",
21
+ "Programming Language :: Python :: 3.8",
22
+ "Programming Language :: Python :: 3.9",
23
+ "Programming Language :: Python :: 3.10",
24
+ "Programming Language :: Python :: 3.11",
25
+ "Topic :: Text Processing :: Markup",
26
+ ]
27
+ dependencies = [
28
+ "httpx>=0.28.1",
29
+ "mcp[cli]>=1.3.0",
30
+ "markdown>=3.7",
31
+ "python-docx>=1.1.2",
32
+ "beautifulsoup4>=4.13.3",
33
+ "pillow>=11.1.0",
34
+ "requests>=2.32.3",
35
+ ]
36
+
37
+ [project.urls]
38
+ "Homepage" = "https://github.com/yourusername/md_to_docx"
39
+ "Bug Tracker" = "https://github.com/yourusername/md_to_docx/issues"
40
+
41
+ [project.scripts]
42
+ md-to-docx = "md_to_docx.cli:main"
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+