mddoco 2.0.0__tar.gz → 2.0.3__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (36) hide show
  1. {mddoco-2.0.0/src/mddoco.egg-info → mddoco-2.0.3}/PKG-INFO +56 -10
  2. {mddoco-2.0.0 → mddoco-2.0.3}/README.md +51 -9
  3. {mddoco-2.0.0 → mddoco-2.0.3}/pyproject.toml +15 -1
  4. mddoco-2.0.3/src/mddoco/__init__.py +1 -0
  5. {mddoco-2.0.0 → mddoco-2.0.3}/src/mddoco/cli.py +18 -9
  6. {mddoco-2.0.0 → mddoco-2.0.3}/src/mddoco/graph.py +58 -37
  7. mddoco-2.0.3/src/mddoco/pdf.py +78 -0
  8. mddoco-2.0.3/src/mddoco/preprocessor.py +132 -0
  9. {mddoco-2.0.0 → mddoco-2.0.3/src/mddoco.egg-info}/PKG-INFO +56 -10
  10. {mddoco-2.0.0 → mddoco-2.0.3}/src/mddoco.egg-info/SOURCES.txt +2 -1
  11. {mddoco-2.0.0 → mddoco-2.0.3}/src/mddoco.egg-info/requires.txt +5 -0
  12. mddoco-2.0.3/tests/test_preprocessor.py +145 -0
  13. mddoco-2.0.0/src/mddoco/__init__.py +0 -1
  14. mddoco-2.0.0/src/mddoco/pdf.py +0 -39
  15. mddoco-2.0.0/src/mddoco/preprocessor.py +0 -32
  16. {mddoco-2.0.0 → mddoco-2.0.3}/LICENSE +0 -0
  17. {mddoco-2.0.0 → mddoco-2.0.3}/setup.cfg +0 -0
  18. {mddoco-2.0.0 → mddoco-2.0.3}/src/mddoco/converter.py +0 -0
  19. {mddoco-2.0.0 → mddoco-2.0.3}/src/mddoco/mermaid.py +0 -0
  20. {mddoco-2.0.0 → mddoco-2.0.3}/src/mddoco/renderer.py +0 -0
  21. {mddoco-2.0.0 → mddoco-2.0.3}/src/mddoco/scanner.py +0 -0
  22. {mddoco-2.0.0 → mddoco-2.0.3}/src/mddoco/themes/academic-wide.html +0 -0
  23. {mddoco-2.0.0 → mddoco-2.0.3}/src/mddoco/themes/academic.html +0 -0
  24. {mddoco-2.0.0 → mddoco-2.0.3}/src/mddoco/themes/dark-wide.html +0 -0
  25. {mddoco-2.0.0 → mddoco-2.0.3}/src/mddoco/themes/dark.html +0 -0
  26. {mddoco-2.0.0 → mddoco-2.0.3}/src/mddoco/themes/default-wide.html +0 -0
  27. {mddoco-2.0.0 → mddoco-2.0.3}/src/mddoco/themes/default.html +0 -0
  28. {mddoco-2.0.0 → mddoco-2.0.3}/src/mddoco/themes/paged-professional.html +0 -0
  29. {mddoco-2.0.0 → mddoco-2.0.3}/src/mddoco/themes/professional-wide.html +0 -0
  30. {mddoco-2.0.0 → mddoco-2.0.3}/src/mddoco/themes/professional.html +0 -0
  31. {mddoco-2.0.0 → mddoco-2.0.3}/src/mddoco/themes/vanilla.html +0 -0
  32. {mddoco-2.0.0 → mddoco-2.0.3}/src/mddoco/toc.py +0 -0
  33. {mddoco-2.0.0 → mddoco-2.0.3}/src/mddoco/writer.py +0 -0
  34. {mddoco-2.0.0 → mddoco-2.0.3}/src/mddoco.egg-info/dependency_links.txt +0 -0
  35. {mddoco-2.0.0 → mddoco-2.0.3}/src/mddoco.egg-info/entry_points.txt +0 -0
  36. {mddoco-2.0.0 → mddoco-2.0.3}/src/mddoco.egg-info/top_level.txt +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: mddoco
3
- Version: 2.0.0
3
+ Version: 2.0.3
4
4
  Summary: Markdown to HTML/PDF document converter
5
5
  Requires-Python: >=3.10
6
6
  Description-Content-Type: text/markdown
@@ -10,6 +10,10 @@ Requires-Dist: markdown>=3.5
10
10
  Requires-Dist: jinja2>=3.1
11
11
  Requires-Dist: playwright>=1.40
12
12
  Requires-Dist: matplotlib>=3.7
13
+ Provides-Extra: dev
14
+ Requires-Dist: pytest>=8.0; extra == "dev"
15
+ Requires-Dist: black>=24.0; extra == "dev"
16
+ Requires-Dist: ruff<0.17,>=0.16; extra == "dev"
13
17
  Dynamic: license-file
14
18
 
15
19
  # mddoco
@@ -20,17 +24,23 @@ A CLI tool that converts markdown files to a single HTML (or future PDF) documen
20
24
 
21
25
  ```bash
22
26
  pip install mddoco
23
- playwright install chromium
24
27
  ```
25
28
 
26
29
  Or for development:
27
30
 
28
31
  ```bash
29
32
  pip install -e .
30
- playwright install chromium
31
33
  ```
32
34
 
33
- > Playwright (Chromium) is required for PDF output only. HTML output works without it.
35
+ > **Playwright (Chromium) is required for PDF output only.** HTML output works without it.
36
+ > The `playwright` Python package is installed automatically, but the Chromium
37
+ > browser it drives is not. The first time you render a PDF, mddoco downloads
38
+ > Chromium for you (~150 MB, one time). To do it ahead of time, or if the
39
+ > automatic download fails, run it yourself:
40
+ >
41
+ > ```bash
42
+ > playwright install chromium
43
+ > ```
34
44
 
35
45
  ## Usage
36
46
 
@@ -79,25 +89,43 @@ All matched markdown files are combined into a single HTML document in the outpu
79
89
 
80
90
  Files ending in `.md.j2` are Jinja2 templates that produce Markdown. They are sorted alongside regular `.md` files (numeric prefixes such as `01_`, `02_` determine order) and converted through the same pipeline.
81
91
 
82
- ### JSON context variables
92
+ ### Context variables (JSON and CSV)
83
93
 
84
- Place `*.json` files in the same input directory. Each file's stem becomes a top-level variable in the Jinja2 context — `report.json` is available as `report`, `config.json` as `config`, and so on.
94
+ Place `*.json` or `*.csv` files in the same input directory. Each file's stem becomes a top-level variable in the Jinja2 context — `report.json` is available as `report`, `people.csv` as `people`, and so on. If both `data.json` and `data.csv` exist, the tool will exit with an error.
85
95
 
86
96
  ```
87
97
  docs/
88
98
  01_intro.md
89
99
  02_summary.md.j2 ← Jinja2 template
90
- report.json ← available as {{ report }} inside *.md.j2 files
100
+ report.json ← available as {{ report }}
101
+ people.csv ← available as {{ people }}
91
102
  ```
92
103
 
93
- Example template (`02_summary.md.j2`):
104
+ **JSON** files are loaded as-is; the variable holds whatever structure the JSON contains.
105
+
106
+ **CSV** files are loaded as a list of row dicts, one dict per row:
94
107
 
95
108
  ```markdown
96
- # Summary for {{ report.project }}
109
+ {% for person in people %}
110
+ - {{ person.name }} ({{ person.role }})
111
+ {% endfor %}
112
+ ```
97
113
 
98
- Generated by: {{ report.author }}
114
+ Cell values containing `;` are automatically split into a list:
115
+
116
+ ```csv
117
+ name,skills
118
+ Alice,python;flask;sql
119
+ Bob,java
99
120
  ```
100
121
 
122
+ ```markdown
123
+ {{ people[0].skills }} {# → ["python", "flask", "sql"] #}
124
+ {{ people[1].skills }} {# → "java" (plain string — no ;) #}
125
+ ```
126
+
127
+ Quoting a field in the CSV prevents `;` splitting — `"python;flask"` remains a single string.
128
+
101
129
  ### Excluding files
102
130
 
103
131
  Files whose name starts with `_` are excluded from scanning. Use this for Jinja2 macro files that should be imported but not rendered as documents:
@@ -109,6 +137,24 @@ docs/
109
137
  02_report.md.j2
110
138
  ```
111
139
 
140
+ ### Using macros
141
+
142
+ Use `{% import %}` or `{% from ... import %}` to make macros from another file callable — **not** `{% include %}`. `{% include %}` injects rendered output only; macros defined in an included file are not visible to the calling template and will raise an `undefined` error.
143
+
144
+ ```jinja
145
+ {# correct — macro is callable after this #}
146
+ {% import "_macros.j2" as macros %}
147
+ {{ macros.val(item) }}
148
+
149
+ {# also correct #}
150
+ {% from "_macros.j2" import val %}
151
+ {{ val(item) }}
152
+
153
+ {# wrong — val will be undefined #}
154
+ {% include "_macros.j2" %}
155
+ {{ val(item) }}
156
+ ```
157
+
112
158
  ## Themes
113
159
 
114
160
  Themes are self-contained Jinja2 HTML files with embedded CSS. Pass a theme name with `--theme NAME`.
@@ -6,17 +6,23 @@ A CLI tool that converts markdown files to a single HTML (or future PDF) documen
6
6
 
7
7
  ```bash
8
8
  pip install mddoco
9
- playwright install chromium
10
9
  ```
11
10
 
12
11
  Or for development:
13
12
 
14
13
  ```bash
15
14
  pip install -e .
16
- playwright install chromium
17
15
  ```
18
16
 
19
- > Playwright (Chromium) is required for PDF output only. HTML output works without it.
17
+ > **Playwright (Chromium) is required for PDF output only.** HTML output works without it.
18
+ > The `playwright` Python package is installed automatically, but the Chromium
19
+ > browser it drives is not. The first time you render a PDF, mddoco downloads
20
+ > Chromium for you (~150 MB, one time). To do it ahead of time, or if the
21
+ > automatic download fails, run it yourself:
22
+ >
23
+ > ```bash
24
+ > playwright install chromium
25
+ > ```
20
26
 
21
27
  ## Usage
22
28
 
@@ -65,25 +71,43 @@ All matched markdown files are combined into a single HTML document in the outpu
65
71
 
66
72
  Files ending in `.md.j2` are Jinja2 templates that produce Markdown. They are sorted alongside regular `.md` files (numeric prefixes such as `01_`, `02_` determine order) and converted through the same pipeline.
67
73
 
68
- ### JSON context variables
74
+ ### Context variables (JSON and CSV)
69
75
 
70
- Place `*.json` files in the same input directory. Each file's stem becomes a top-level variable in the Jinja2 context — `report.json` is available as `report`, `config.json` as `config`, and so on.
76
+ Place `*.json` or `*.csv` files in the same input directory. Each file's stem becomes a top-level variable in the Jinja2 context — `report.json` is available as `report`, `people.csv` as `people`, and so on. If both `data.json` and `data.csv` exist, the tool will exit with an error.
71
77
 
72
78
  ```
73
79
  docs/
74
80
  01_intro.md
75
81
  02_summary.md.j2 ← Jinja2 template
76
- report.json ← available as {{ report }} inside *.md.j2 files
82
+ report.json ← available as {{ report }}
83
+ people.csv ← available as {{ people }}
77
84
  ```
78
85
 
79
- Example template (`02_summary.md.j2`):
86
+ **JSON** files are loaded as-is; the variable holds whatever structure the JSON contains.
87
+
88
+ **CSV** files are loaded as a list of row dicts, one dict per row:
80
89
 
81
90
  ```markdown
82
- # Summary for {{ report.project }}
91
+ {% for person in people %}
92
+ - {{ person.name }} ({{ person.role }})
93
+ {% endfor %}
94
+ ```
83
95
 
84
- Generated by: {{ report.author }}
96
+ Cell values containing `;` are automatically split into a list:
97
+
98
+ ```csv
99
+ name,skills
100
+ Alice,python;flask;sql
101
+ Bob,java
85
102
  ```
86
103
 
104
+ ```markdown
105
+ {{ people[0].skills }} {# → ["python", "flask", "sql"] #}
106
+ {{ people[1].skills }} {# → "java" (plain string — no ;) #}
107
+ ```
108
+
109
+ Quoting a field in the CSV prevents `;` splitting — `"python;flask"` remains a single string.
110
+
87
111
  ### Excluding files
88
112
 
89
113
  Files whose name starts with `_` are excluded from scanning. Use this for Jinja2 macro files that should be imported but not rendered as documents:
@@ -95,6 +119,24 @@ docs/
95
119
  02_report.md.j2
96
120
  ```
97
121
 
122
+ ### Using macros
123
+
124
+ Use `{% import %}` or `{% from ... import %}` to make macros from another file callable — **not** `{% include %}`. `{% include %}` injects rendered output only; macros defined in an included file are not visible to the calling template and will raise an `undefined` error.
125
+
126
+ ```jinja
127
+ {# correct — macro is callable after this #}
128
+ {% import "_macros.j2" as macros %}
129
+ {{ macros.val(item) }}
130
+
131
+ {# also correct #}
132
+ {% from "_macros.j2" import val %}
133
+ {{ val(item) }}
134
+
135
+ {# wrong — val will be undefined #}
136
+ {% include "_macros.j2" %}
137
+ {{ val(item) }}
138
+ ```
139
+
98
140
  ## Themes
99
141
 
100
142
  Themes are self-contained Jinja2 HTML files with embedded CSS. Pass a theme name with `--theme NAME`.
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "mddoco"
7
- version = "2.0.0"
7
+ version = "2.0.3"
8
8
  description = "Markdown to HTML/PDF document converter"
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.10"
@@ -16,6 +16,13 @@ dependencies = [
16
16
  "matplotlib>=3.7",
17
17
  ]
18
18
 
19
+ [project.optional-dependencies]
20
+ dev = [
21
+ "pytest>=8.0",
22
+ "black>=24.0",
23
+ "ruff>=0.16,<0.17",
24
+ ]
25
+
19
26
  [project.scripts]
20
27
  mddoco = "mddoco.cli:main"
21
28
 
@@ -25,9 +32,16 @@ where = ["src"]
25
32
  [tool.setuptools.package-data]
26
33
  mddoco = ["themes/*.html"]
27
34
 
35
+ [tool.black]
36
+ line-length = 88
37
+ target-version = ["py310"]
38
+
28
39
  [tool.ruff]
29
40
  src = ["src"]
30
41
  line-length = 88
31
42
 
32
43
  [tool.ruff.lint]
33
44
  select = ["E", "F", "I", "UP"]
45
+
46
+ [tool.pytest.ini_options]
47
+ testpaths = ["tests"]
@@ -0,0 +1 @@
1
+ __version__ = "2.0.3"
@@ -5,7 +5,7 @@ import click
5
5
 
6
6
  from mddoco.converter import convert_files
7
7
  from mddoco.pdf import html_to_pdf
8
- from mddoco.preprocessor import load_json_context
8
+ from mddoco.preprocessor import load_context
9
9
  from mddoco.renderer import render_html
10
10
  from mddoco.scanner import find_markdown_files
11
11
  from mddoco.toc import combine_toc
@@ -15,31 +15,40 @@ from mddoco.writer import write_output
15
15
  @click.command()
16
16
  @click.argument("input_path", type=click.Path(exists=True, path_type=Path))
17
17
  @click.option(
18
- "--output", "-o", "output_path",
19
- default=".", show_default=True,
18
+ "--output",
19
+ "-o",
20
+ "output_path",
21
+ default=".",
22
+ show_default=True,
20
23
  type=click.Path(path_type=Path),
21
24
  help="Output directory, or explicit output file path.",
22
25
  )
23
26
  @click.option(
24
- "--format", "-f", "fmt",
25
- default="html", show_default=True,
27
+ "--format",
28
+ "-f",
29
+ "fmt",
30
+ default="html",
31
+ show_default=True,
26
32
  type=click.Choice(["html", "pdf"], case_sensitive=False),
27
33
  help="Output format.",
28
34
  )
29
35
  @click.option("--title", "-t", default=None, help="Document title.")
30
36
  @click.option(
31
37
  "--theme",
32
- default="default", show_default=True,
38
+ default="default",
39
+ show_default=True,
33
40
  help="Theme name to use for rendering.",
34
41
  )
35
42
  @click.option(
36
43
  "--toc/--no-toc",
37
- default=False, show_default=True,
44
+ default=False,
45
+ show_default=True,
38
46
  help="Generate a table of contents.",
39
47
  )
40
48
  @click.option(
41
49
  "--toc-depth",
42
- default=3, show_default=True,
50
+ default=3,
51
+ show_default=True,
43
52
  type=click.IntRange(1, 6),
44
53
  help="Maximum heading depth included in the TOC.",
45
54
  )
@@ -63,7 +72,7 @@ def main(
63
72
  click.echo(f"Found {len(md_files)} markdown file(s).")
64
73
 
65
74
  try:
66
- context = load_json_context(input_path)
75
+ context = load_context(input_path)
67
76
  raw = convert_files(md_files, toc=toc, toc_depth=toc_depth, context=context)
68
77
  except Exception as exc:
69
78
  raise click.ClickException(str(exc)) from exc
@@ -6,84 +6,105 @@ import re
6
6
  from html import unescape
7
7
 
8
8
  import matplotlib
9
- matplotlib.use('Agg') # non-interactive backend — must be set before importing pyplot
9
+
10
+ matplotlib.use("Agg") # non-interactive backend — must be set before importing pyplot
10
11
  import matplotlib.pyplot as plt
11
12
 
12
13
  log = logging.getLogger(__name__)
13
14
 
14
- _JSON_TO_PY = re.compile(r'\b(true|false|null)\b')
15
- _JSON_TO_PY_MAP = {'true': 'True', 'false': 'False', 'null': 'None'}
15
+ _JSON_TO_PY = re.compile(r"\b(true|false|null)\b")
16
+ _JSON_TO_PY_MAP = {"true": "True", "false": "False", "null": "None"}
16
17
 
17
18
  _DEFAULT_COLOURS = [
18
- '#2d6cbe', '#e74c3c', '#27ae60', '#e67e22',
19
- '#8e44ad', '#16a085', '#2c3e50', '#d35400',
19
+ "#2d6cbe",
20
+ "#e74c3c",
21
+ "#27ae60",
22
+ "#e67e22",
23
+ "#8e44ad",
24
+ "#16a085",
25
+ "#2c3e50",
26
+ "#d35400",
20
27
  ]
21
28
 
22
29
 
23
30
  def graph_it(data, output=None):
24
31
  dpi = 100
25
- width_px = data.get('width_px', 640)
26
- height_px = data.get('height_px', 480)
32
+ width_px = data.get("width_px", 640)
33
+ height_px = data.get("height_px", 480)
27
34
  fig, ax = plt.subplots(figsize=(width_px / dpi, height_px / dpi), dpi=dpi)
28
35
 
29
- orientation = data.get('orientation', 'vertical').lower()
30
- if orientation not in ('horizontal', 'vertical'):
36
+ orientation = data.get("orientation", "vertical").lower()
37
+ if orientation not in ("horizontal", "vertical"):
31
38
  raise ValueError(
32
39
  f"Invalid orientation '{orientation}' — must be 'horizontal' or 'vertical'"
33
40
  )
34
- horizontal = orientation == 'horizontal'
41
+ horizontal = orientation == "horizontal"
35
42
 
36
- x = data['data']['x']
43
+ x = data["data"]["x"]
37
44
 
38
45
  # Auto-generate series from data keys (everything except 'x') if not provided
39
- series_list = data.get('series') or [
40
- {'label': k} for k in data['data'] if k != 'x'
41
- ]
46
+ series_list = data.get("series") or [{"label": k} for k in data["data"] if k != "x"]
42
47
 
43
48
  for i, series in enumerate(series_list):
44
- label = series['label']
45
- values = data['data'][label]
46
- series_type = series.get('type', 'line')
47
- colour = series.get('colour', _DEFAULT_COLOURS[i % len(_DEFAULT_COLOURS)])
49
+ label = series["label"]
50
+ values = data["data"][label]
51
+ series_type = series.get("type", "line")
52
+ colour = series.get("colour", _DEFAULT_COLOURS[i % len(_DEFAULT_COLOURS)])
48
53
 
49
- if series_type == 'bar':
54
+ if series_type == "bar":
50
55
  colour_val = colour if isinstance(colour, list) else [colour] * len(values)
51
56
  if horizontal:
52
57
  ax.barh(x, values, color=colour_val, label=label)
53
58
  else:
54
59
  ax.bar(x, values, color=colour_val, label=label)
55
- elif series_type in ('line', 'line2'):
56
- marker = 'o' if series.get('marker', False) else 'None'
57
- linestyle = ':' if series_type == 'line2' else '-'
60
+ elif series_type in ("line", "line2"):
61
+ marker = "o" if series.get("marker", False) else "None"
62
+ linestyle = ":" if series_type == "line2" else "-"
58
63
  if horizontal:
59
- ax.plot(values, x, color=colour, marker=marker, linewidth=2, linestyle=linestyle, label=label)
64
+ ax.plot(
65
+ values,
66
+ x,
67
+ color=colour,
68
+ marker=marker,
69
+ linewidth=2,
70
+ linestyle=linestyle,
71
+ label=label,
72
+ )
60
73
  else:
61
- ax.plot(x, values, color=colour, marker=marker, linewidth=2, linestyle=linestyle, label=label)
62
-
63
- if 'min' in data or 'max' in data:
64
- lo = data.get('min', None)
65
- hi = data.get('max', None)
74
+ ax.plot(
75
+ x,
76
+ values,
77
+ color=colour,
78
+ marker=marker,
79
+ linewidth=2,
80
+ linestyle=linestyle,
81
+ label=label,
82
+ )
83
+
84
+ if "min" in data or "max" in data:
85
+ lo = data.get("min", None)
86
+ hi = data.get("max", None)
66
87
  if horizontal:
67
88
  ax.set_xlim(lo, hi)
68
89
  else:
69
90
  ax.set_ylim(lo, hi)
70
91
 
71
- title = data.get('title', '')
92
+ title = data.get("title", "")
72
93
  if title:
73
94
  ax.set_title(title)
74
95
 
75
- if data.get('show_legend', True):
76
- ax.legend(loc='upper left')
96
+ if data.get("show_legend", True):
97
+ ax.legend(loc="upper left")
77
98
 
78
- ax.spines['top'].set_visible(False)
79
- ax.spines['right'].set_visible(False)
99
+ ax.spines["top"].set_visible(False)
100
+ ax.spines["right"].set_visible(False)
80
101
 
81
102
  if output:
82
- plt.savefig(output, format="svg", bbox_inches='tight')
103
+ plt.savefig(output, format="svg", bbox_inches="tight")
83
104
  plt.close(fig)
84
105
  else:
85
106
  buf = io.StringIO()
86
- plt.savefig(buf, format="svg", bbox_inches='tight')
107
+ plt.savefig(buf, format="svg", bbox_inches="tight")
87
108
  plt.close(fig)
88
109
  return buf.getvalue()
89
110
 
@@ -93,7 +114,7 @@ _GRAPH_BLOCK = re.compile(
93
114
  re.DOTALL,
94
115
  )
95
116
 
96
- _SVG_HEADER = re.compile(r'^.*?(?=<svg)', re.DOTALL)
117
+ _SVG_HEADER = re.compile(r"^.*?(?=<svg)", re.DOTALL)
97
118
 
98
119
 
99
120
  def _parse_source(source: str) -> dict:
@@ -113,7 +134,7 @@ def _render_graph(source: str) -> str:
113
134
  """Parse source, render via graph_it, and return an inline SVG element."""
114
135
  data = _parse_source(source)
115
136
  svg = graph_it(data)
116
- svg = _SVG_HEADER.sub('', svg) # strip XML declaration and DOCTYPE
137
+ svg = _SVG_HEADER.sub("", svg) # strip XML declaration and DOCTYPE
117
138
  return f'<div class="graph">{svg}</div>'
118
139
 
119
140
 
@@ -0,0 +1,78 @@
1
+ import logging
2
+ import subprocess
3
+ import sys
4
+ import tempfile
5
+ from pathlib import Path
6
+
7
+ log = logging.getLogger(__name__)
8
+
9
+ _INSTALL_HINT = (
10
+ "Chromium is required for PDF output and could not be installed "
11
+ "automatically. Run: playwright install chromium"
12
+ )
13
+
14
+
15
+ def html_to_pdf(html_content: str, dest: Path) -> None:
16
+ """Render an HTML string to a PDF file using a headless Chromium browser.
17
+
18
+ Navigates via a temporary file:// URL so that external resources (CDN
19
+ scripts, Mermaid.js, etc.) load correctly before the page is printed.
20
+
21
+ The Chromium binary that Playwright drives is not shipped with the Python
22
+ package. The first time a PDF is rendered on a fresh machine it is
23
+ downloaded automatically; if that download fails the user is told how to
24
+ install it by hand.
25
+ """
26
+ try:
27
+ from playwright.sync_api import Error as PlaywrightError
28
+ from playwright.sync_api import sync_playwright
29
+ except ImportError:
30
+ raise RuntimeError(
31
+ "Playwright is required for PDF output. "
32
+ "Run: pip install playwright && playwright install chromium"
33
+ )
34
+
35
+ with tempfile.NamedTemporaryFile(
36
+ suffix=".html", delete=False, mode="w", encoding="utf-8"
37
+ ) as f:
38
+ f.write(html_content)
39
+ tmp_path = Path(f.name)
40
+
41
+ try:
42
+ try:
43
+ _render(sync_playwright, tmp_path, dest)
44
+ except PlaywrightError as exc:
45
+ if "Executable doesn't exist" not in str(exc):
46
+ raise
47
+ _install_chromium()
48
+ _render(sync_playwright, tmp_path, dest)
49
+ finally:
50
+ tmp_path.unlink(missing_ok=True)
51
+
52
+
53
+ def _render(sync_playwright, source: Path, dest: Path) -> None:
54
+ """Drive headless Chromium to print ``source`` to ``dest`` as A4 PDF."""
55
+ with sync_playwright() as p:
56
+ browser = p.chromium.launch()
57
+ page = browser.new_page()
58
+ page.goto(source.as_uri(), wait_until="networkidle")
59
+ # If Paged.js is present, wait for it to finish paginating before
60
+ # capturing — networkidle fires before its JS layout pass completes.
61
+ page.wait_for_function(
62
+ "typeof window.PagedPolyfill === 'undefined'"
63
+ " || !!document.querySelector('.pagedjs_pages')"
64
+ )
65
+ page.pdf(path=str(dest), format="A4", print_background=True)
66
+ browser.close()
67
+
68
+
69
+ def _install_chromium() -> None:
70
+ """Download the Chromium binary Playwright needs, or explain how to."""
71
+ log.warning("Chromium not found — downloading it now (~150 MB, one time).")
72
+ try:
73
+ subprocess.run(
74
+ [sys.executable, "-m", "playwright", "install", "chromium"],
75
+ check=True,
76
+ )
77
+ except (subprocess.CalledProcessError, OSError) as exc:
78
+ raise RuntimeError(_INSTALL_HINT) from exc
@@ -0,0 +1,132 @@
1
+ import csv
2
+ import json
3
+ from pathlib import Path
4
+
5
+ from jinja2 import Environment, FileSystemLoader
6
+
7
+
8
+ def load_json_context(input_path: Path) -> dict:
9
+ """Load all *.json files from the input directory into a template context dict."""
10
+ directory = input_path if input_path.is_dir() else input_path.parent
11
+ context: dict = {}
12
+ for json_file in sorted(directory.glob("*.json")):
13
+ try:
14
+ data = json.loads(json_file.read_text(encoding="utf-8"))
15
+ except json.JSONDecodeError as exc:
16
+ raise ValueError(f"Invalid JSON in {json_file.name}: {exc}") from exc
17
+ context[json_file.stem] = data
18
+ return context
19
+
20
+
21
+ def _parse_csv_line(line: str) -> list[tuple[str, bool]]:
22
+ """Parse a single CSV line into (value, was_quoted) pairs.
23
+
24
+ Handles quoted fields (including commas and escaped quotes inside them).
25
+ """
26
+ fields: list[tuple[str, bool]] = []
27
+ i = 0
28
+ n = len(line)
29
+
30
+ while True:
31
+ if i >= n:
32
+ break
33
+
34
+ if line[i] == '"':
35
+ i += 1
36
+ parts: list[str] = []
37
+ while i < n:
38
+ if line[i] == '"':
39
+ if i + 1 < n and line[i + 1] == '"':
40
+ parts.append('"')
41
+ i += 2
42
+ else:
43
+ i += 1
44
+ break
45
+ else:
46
+ parts.append(line[i])
47
+ i += 1
48
+ value = "".join(parts)
49
+ while i < n and line[i] != ",":
50
+ i += 1
51
+ fields.append((value, True))
52
+ else:
53
+ start = i
54
+ while i < n and line[i] != ",":
55
+ i += 1
56
+ fields.append((line[start:i], False))
57
+
58
+ if i < n and line[i] == ",":
59
+ i += 1
60
+ if i >= n:
61
+ fields.append(("", False))
62
+
63
+ return fields
64
+
65
+
66
+ def _apply_semicolon_split(value: str, was_quoted: bool) -> str | list[str]:
67
+ """Return value split on ';' unless the field was originally quoted."""
68
+ if was_quoted or ";" not in value:
69
+ return value
70
+ return [part.strip() for part in value.split(";")]
71
+
72
+
73
+ def load_csv_context(input_path: Path) -> dict:
74
+ """Load all *.csv files from the input directory into a template context dict.
75
+
76
+ Each file's stem becomes the variable name (people.csv → ``people``).
77
+ The value is a list of row dicts. Cell values containing ';' become lists;
78
+ quoted fields are never split.
79
+ """
80
+ directory = input_path if input_path.is_dir() else input_path.parent
81
+ context: dict = {}
82
+
83
+ for csv_file in sorted(directory.glob("*.csv")):
84
+ lines = csv_file.read_text(encoding="utf-8").splitlines()
85
+ rows: list[dict] = []
86
+
87
+ if not lines:
88
+ context[csv_file.stem] = rows
89
+ continue
90
+
91
+ header = next(csv.reader([lines[0]]))
92
+
93
+ for raw_line in lines[1:]:
94
+ if not raw_line.strip():
95
+ continue
96
+ fields = _parse_csv_line(raw_line)
97
+ row: dict = {}
98
+ for idx, key in enumerate(header):
99
+ if idx < len(fields):
100
+ value, was_quoted = fields[idx]
101
+ else:
102
+ value, was_quoted = "", False
103
+ row[key] = _apply_semicolon_split(value, was_quoted)
104
+ rows.append(row)
105
+
106
+ context[csv_file.stem] = rows
107
+
108
+ return context
109
+
110
+
111
+ def load_context(input_path: Path) -> dict:
112
+ """Load JSON and CSV context files, raising ValueError on stem conflicts."""
113
+ json_ctx = load_json_context(input_path)
114
+ csv_ctx = load_csv_context(input_path)
115
+ conflicts = set(json_ctx) & set(csv_ctx)
116
+ if conflicts:
117
+ names = ", ".join(sorted(conflicts))
118
+ raise ValueError(
119
+ f"Context name conflict — both a .json and .csv exist for: {names}"
120
+ )
121
+ return {**json_ctx, **csv_ctx}
122
+
123
+
124
+ def render_j2(path: Path, context: dict) -> str:
125
+ """Render a .md.j2 Jinja2 template and return the resulting Markdown text."""
126
+ env = Environment(
127
+ loader=FileSystemLoader(str(path.parent)),
128
+ autoescape=False,
129
+ keep_trailing_newline=True,
130
+ )
131
+ template = env.get_template(path.name)
132
+ return template.render(**context)
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: mddoco
3
- Version: 2.0.0
3
+ Version: 2.0.3
4
4
  Summary: Markdown to HTML/PDF document converter
5
5
  Requires-Python: >=3.10
6
6
  Description-Content-Type: text/markdown
@@ -10,6 +10,10 @@ Requires-Dist: markdown>=3.5
10
10
  Requires-Dist: jinja2>=3.1
11
11
  Requires-Dist: playwright>=1.40
12
12
  Requires-Dist: matplotlib>=3.7
13
+ Provides-Extra: dev
14
+ Requires-Dist: pytest>=8.0; extra == "dev"
15
+ Requires-Dist: black>=24.0; extra == "dev"
16
+ Requires-Dist: ruff<0.17,>=0.16; extra == "dev"
13
17
  Dynamic: license-file
14
18
 
15
19
  # mddoco
@@ -20,17 +24,23 @@ A CLI tool that converts markdown files to a single HTML (or future PDF) documen
20
24
 
21
25
  ```bash
22
26
  pip install mddoco
23
- playwright install chromium
24
27
  ```
25
28
 
26
29
  Or for development:
27
30
 
28
31
  ```bash
29
32
  pip install -e .
30
- playwright install chromium
31
33
  ```
32
34
 
33
- > Playwright (Chromium) is required for PDF output only. HTML output works without it.
35
+ > **Playwright (Chromium) is required for PDF output only.** HTML output works without it.
36
+ > The `playwright` Python package is installed automatically, but the Chromium
37
+ > browser it drives is not. The first time you render a PDF, mddoco downloads
38
+ > Chromium for you (~150 MB, one time). To do it ahead of time, or if the
39
+ > automatic download fails, run it yourself:
40
+ >
41
+ > ```bash
42
+ > playwright install chromium
43
+ > ```
34
44
 
35
45
  ## Usage
36
46
 
@@ -79,25 +89,43 @@ All matched markdown files are combined into a single HTML document in the outpu
79
89
 
80
90
  Files ending in `.md.j2` are Jinja2 templates that produce Markdown. They are sorted alongside regular `.md` files (numeric prefixes such as `01_`, `02_` determine order) and converted through the same pipeline.
81
91
 
82
- ### JSON context variables
92
+ ### Context variables (JSON and CSV)
83
93
 
84
- Place `*.json` files in the same input directory. Each file's stem becomes a top-level variable in the Jinja2 context — `report.json` is available as `report`, `config.json` as `config`, and so on.
94
+ Place `*.json` or `*.csv` files in the same input directory. Each file's stem becomes a top-level variable in the Jinja2 context — `report.json` is available as `report`, `people.csv` as `people`, and so on. If both `data.json` and `data.csv` exist, the tool will exit with an error.
85
95
 
86
96
  ```
87
97
  docs/
88
98
  01_intro.md
89
99
  02_summary.md.j2 ← Jinja2 template
90
- report.json ← available as {{ report }} inside *.md.j2 files
100
+ report.json ← available as {{ report }}
101
+ people.csv ← available as {{ people }}
91
102
  ```
92
103
 
93
- Example template (`02_summary.md.j2`):
104
+ **JSON** files are loaded as-is; the variable holds whatever structure the JSON contains.
105
+
106
+ **CSV** files are loaded as a list of row dicts, one dict per row:
94
107
 
95
108
  ```markdown
96
- # Summary for {{ report.project }}
109
+ {% for person in people %}
110
+ - {{ person.name }} ({{ person.role }})
111
+ {% endfor %}
112
+ ```
97
113
 
98
- Generated by: {{ report.author }}
114
+ Cell values containing `;` are automatically split into a list:
115
+
116
+ ```csv
117
+ name,skills
118
+ Alice,python;flask;sql
119
+ Bob,java
99
120
  ```
100
121
 
122
+ ```markdown
123
+ {{ people[0].skills }} {# → ["python", "flask", "sql"] #}
124
+ {{ people[1].skills }} {# → "java" (plain string — no ;) #}
125
+ ```
126
+
127
+ Quoting a field in the CSV prevents `;` splitting — `"python;flask"` remains a single string.
128
+
101
129
  ### Excluding files
102
130
 
103
131
  Files whose name starts with `_` are excluded from scanning. Use this for Jinja2 macro files that should be imported but not rendered as documents:
@@ -109,6 +137,24 @@ docs/
109
137
  02_report.md.j2
110
138
  ```
111
139
 
140
+ ### Using macros
141
+
142
+ Use `{% import %}` or `{% from ... import %}` to make macros from another file callable — **not** `{% include %}`. `{% include %}` injects rendered output only; macros defined in an included file are not visible to the calling template and will raise an `undefined` error.
143
+
144
+ ```jinja
145
+ {# correct — macro is callable after this #}
146
+ {% import "_macros.j2" as macros %}
147
+ {{ macros.val(item) }}
148
+
149
+ {# also correct #}
150
+ {% from "_macros.j2" import val %}
151
+ {{ val(item) }}
152
+
153
+ {# wrong — val will be undefined #}
154
+ {% include "_macros.j2" %}
155
+ {{ val(item) }}
156
+ ```
157
+
112
158
  ## Themes
113
159
 
114
160
  Themes are self-contained Jinja2 HTML files with embedded CSS. Pass a theme name with `--theme NAME`.
@@ -27,4 +27,5 @@ src/mddoco/themes/default.html
27
27
  src/mddoco/themes/paged-professional.html
28
28
  src/mddoco/themes/professional-wide.html
29
29
  src/mddoco/themes/professional.html
30
- src/mddoco/themes/vanilla.html
30
+ src/mddoco/themes/vanilla.html
31
+ tests/test_preprocessor.py
@@ -3,3 +3,8 @@ markdown>=3.5
3
3
  jinja2>=3.1
4
4
  playwright>=1.40
5
5
  matplotlib>=3.7
6
+
7
+ [dev]
8
+ pytest>=8.0
9
+ black>=24.0
10
+ ruff<0.17,>=0.16
@@ -0,0 +1,145 @@
1
+ from pathlib import Path
2
+
3
+ import pytest
4
+
5
+ import mddoco
6
+ from mddoco.preprocessor import (
7
+ _apply_semicolon_split,
8
+ _parse_csv_line,
9
+ load_context,
10
+ load_csv_context,
11
+ )
12
+
13
+ # --- version ---
14
+
15
+
16
+ def test_version():
17
+ assert mddoco.__version__ == "2.0.3"
18
+
19
+
20
+ # --- _parse_csv_line ---
21
+
22
+
23
+ def test_parse_csv_line_simple():
24
+ assert _parse_csv_line("a,b,c") == [("a", False), ("b", False), ("c", False)]
25
+
26
+
27
+ def test_parse_csv_line_quoted_field():
28
+ assert _parse_csv_line('a,"b;c",d') == [("a", False), ("b;c", True), ("d", False)]
29
+
30
+
31
+ def test_parse_csv_line_quoted_field_with_comma():
32
+ assert _parse_csv_line('a,"b,c",d') == [("a", False), ("b,c", True), ("d", False)]
33
+
34
+
35
+ def test_parse_csv_line_escaped_quote_inside_quoted():
36
+ assert _parse_csv_line('"say ""hello""",b') == [('say "hello"', True), ("b", False)]
37
+
38
+
39
+ def test_parse_csv_line_trailing_comma():
40
+ assert _parse_csv_line("a,b,") == [("a", False), ("b", False), ("", False)]
41
+
42
+
43
+ def test_parse_csv_line_single_field():
44
+ assert _parse_csv_line("hello") == [("hello", False)]
45
+
46
+
47
+ # --- _apply_semicolon_split ---
48
+
49
+
50
+ def test_split_unquoted_with_semicolon():
51
+ assert _apply_semicolon_split("python;flask", False) == ["python", "flask"]
52
+
53
+
54
+ def test_split_trims_whitespace():
55
+ assert _apply_semicolon_split("python ; flask ; sql", False) == [
56
+ "python",
57
+ "flask",
58
+ "sql",
59
+ ]
60
+
61
+
62
+ def test_no_split_when_quoted():
63
+ assert _apply_semicolon_split("python;flask", True) == "python;flask"
64
+
65
+
66
+ def test_no_split_no_semicolon():
67
+ assert _apply_semicolon_split("python", False) == "python"
68
+
69
+
70
+ def test_single_element_after_split():
71
+ assert _apply_semicolon_split("python;", False) == ["python", ""]
72
+
73
+
74
+ # --- load_csv_context ---
75
+
76
+
77
+ def test_load_csv_basic(tmp_path: Path):
78
+ (tmp_path / "people.csv").write_text(
79
+ "name,role\nAlice,Engineer\nBob,Manager\n", encoding="utf-8"
80
+ )
81
+ ctx = load_csv_context(tmp_path)
82
+ assert ctx == {
83
+ "people": [
84
+ {"name": "Alice", "role": "Engineer"},
85
+ {"name": "Bob", "role": "Manager"},
86
+ ]
87
+ }
88
+
89
+
90
+ def test_load_csv_semicolon_becomes_list(tmp_path: Path):
91
+ (tmp_path / "data.csv").write_text(
92
+ "name,skills\nAlice,python;flask\n", encoding="utf-8"
93
+ )
94
+ ctx = load_csv_context(tmp_path)
95
+ assert ctx["data"][0]["skills"] == ["python", "flask"]
96
+
97
+
98
+ def test_load_csv_quoted_semicolon_not_split(tmp_path: Path):
99
+ (tmp_path / "data.csv").write_text('id,label\n1,"a;b;c"\n', encoding="utf-8")
100
+ ctx = load_csv_context(tmp_path)
101
+ assert ctx["data"][0]["label"] == "a;b;c"
102
+
103
+
104
+ def test_load_csv_multiple_files(tmp_path: Path):
105
+ (tmp_path / "a.csv").write_text("x\n1\n", encoding="utf-8")
106
+ (tmp_path / "b.csv").write_text("y\n2\n", encoding="utf-8")
107
+ ctx = load_csv_context(tmp_path)
108
+ assert "a" in ctx and "b" in ctx
109
+
110
+
111
+ def test_load_csv_empty_file(tmp_path: Path):
112
+ (tmp_path / "empty.csv").write_text("", encoding="utf-8")
113
+ ctx = load_csv_context(tmp_path)
114
+ assert ctx == {"empty": []}
115
+
116
+
117
+ def test_load_csv_blank_lines_skipped(tmp_path: Path):
118
+ (tmp_path / "data.csv").write_text("name\nAlice\n\nBob\n", encoding="utf-8")
119
+ ctx = load_csv_context(tmp_path)
120
+ assert len(ctx["data"]) == 2
121
+
122
+
123
+ def test_load_csv_from_file_path(tmp_path: Path):
124
+ (tmp_path / "data.csv").write_text("col\nval\n", encoding="utf-8")
125
+ f = tmp_path / "doc.md"
126
+ f.write_text("")
127
+ ctx = load_csv_context(f)
128
+ assert "data" in ctx
129
+
130
+
131
+ # --- load_context ---
132
+
133
+
134
+ def test_load_context_merges_json_and_csv(tmp_path: Path):
135
+ (tmp_path / "config.json").write_text('{"env": "prod"}', encoding="utf-8")
136
+ (tmp_path / "people.csv").write_text("name\nAlice\n", encoding="utf-8")
137
+ ctx = load_context(tmp_path)
138
+ assert "config" in ctx and "people" in ctx
139
+
140
+
141
+ def test_load_context_conflict_raises(tmp_path: Path):
142
+ (tmp_path / "data.json").write_text('{"key": "val"}', encoding="utf-8")
143
+ (tmp_path / "data.csv").write_text("col\nval\n", encoding="utf-8")
144
+ with pytest.raises(ValueError, match="data"):
145
+ load_context(tmp_path)
@@ -1 +0,0 @@
1
- __version__ = "2.0.0"
@@ -1,39 +0,0 @@
1
- import tempfile
2
- from pathlib import Path
3
-
4
-
5
- def html_to_pdf(html_content: str, dest: Path) -> None:
6
- """Render an HTML string to a PDF file using a headless Chromium browser.
7
-
8
- Navigates via a temporary file:// URL so that external resources (CDN
9
- scripts, Mermaid.js, etc.) load correctly before the page is printed.
10
- """
11
- try:
12
- from playwright.sync_api import sync_playwright
13
- except ImportError:
14
- raise RuntimeError(
15
- "Playwright is required for PDF output. "
16
- "Run: pip install playwright && playwright install chromium"
17
- )
18
-
19
- with tempfile.NamedTemporaryFile(
20
- suffix=".html", delete=False, mode="w", encoding="utf-8"
21
- ) as f:
22
- f.write(html_content)
23
- tmp_path = Path(f.name)
24
-
25
- try:
26
- with sync_playwright() as p:
27
- browser = p.chromium.launch()
28
- page = browser.new_page()
29
- page.goto(tmp_path.as_uri(), wait_until="networkidle")
30
- # If Paged.js is present, wait for it to finish paginating before
31
- # capturing — networkidle fires before its JS layout pass completes.
32
- page.wait_for_function(
33
- "typeof window.PagedPolyfill === 'undefined'"
34
- " || !!document.querySelector('.pagedjs_pages')"
35
- )
36
- page.pdf(path=str(dest), format="A4", print_background=True)
37
- browser.close()
38
- finally:
39
- tmp_path.unlink(missing_ok=True)
@@ -1,32 +0,0 @@
1
- import json
2
- from pathlib import Path
3
-
4
- from jinja2 import Environment, FileSystemLoader
5
-
6
-
7
- def load_json_context(input_path: Path) -> dict:
8
- """Load all *.json files from the input directory into a template context dict.
9
-
10
- The stem of each JSON file becomes the variable name in the context.
11
- E.g. report.json → available as ``report`` inside .md.j2 templates.
12
- """
13
- directory = input_path if input_path.is_dir() else input_path.parent
14
- context: dict = {}
15
- for json_file in sorted(directory.glob("*.json")):
16
- try:
17
- data = json.loads(json_file.read_text(encoding="utf-8"))
18
- except json.JSONDecodeError as exc:
19
- raise ValueError(f"Invalid JSON in {json_file.name}: {exc}") from exc
20
- context[json_file.stem] = data
21
- return context
22
-
23
-
24
- def render_j2(path: Path, context: dict) -> str:
25
- """Render a .md.j2 Jinja2 template and return the resulting Markdown text."""
26
- env = Environment(
27
- loader=FileSystemLoader(str(path.parent)),
28
- autoescape=False,
29
- keep_trailing_newline=True,
30
- )
31
- template = env.get_template(path.name)
32
- return template.render(**context)
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes