lhtml-markup 2.2.0__tar.gz → 2.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {lhtml_markup-2.2.0 → lhtml_markup-2.3.0}/PKG-INFO +48 -7
- {lhtml_markup-2.2.0 → lhtml_markup-2.3.0}/README.md +47 -6
- {lhtml_markup-2.2.0 → lhtml_markup-2.3.0}/pyproject.toml +1 -1
- {lhtml_markup-2.2.0 → lhtml_markup-2.3.0}/src/lhtml/__init__.py +6 -14
- {lhtml_markup-2.2.0 → lhtml_markup-2.3.0}/src/lhtml/cli.py +53 -14
- {lhtml_markup-2.2.0 → lhtml_markup-2.3.0}/src/lhtml/code.py +20 -8
- {lhtml_markup-2.2.0 → lhtml_markup-2.3.0}/src/lhtml/errors.py +26 -6
- {lhtml_markup-2.2.0 → lhtml_markup-2.3.0}/src/lhtml/export_html.py +17 -9
- lhtml_markup-2.3.0/src/lhtml/insert_in_text.py +30 -0
- lhtml_markup-2.3.0/src/lhtml/patterns.py +156 -0
- {lhtml_markup-2.2.0 → lhtml_markup-2.3.0}/src/lhtml/pipeline.py +50 -43
- lhtml_markup-2.3.0/src/lhtml/process.py +590 -0
- {lhtml_markup-2.2.0 → lhtml_markup-2.3.0}/src/lhtml/tag_element.lark +7 -3
- {lhtml_markup-2.2.0 → lhtml_markup-2.3.0}/src/lhtml/tag_parser.py +34 -36
- {lhtml_markup-2.2.0 → lhtml_markup-2.3.0}/src/lhtml/wrap_html.py +5 -3
- {lhtml_markup-2.2.0 → lhtml_markup-2.3.0}/src/lhtml_markup.egg-info/PKG-INFO +48 -7
- {lhtml_markup-2.2.0 → lhtml_markup-2.3.0}/src/lhtml_markup.egg-info/SOURCES.txt +0 -1
- lhtml_markup-2.3.0/test/test_lhtml.py +1120 -0
- lhtml_markup-2.2.0/src/lhtml/ast_nodes.py +0 -124
- lhtml_markup-2.2.0/src/lhtml/insert_in_text.py +0 -7
- lhtml_markup-2.2.0/src/lhtml/patterns.py +0 -92
- lhtml_markup-2.2.0/src/lhtml/process.py +0 -253
- lhtml_markup-2.2.0/test/test_lhtml.py +0 -385
- {lhtml_markup-2.2.0 → lhtml_markup-2.3.0}/LICENSE.md +0 -0
- {lhtml_markup-2.2.0 → lhtml_markup-2.3.0}/setup.cfg +0 -0
- {lhtml_markup-2.2.0 → lhtml_markup-2.3.0}/src/lhtml/__main__.py +0 -0
- {lhtml_markup-2.2.0 → lhtml_markup-2.3.0}/src/lhtml/element_extract.py +0 -0
- {lhtml_markup-2.2.0 → lhtml_markup-2.3.0}/src/lhtml/listing.py +0 -0
- {lhtml_markup-2.2.0 → lhtml_markup-2.3.0}/src/lhtml_markup.egg-info/dependency_links.txt +0 -0
- {lhtml_markup-2.2.0 → lhtml_markup-2.3.0}/src/lhtml_markup.egg-info/entry_points.txt +0 -0
- {lhtml_markup-2.2.0 → lhtml_markup-2.3.0}/src/lhtml_markup.egg-info/requires.txt +0 -0
- {lhtml_markup-2.2.0 → lhtml_markup-2.3.0}/src/lhtml_markup.egg-info/top_level.txt +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: lhtml-markup
|
|
3
|
-
Version: 2.
|
|
3
|
+
Version: 2.3.0
|
|
4
4
|
Summary: Lightweight HTML markup language — simplifies HTML authoring with shorthand syntax
|
|
5
5
|
License: MIT
|
|
6
6
|
Project-URL: Homepage, https://github.com/drohmer/lhtml
|
|
@@ -59,9 +59,14 @@ Dependencies (`lark`, `pygments`, `pyyaml`) are installed automatically.
|
|
|
59
59
|
lhtml input.l.html # Convert to stdout
|
|
60
60
|
lhtml input.l.html -o output.html # Convert to file
|
|
61
61
|
lhtml input.l.html -w # Wrap in full HTML document
|
|
62
|
+
lhtml input.l.html -b # Render source line breaks as <br>
|
|
63
|
+
lhtml a.l.html b.l.html # Several files: a.html, b.html next to sources
|
|
64
|
+
lhtml a.l.html b.l.html -o build/ # Several files into a directory
|
|
62
65
|
python -m lhtml input.l.html # Alternative invocation
|
|
63
66
|
```
|
|
64
67
|
|
|
68
|
+
Files are read and written as UTF-8 (a BOM is accepted). Errors and warnings are reported on stderr with the file name, the remaining files are still processed, and the exit code is non-zero if any file failed. An input file is never overwritten (e.g. `lhtml page.html` without `-o`). Includes are looked up in the input file's directory first, then in the current directory.
|
|
69
|
+
|
|
65
70
|
### Python API
|
|
66
71
|
|
|
67
72
|
```python
|
|
@@ -133,6 +138,8 @@ This is <em>italic</em> text.
|
|
|
133
138
|
This is <code class="code-inline">inline code</code> text.
|
|
134
139
|
```
|
|
135
140
|
|
|
141
|
+
Inside inline code, `<` and `>` are escaped (so `` `std::vector<float>` `` displays correctly), and `**`, `__`, `$`, `include::`, `code::` and `::#` are not interpreted. Existing entities such as `<` are kept. Named tags such as `link::` remain active, but a bare `::` is C++ (`` `::glfwInit()` ``), not a closing tag.
|
|
142
|
+
|
|
136
143
|
|
|
137
144
|
### Tag Elements (the `::` system)
|
|
138
145
|
|
|
@@ -144,6 +151,12 @@ tagName::(.classes #id)[cssStyle]{htmlAttributes} content ::
|
|
|
144
151
|
|
|
145
152
|
All bracket groups are optional. If `tagName` is omitted, defaults to `div`.
|
|
146
153
|
|
|
154
|
+
Tag names start with a letter and may contain letters, digits, `_` and `-` (useful for custom tags). A tag is only recognized when it is not glued to a preceding word: `std::chrono::seconds`, `obj.link::x` or the Python slice `a[::2]` are left untouched.
|
|
155
|
+
|
|
156
|
+
A closing `::` may be directly followed by punctuation or HTML (`**span::[c] x ::**`, `important ::,`). In a run of colons, `::` markers are the pairs ending the run: `::::[...]` is a closing `::` followed by `::[...]`, and `x:::nl` is `x:` followed by `::nl`. A `::` between two HTML tags (as in pre-highlighted code `<span>::</span>`) is left untouched.
|
|
157
|
+
|
|
158
|
+
A tag that is opened but never closed, or a `::` without matching opening tag, produces an `LHTMLWarning` quoting the beginning of the tag.
|
|
159
|
+
|
|
147
160
|
#### Div / Span with Styles
|
|
148
161
|
|
|
149
162
|
```
|
|
@@ -209,6 +222,8 @@ Output:
|
|
|
209
222
|
|
|
210
223
|
### Links
|
|
211
224
|
|
|
225
|
+
The URL of `link::`, `img::`, `video::` and `videoplay::` is never modified (`__`, `**`, `$` are kept). Parentheses that are part of the URL are kept (`Mercury_(planet)`, `fig(1).png`), while a group starting with `.` or `#` is a class/id group.
|
|
226
|
+
|
|
212
227
|
```
|
|
213
228
|
link::https://example.com[Click here]
|
|
214
229
|
link::page.html(.nav)[Back to home]
|
|
@@ -252,7 +267,7 @@ def hello():
|
|
|
252
267
|
code::[-]
|
|
253
268
|
````
|
|
254
269
|
|
|
255
|
-
Syntax highlighting is powered by Pygments. Any language supported by Pygments can be used.
|
|
270
|
+
`include::file` directives inside a code block insert the file as raw code. Syntax highlighting is powered by Pygments. Any language supported by Pygments can be used. An empty language (`code::[]`) renders plain text; an unknown language renders plain text with a warning.
|
|
256
271
|
|
|
257
272
|
|
|
258
273
|
### Spacer
|
|
@@ -280,12 +295,28 @@ verbatim::[-]
|
|
|
280
295
|
```
|
|
281
296
|
|
|
282
297
|
|
|
298
|
+
### Line breaks
|
|
299
|
+
|
|
300
|
+
By default, line breaks of the source have no visual effect (as in HTML). With the option `-b` / `--line-breaks` (or `line-breaks: true` in the YAML front matter, or `{'line-breaks': True}` in the `meta` passed to `lhtml.run()`), each line break of the text is rendered with `<br>`, blank lines included:
|
|
301
|
+
|
|
302
|
+
```
|
|
303
|
+
Line 1 Line 1<br>
|
|
304
|
+
Line 2 => Line 2<br>
|
|
305
|
+
<br>
|
|
306
|
+
Other paragraph Other paragraph
|
|
307
|
+
```
|
|
308
|
+
|
|
309
|
+
Images and videos are inline: images written one per line are stacked (write them on the same line to keep them side by side). Lines that start or end with a block element (headings, lists, `div::`, block-level HTML tags such as `<div>`, `<p>`, `<li>`, code and verbatim blocks, `<script>`, display math, ...) are structure: they never get a `<br>` and blank lines next to them are ignored. The content of code blocks, verbatim blocks, scripts, comments and math is never modified.
|
|
310
|
+
|
|
311
|
+
|
|
283
312
|
### Comments
|
|
284
313
|
|
|
285
314
|
```
|
|
286
315
|
Some text ::# This comment will be removed
|
|
287
316
|
```
|
|
288
317
|
|
|
318
|
+
A comment starts a line or follows whitespace, so `link::#intro[...]` (link to an anchor) is not a comment.
|
|
319
|
+
|
|
289
320
|
|
|
290
321
|
### File Inclusion
|
|
291
322
|
|
|
@@ -294,11 +325,13 @@ include::header.html
|
|
|
294
325
|
include::components/nav.html
|
|
295
326
|
```
|
|
296
327
|
|
|
297
|
-
Included files are recursively processed (up to 20 levels).
|
|
328
|
+
Included files are recursively processed (up to 20 levels); their own YAML front matter is ignored. An included file looks for its own includes first in its own directory, then in `directory_include`. A circular include raises `LHTMLIncludeLoopError`.
|
|
298
329
|
|
|
299
330
|
|
|
300
331
|
### YAML Front Matter
|
|
301
332
|
|
|
333
|
+
The front matter must be at the very beginning of the file (`---` separators elsewhere are kept as text). Input text is normalized first: a leading BOM is removed and CRLF line endings become LF.
|
|
334
|
+
|
|
302
335
|
```
|
|
303
336
|
---
|
|
304
337
|
title: "My Page"
|
|
@@ -318,6 +351,7 @@ Supported metadata keys:
|
|
|
318
351
|
| `css` | string or list | CSS files to include |
|
|
319
352
|
| `js` | string or list | JavaScript files to include |
|
|
320
353
|
| `wrap-auto` | boolean | Wrap output in full HTML document |
|
|
354
|
+
| `line-breaks` | boolean | Render source line breaks as `<br>` |
|
|
321
355
|
| `directory_include` | list | Directories to search for includes |
|
|
322
356
|
|
|
323
357
|
|
|
@@ -331,9 +365,10 @@ Register handlers for new `::` tag types:
|
|
|
331
365
|
from lhtml.pipeline import tag_registry
|
|
332
366
|
|
|
333
367
|
def handle_alert(element, tag_to_close, current_directory):
|
|
368
|
+
"""Open <div class="alert">; the following :: closes it."""
|
|
334
369
|
style = element.get('[]', '')
|
|
335
|
-
|
|
336
|
-
return f'<div class="alert" style="{style}">
|
|
370
|
+
tag_to_close.append('div')
|
|
371
|
+
return f'<div class="alert" style="{style}">', True
|
|
337
372
|
|
|
338
373
|
tag_registry.register('alert', handle_alert)
|
|
339
374
|
```
|
|
@@ -343,6 +378,12 @@ Then use in LHTML:
|
|
|
343
378
|
alert::[background:yellow; padding:10px;] Warning message ::
|
|
344
379
|
```
|
|
345
380
|
|
|
381
|
+
A handler receives the parsed element (`'[]'`, `'()'`, `'{}'`, `'text'`: the
|
|
382
|
+
word glued after the brackets) and returns `(html, is_real_tag)`. To wrap the
|
|
383
|
+
content that follows, push the HTML tag name on `tag_to_close`: the next `::`
|
|
384
|
+
(or `::name[-]`) closes it. Returning `is_real_tag=False` leaves the source
|
|
385
|
+
text unchanged.
|
|
386
|
+
|
|
346
387
|
### Custom Code Lexers
|
|
347
388
|
|
|
348
389
|
Register custom Pygments lexers for syntax highlighting:
|
|
@@ -376,6 +417,7 @@ All keys for the `meta` dict passed to `lhtml.run()`:
|
|
|
376
417
|
```python
|
|
377
418
|
{
|
|
378
419
|
'wrap-auto': False, # Wrap in HTML document
|
|
420
|
+
'line-breaks': False, # Render source line breaks as <br>
|
|
379
421
|
'title': 'Webpage', # Document title
|
|
380
422
|
'css': [], # CSS files (string or list)
|
|
381
423
|
'js': [], # JS files (string or list)
|
|
@@ -387,7 +429,7 @@ All keys for the `meta` dict passed to `lhtml.run()`:
|
|
|
387
429
|
|
|
388
430
|
## Design Principles
|
|
389
431
|
|
|
390
|
-
- **HTML-first**: Raw HTML is never modified. Only LHTML syntax triggers conversions.
|
|
432
|
+
- **HTML-first**: Raw HTML is never modified. Only LHTML syntax triggers conversions. In particular, the following are never transformed (not even by `include::` or `::#`): HTML tags and their attributes (URLs containing `__`, quoted values containing `>`, ...), `<script>` and `<style>` blocks (CSS `::before`, ...), HTML comments, and math (`$...$`, `$$...$$`, `\(...\)`, `\[...\]`, so that `$x**2$` reaches MathJax/KaTeX intact). Text between HTML tags is still processed.
|
|
391
433
|
- **Island grammar**: LHTML syntax "islands" float in a sea of opaque content (HTML, Jinja2 templates, LaTeX, etc.) that passes through untouched.
|
|
392
434
|
- **Minimal**: A few symbols (`::`, `=`, `*`, `**`, `__`, `` ` ``) cover most needs. No complex configuration required.
|
|
393
435
|
- **Composable**: LHTML works seamlessly with Jinja2 templates, making it suitable for static site generators.
|
|
@@ -408,7 +450,6 @@ src/lhtml/
|
|
|
408
450
|
listing.py # List processing
|
|
409
451
|
code.py # Code syntax highlighting (Pygments)
|
|
410
452
|
wrap_html.py # HTML document wrapping
|
|
411
|
-
ast_nodes.py # AST node dataclasses
|
|
412
453
|
errors.py # Structured error types
|
|
413
454
|
```
|
|
414
455
|
|
|
@@ -40,9 +40,14 @@ Dependencies (`lark`, `pygments`, `pyyaml`) are installed automatically.
|
|
|
40
40
|
lhtml input.l.html # Convert to stdout
|
|
41
41
|
lhtml input.l.html -o output.html # Convert to file
|
|
42
42
|
lhtml input.l.html -w # Wrap in full HTML document
|
|
43
|
+
lhtml input.l.html -b # Render source line breaks as <br>
|
|
44
|
+
lhtml a.l.html b.l.html # Several files: a.html, b.html next to sources
|
|
45
|
+
lhtml a.l.html b.l.html -o build/ # Several files into a directory
|
|
43
46
|
python -m lhtml input.l.html # Alternative invocation
|
|
44
47
|
```
|
|
45
48
|
|
|
49
|
+
Files are read and written as UTF-8 (a BOM is accepted). Errors and warnings are reported on stderr with the file name, the remaining files are still processed, and the exit code is non-zero if any file failed. An input file is never overwritten (e.g. `lhtml page.html` without `-o`). Includes are looked up in the input file's directory first, then in the current directory.
|
|
50
|
+
|
|
46
51
|
### Python API
|
|
47
52
|
|
|
48
53
|
```python
|
|
@@ -114,6 +119,8 @@ This is <em>italic</em> text.
|
|
|
114
119
|
This is <code class="code-inline">inline code</code> text.
|
|
115
120
|
```
|
|
116
121
|
|
|
122
|
+
Inside inline code, `<` and `>` are escaped (so `` `std::vector<float>` `` displays correctly), and `**`, `__`, `$`, `include::`, `code::` and `::#` are not interpreted. Existing entities such as `<` are kept. Named tags such as `link::` remain active, but a bare `::` is C++ (`` `::glfwInit()` ``), not a closing tag.
|
|
123
|
+
|
|
117
124
|
|
|
118
125
|
### Tag Elements (the `::` system)
|
|
119
126
|
|
|
@@ -125,6 +132,12 @@ tagName::(.classes #id)[cssStyle]{htmlAttributes} content ::
|
|
|
125
132
|
|
|
126
133
|
All bracket groups are optional. If `tagName` is omitted, defaults to `div`.
|
|
127
134
|
|
|
135
|
+
Tag names start with a letter and may contain letters, digits, `_` and `-` (useful for custom tags). A tag is only recognized when it is not glued to a preceding word: `std::chrono::seconds`, `obj.link::x` or the Python slice `a[::2]` are left untouched.
|
|
136
|
+
|
|
137
|
+
A closing `::` may be directly followed by punctuation or HTML (`**span::[c] x ::**`, `important ::,`). In a run of colons, `::` markers are the pairs ending the run: `::::[...]` is a closing `::` followed by `::[...]`, and `x:::nl` is `x:` followed by `::nl`. A `::` between two HTML tags (as in pre-highlighted code `<span>::</span>`) is left untouched.
|
|
138
|
+
|
|
139
|
+
A tag that is opened but never closed, or a `::` without matching opening tag, produces an `LHTMLWarning` quoting the beginning of the tag.
|
|
140
|
+
|
|
128
141
|
#### Div / Span with Styles
|
|
129
142
|
|
|
130
143
|
```
|
|
@@ -190,6 +203,8 @@ Output:
|
|
|
190
203
|
|
|
191
204
|
### Links
|
|
192
205
|
|
|
206
|
+
The URL of `link::`, `img::`, `video::` and `videoplay::` is never modified (`__`, `**`, `$` are kept). Parentheses that are part of the URL are kept (`Mercury_(planet)`, `fig(1).png`), while a group starting with `.` or `#` is a class/id group.
|
|
207
|
+
|
|
193
208
|
```
|
|
194
209
|
link::https://example.com[Click here]
|
|
195
210
|
link::page.html(.nav)[Back to home]
|
|
@@ -233,7 +248,7 @@ def hello():
|
|
|
233
248
|
code::[-]
|
|
234
249
|
````
|
|
235
250
|
|
|
236
|
-
Syntax highlighting is powered by Pygments. Any language supported by Pygments can be used.
|
|
251
|
+
`include::file` directives inside a code block insert the file as raw code. Syntax highlighting is powered by Pygments. Any language supported by Pygments can be used. An empty language (`code::[]`) renders plain text; an unknown language renders plain text with a warning.
|
|
237
252
|
|
|
238
253
|
|
|
239
254
|
### Spacer
|
|
@@ -261,12 +276,28 @@ verbatim::[-]
|
|
|
261
276
|
```
|
|
262
277
|
|
|
263
278
|
|
|
279
|
+
### Line breaks
|
|
280
|
+
|
|
281
|
+
By default, line breaks of the source have no visual effect (as in HTML). With the option `-b` / `--line-breaks` (or `line-breaks: true` in the YAML front matter, or `{'line-breaks': True}` in the `meta` passed to `lhtml.run()`), each line break of the text is rendered with `<br>`, blank lines included:
|
|
282
|
+
|
|
283
|
+
```
|
|
284
|
+
Line 1 Line 1<br>
|
|
285
|
+
Line 2 => Line 2<br>
|
|
286
|
+
<br>
|
|
287
|
+
Other paragraph Other paragraph
|
|
288
|
+
```
|
|
289
|
+
|
|
290
|
+
Images and videos are inline: images written one per line are stacked (write them on the same line to keep them side by side). Lines that start or end with a block element (headings, lists, `div::`, block-level HTML tags such as `<div>`, `<p>`, `<li>`, code and verbatim blocks, `<script>`, display math, ...) are structure: they never get a `<br>` and blank lines next to them are ignored. The content of code blocks, verbatim blocks, scripts, comments and math is never modified.
|
|
291
|
+
|
|
292
|
+
|
|
264
293
|
### Comments
|
|
265
294
|
|
|
266
295
|
```
|
|
267
296
|
Some text ::# This comment will be removed
|
|
268
297
|
```
|
|
269
298
|
|
|
299
|
+
A comment starts a line or follows whitespace, so `link::#intro[...]` (link to an anchor) is not a comment.
|
|
300
|
+
|
|
270
301
|
|
|
271
302
|
### File Inclusion
|
|
272
303
|
|
|
@@ -275,11 +306,13 @@ include::header.html
|
|
|
275
306
|
include::components/nav.html
|
|
276
307
|
```
|
|
277
308
|
|
|
278
|
-
Included files are recursively processed (up to 20 levels).
|
|
309
|
+
Included files are recursively processed (up to 20 levels); their own YAML front matter is ignored. An included file looks for its own includes first in its own directory, then in `directory_include`. A circular include raises `LHTMLIncludeLoopError`.
|
|
279
310
|
|
|
280
311
|
|
|
281
312
|
### YAML Front Matter
|
|
282
313
|
|
|
314
|
+
The front matter must be at the very beginning of the file (`---` separators elsewhere are kept as text). Input text is normalized first: a leading BOM is removed and CRLF line endings become LF.
|
|
315
|
+
|
|
283
316
|
```
|
|
284
317
|
---
|
|
285
318
|
title: "My Page"
|
|
@@ -299,6 +332,7 @@ Supported metadata keys:
|
|
|
299
332
|
| `css` | string or list | CSS files to include |
|
|
300
333
|
| `js` | string or list | JavaScript files to include |
|
|
301
334
|
| `wrap-auto` | boolean | Wrap output in full HTML document |
|
|
335
|
+
| `line-breaks` | boolean | Render source line breaks as `<br>` |
|
|
302
336
|
| `directory_include` | list | Directories to search for includes |
|
|
303
337
|
|
|
304
338
|
|
|
@@ -312,9 +346,10 @@ Register handlers for new `::` tag types:
|
|
|
312
346
|
from lhtml.pipeline import tag_registry
|
|
313
347
|
|
|
314
348
|
def handle_alert(element, tag_to_close, current_directory):
|
|
349
|
+
"""Open <div class="alert">; the following :: closes it."""
|
|
315
350
|
style = element.get('[]', '')
|
|
316
|
-
|
|
317
|
-
return f'<div class="alert" style="{style}">
|
|
351
|
+
tag_to_close.append('div')
|
|
352
|
+
return f'<div class="alert" style="{style}">', True
|
|
318
353
|
|
|
319
354
|
tag_registry.register('alert', handle_alert)
|
|
320
355
|
```
|
|
@@ -324,6 +359,12 @@ Then use in LHTML:
|
|
|
324
359
|
alert::[background:yellow; padding:10px;] Warning message ::
|
|
325
360
|
```
|
|
326
361
|
|
|
362
|
+
A handler receives the parsed element (`'[]'`, `'()'`, `'{}'`, `'text'`: the
|
|
363
|
+
word glued after the brackets) and returns `(html, is_real_tag)`. To wrap the
|
|
364
|
+
content that follows, push the HTML tag name on `tag_to_close`: the next `::`
|
|
365
|
+
(or `::name[-]`) closes it. Returning `is_real_tag=False` leaves the source
|
|
366
|
+
text unchanged.
|
|
367
|
+
|
|
327
368
|
### Custom Code Lexers
|
|
328
369
|
|
|
329
370
|
Register custom Pygments lexers for syntax highlighting:
|
|
@@ -357,6 +398,7 @@ All keys for the `meta` dict passed to `lhtml.run()`:
|
|
|
357
398
|
```python
|
|
358
399
|
{
|
|
359
400
|
'wrap-auto': False, # Wrap in HTML document
|
|
401
|
+
'line-breaks': False, # Render source line breaks as <br>
|
|
360
402
|
'title': 'Webpage', # Document title
|
|
361
403
|
'css': [], # CSS files (string or list)
|
|
362
404
|
'js': [], # JS files (string or list)
|
|
@@ -368,7 +410,7 @@ All keys for the `meta` dict passed to `lhtml.run()`:
|
|
|
368
410
|
|
|
369
411
|
## Design Principles
|
|
370
412
|
|
|
371
|
-
- **HTML-first**: Raw HTML is never modified. Only LHTML syntax triggers conversions.
|
|
413
|
+
- **HTML-first**: Raw HTML is never modified. Only LHTML syntax triggers conversions. In particular, the following are never transformed (not even by `include::` or `::#`): HTML tags and their attributes (URLs containing `__`, quoted values containing `>`, ...), `<script>` and `<style>` blocks (CSS `::before`, ...), HTML comments, and math (`$...$`, `$$...$$`, `\(...\)`, `\[...\]`, so that `$x**2$` reaches MathJax/KaTeX intact). Text between HTML tags is still processed.
|
|
372
414
|
- **Island grammar**: LHTML syntax "islands" float in a sea of opaque content (HTML, Jinja2 templates, LaTeX, etc.) that passes through untouched.
|
|
373
415
|
- **Minimal**: A few symbols (`::`, `=`, `*`, `**`, `__`, `` ` ``) cover most needs. No complex configuration required.
|
|
374
416
|
- **Composable**: LHTML works seamlessly with Jinja2 templates, making it suitable for static site generators.
|
|
@@ -389,7 +431,6 @@ src/lhtml/
|
|
|
389
431
|
listing.py # List processing
|
|
390
432
|
code.py # Code syntax highlighting (Pygments)
|
|
391
433
|
wrap_html.py # HTML document wrapping
|
|
392
|
-
ast_nodes.py # AST node dataclasses
|
|
393
434
|
errors.py # Structured error types
|
|
394
435
|
```
|
|
395
436
|
|
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "lhtml-markup"
|
|
7
|
-
version = "2.
|
|
7
|
+
version = "2.3.0"
|
|
8
8
|
description = "Lightweight HTML markup language — simplifies HTML authoring with shorthand syntax"
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
license = {text = "MIT"}
|
|
@@ -15,7 +15,8 @@ from .wrap_html import wrap_auto
|
|
|
15
15
|
|
|
16
16
|
from .process import (
|
|
17
17
|
process_yaml, process_verbatim_to_index, process_verbatim_back_from_index,
|
|
18
|
-
process_remove_comment, process_include, find_file,
|
|
18
|
+
process_remove_comment, process_include, process_include_recursive, find_file,
|
|
19
|
+
process_protect, process_unprotect, process_line_breaks,
|
|
19
20
|
process_bold, process_italic, process_code_inline,
|
|
20
21
|
process_title, process_tag, process_code,
|
|
21
22
|
process_listing,
|
|
@@ -23,13 +24,7 @@ from .process import (
|
|
|
23
24
|
|
|
24
25
|
from .errors import (
|
|
25
26
|
LHTMLError, LHTMLParseError, LHTMLFileNotFound,
|
|
26
|
-
LHTMLTagStackError, LHTMLIncludeLoopError,
|
|
27
|
-
)
|
|
28
|
-
|
|
29
|
-
from .ast_nodes import (
|
|
30
|
-
LHTMLDocument, TextNode, TagElement, HeadingNode,
|
|
31
|
-
ListNode, ListItem, InlineFormat, CodeBlock,
|
|
32
|
-
VerbatimBlock, IncludeDirective, Comment, SpacerNode, ClosingTag,
|
|
27
|
+
LHTMLTagStackError, LHTMLIncludeLoopError, LHTMLWarning,
|
|
33
28
|
)
|
|
34
29
|
|
|
35
30
|
from .pipeline import ProcessingPipeline, tag_registry, lexer_registry
|
|
@@ -79,7 +74,8 @@ __all__ = [
|
|
|
79
74
|
# Processing functions
|
|
80
75
|
'extract_bracket_elements',
|
|
81
76
|
'process_yaml', 'process_verbatim_to_index', 'process_verbatim_back_from_index',
|
|
82
|
-
'process_remove_comment', 'process_include', 'find_file',
|
|
77
|
+
'process_remove_comment', 'process_include', 'process_include_recursive', 'find_file',
|
|
78
|
+
'process_protect', 'process_unprotect', 'process_line_breaks',
|
|
83
79
|
'process_bold', 'process_italic', 'process_code_inline',
|
|
84
80
|
'process_title', 'process_tag', 'process_code',
|
|
85
81
|
'process_listing',
|
|
@@ -87,11 +83,7 @@ __all__ = [
|
|
|
87
83
|
'wrap_auto',
|
|
88
84
|
# Pipeline & plugins
|
|
89
85
|
'ProcessingPipeline', 'tag_registry', 'lexer_registry',
|
|
90
|
-
# AST
|
|
91
|
-
'LHTMLDocument', 'TextNode', 'TagElement', 'HeadingNode',
|
|
92
|
-
'ListNode', 'ListItem', 'InlineFormat', 'CodeBlock',
|
|
93
|
-
'VerbatimBlock', 'IncludeDirective', 'Comment', 'SpacerNode', 'ClosingTag',
|
|
94
86
|
# Errors
|
|
95
87
|
'LHTMLError', 'LHTMLParseError', 'LHTMLFileNotFound',
|
|
96
|
-
'LHTMLTagStackError', 'LHTMLIncludeLoopError',
|
|
88
|
+
'LHTMLTagStackError', 'LHTMLIncludeLoopError', 'LHTMLWarning',
|
|
97
89
|
]
|
|
@@ -5,13 +5,18 @@ Usage:
|
|
|
5
5
|
lhtml [-w] file.l.html -o output.html # Single file → output file
|
|
6
6
|
lhtml [-w] a.l.html b.l.html # Multiple files → .html next to sources
|
|
7
7
|
lhtml [-w] a.l.html b.l.html -o build/ # Multiple files → output directory
|
|
8
|
+
lhtml -b file.l.html # Source line breaks rendered as <br>
|
|
8
9
|
python -m lhtml [same options]
|
|
9
10
|
"""
|
|
10
11
|
|
|
11
12
|
import os
|
|
13
|
+
import sys
|
|
12
14
|
import argparse
|
|
15
|
+
import warnings
|
|
13
16
|
|
|
17
|
+
from .errors import LHTMLError
|
|
14
18
|
from .pipeline import ProcessingPipeline
|
|
19
|
+
from .process import read_source
|
|
15
20
|
|
|
16
21
|
|
|
17
22
|
_pipeline = ProcessingPipeline()
|
|
@@ -49,16 +54,22 @@ def _output_path_for(input_path, output_arg):
|
|
|
49
54
|
|
|
50
55
|
|
|
51
56
|
def _process_file(f_in, meta_base):
|
|
52
|
-
"""Process a single LHTML file and return
|
|
53
|
-
meta = dict(meta_base)
|
|
54
|
-
dir_to_include = os.path.dirname(os.path.abspath(f_in))
|
|
55
|
-
if dir_to_include:
|
|
56
|
-
meta['directory_include'] = meta.get('directory_include', []) + [dir_to_include + '/']
|
|
57
|
+
"""Process a single LHTML file and return (html, warning_messages).
|
|
57
58
|
|
|
58
|
-
|
|
59
|
-
|
|
59
|
+
Includes are looked up first in the file's directory, then in the
|
|
60
|
+
current directory.
|
|
61
|
+
"""
|
|
62
|
+
meta = dict(meta_base)
|
|
63
|
+
dir_of_file = os.path.dirname(os.path.abspath(f_in)) + '/'
|
|
64
|
+
meta['directory_include'] = [dir_of_file] + [
|
|
65
|
+
d for d in meta.get('directory_include', []) if d != dir_of_file]
|
|
66
|
+
meta['current_directory'] = dir_of_file
|
|
60
67
|
|
|
61
|
-
|
|
68
|
+
txt = _ensure_trailing_newline(read_source(f_in))
|
|
69
|
+
with warnings.catch_warnings(record=True) as caught:
|
|
70
|
+
warnings.simplefilter('always')
|
|
71
|
+
html = _ensure_trailing_newline(_pipeline.run(txt, meta))
|
|
72
|
+
return html, [str(w.message) for w in caught]
|
|
62
73
|
|
|
63
74
|
|
|
64
75
|
def main():
|
|
@@ -68,6 +79,9 @@ def main():
|
|
|
68
79
|
parser.add_argument('-w', '--wrapAuto',
|
|
69
80
|
help='Wrap content in basic HTML template',
|
|
70
81
|
action='store_true')
|
|
82
|
+
parser.add_argument('-b', '--line-breaks',
|
|
83
|
+
help='Render the line breaks of the source text as <br>',
|
|
84
|
+
action='store_true')
|
|
71
85
|
parser.add_argument('-o', '--output',
|
|
72
86
|
help='Output file (single input) or directory (multiple inputs)')
|
|
73
87
|
args = parser.parse_args()
|
|
@@ -75,23 +89,48 @@ def main():
|
|
|
75
89
|
meta = {'directory_include': [os.getcwd() + '/']}
|
|
76
90
|
if args.wrapAuto:
|
|
77
91
|
meta['wrap-auto'] = True
|
|
92
|
+
if args.line_breaks:
|
|
93
|
+
meta['line-breaks'] = True
|
|
78
94
|
|
|
79
95
|
single_file = len(args.inputFiles) == 1
|
|
80
96
|
single_to_stdout = single_file and args.output is None
|
|
81
97
|
|
|
98
|
+
if (not single_file and args.output is not None
|
|
99
|
+
and not os.path.isdir(args.output) and not args.output.endswith('/')):
|
|
100
|
+
parser.error(f'-o must be a directory when several input files are given '
|
|
101
|
+
f'(got {args.output!r}; add a trailing / to create it)')
|
|
102
|
+
|
|
103
|
+
errors = 0
|
|
82
104
|
for f_in in args.inputFiles:
|
|
83
105
|
if not os.path.isfile(f_in):
|
|
84
|
-
print(f'
|
|
106
|
+
print(f'lhtml: error: file not found [{f_in}]', file=sys.stderr)
|
|
107
|
+
errors += 1
|
|
85
108
|
continue
|
|
86
109
|
|
|
87
|
-
|
|
110
|
+
try:
|
|
111
|
+
html, messages = _process_file(f_in, meta)
|
|
112
|
+
for message in messages:
|
|
113
|
+
print(f'lhtml: warning in {f_in}: {message}', file=sys.stderr)
|
|
114
|
+
|
|
115
|
+
if single_to_stdout:
|
|
116
|
+
sys.stdout.buffer.write(html.encode('utf-8'))
|
|
117
|
+
sys.stdout.flush()
|
|
118
|
+
continue
|
|
88
119
|
|
|
89
|
-
if single_to_stdout:
|
|
90
|
-
print(html)
|
|
91
|
-
else:
|
|
92
120
|
out_path = _output_path_for(f_in, args.output)
|
|
93
|
-
|
|
121
|
+
if os.path.abspath(out_path) == os.path.abspath(f_in):
|
|
122
|
+
raise ValueError(f'output file would overwrite the input file [{out_path}]')
|
|
123
|
+
out_dir = os.path.dirname(out_path)
|
|
124
|
+
if out_dir:
|
|
125
|
+
os.makedirs(out_dir, exist_ok=True)
|
|
126
|
+
with open(out_path, 'w', encoding='utf-8') as f_out:
|
|
94
127
|
f_out.write(html)
|
|
128
|
+
except (LHTMLError, OSError, UnicodeError, ValueError) as e:
|
|
129
|
+
print(f'lhtml: error in {f_in}: {e}', file=sys.stderr)
|
|
130
|
+
errors += 1
|
|
131
|
+
|
|
132
|
+
if errors:
|
|
133
|
+
sys.exit(1)
|
|
95
134
|
|
|
96
135
|
|
|
97
136
|
if __name__ == '__main__':
|
|
@@ -4,8 +4,11 @@ Uses Pygments for highlighting. Custom lexers can be registered
|
|
|
4
4
|
via the lexer_registry in pipeline.py.
|
|
5
5
|
"""
|
|
6
6
|
|
|
7
|
+
import warnings
|
|
8
|
+
|
|
7
9
|
from pygments import highlight
|
|
8
|
-
from pygments.lexers import get_lexer_by_name
|
|
10
|
+
from pygments.lexers import get_lexer_by_name, TextLexer
|
|
11
|
+
from pygments.util import ClassNotFound
|
|
9
12
|
from pygments.formatters import HtmlFormatter
|
|
10
13
|
from pygments.lexer import words, inherit
|
|
11
14
|
import pygments.lexers
|
|
@@ -51,24 +54,33 @@ _BUILTIN_LEXERS = {
|
|
|
51
54
|
|
|
52
55
|
def export_html_code(text, language, cssclass='code'):
|
|
53
56
|
"""Highlight a code block and return HTML."""
|
|
54
|
-
|
|
57
|
+
language = (language or '').strip()
|
|
58
|
+
|
|
59
|
+
# Check plugin registry first, then built-in lexers (case-insensitive)
|
|
55
60
|
lexer_class = None
|
|
56
61
|
try:
|
|
57
62
|
from .pipeline import lexer_registry
|
|
58
|
-
lexer_class = lexer_registry.get(language)
|
|
63
|
+
lexer_class = lexer_registry.get(language) or lexer_registry.get(language.lower())
|
|
59
64
|
except ImportError:
|
|
60
65
|
pass
|
|
61
66
|
|
|
62
67
|
if lexer_class is None:
|
|
63
|
-
lexer_class = _BUILTIN_LEXERS.get(language)
|
|
68
|
+
lexer_class = _BUILTIN_LEXERS.get(language.lower())
|
|
64
69
|
|
|
70
|
+
lexer_options = dict(stripall=False, stripnl=True, ensurenl=True,
|
|
71
|
+
tabsize=2, encoding='utf-8')
|
|
65
72
|
if lexer_class is not None:
|
|
66
73
|
lexer = lexer_class()
|
|
74
|
+
elif not language:
|
|
75
|
+
lexer = TextLexer(**lexer_options)
|
|
67
76
|
else:
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
|
|
77
|
+
try:
|
|
78
|
+
lexer = get_lexer_by_name(language, **lexer_options)
|
|
79
|
+
except ClassNotFound:
|
|
80
|
+
from .errors import LHTMLWarning
|
|
81
|
+
warnings.warn(f'Unknown language {language!r} for code::[...] block, '
|
|
82
|
+
'rendered as plain text', LHTMLWarning, stacklevel=2)
|
|
83
|
+
lexer = TextLexer(**lexer_options)
|
|
72
84
|
|
|
73
85
|
formatter = HtmlFormatter(linenos=False, cssclass=cssclass)
|
|
74
86
|
return highlight(text, lexer, formatter)
|
|
@@ -20,6 +20,10 @@ class LHTMLError(Exception):
|
|
|
20
20
|
super().__init__(full_message)
|
|
21
21
|
|
|
22
22
|
|
|
23
|
+
class LHTMLWarning(UserWarning):
|
|
24
|
+
"""Category of the warnings emitted while processing LHTML markup."""
|
|
25
|
+
|
|
26
|
+
|
|
23
27
|
class LHTMLParseError(LHTMLError):
|
|
24
28
|
"""Error during parsing of LHTML markup (unclosed brackets, etc.)."""
|
|
25
29
|
pass
|
|
@@ -36,18 +40,34 @@ class LHTMLFileNotFound(LHTMLError):
|
|
|
36
40
|
|
|
37
41
|
|
|
38
42
|
class LHTMLTagStackError(LHTMLError):
|
|
39
|
-
"""A closing :: tag has no matching opening tag."""
|
|
40
|
-
|
|
41
|
-
def __init__(self,
|
|
42
|
-
|
|
43
|
+
"""A closing :: tag has no matching opening tag, or a tag is never closed."""
|
|
44
|
+
|
|
45
|
+
def __init__(self, context: str | int = '', unclosed: str | None = None,
|
|
46
|
+
source_pos: int = -1, source_line: int = -1):
|
|
47
|
+
if isinstance(context, int):
|
|
48
|
+
# LHTML 2.2 signature: LHTMLTagStackError(source_pos, source_line)
|
|
49
|
+
if isinstance(unclosed, int):
|
|
50
|
+
source_line, unclosed = unclosed, None
|
|
51
|
+
source_pos, context = context, ''
|
|
52
|
+
self.context = context
|
|
53
|
+
self.unclosed = unclosed
|
|
54
|
+
if unclosed:
|
|
55
|
+
message = f'Tag <{unclosed}> is never closed'
|
|
56
|
+
else:
|
|
57
|
+
message = 'Closing tag :: has no matching opening tag'
|
|
58
|
+
if context:
|
|
59
|
+
message += f' (near {context!r})'
|
|
43
60
|
super().__init__(message, source_pos, source_line)
|
|
44
61
|
|
|
45
62
|
|
|
46
63
|
class LHTMLIncludeLoopError(LHTMLError):
|
|
47
64
|
"""Too many include iterations — likely a circular include."""
|
|
48
65
|
|
|
49
|
-
def __init__(self, max_iterations: int = 20):
|
|
50
|
-
|
|
66
|
+
def __init__(self, max_iterations: int = 20, chain: list[str] | None = None):
|
|
67
|
+
self.chain = chain or []
|
|
68
|
+
message = f'Circular or too deep include (max depth {max_iterations})'
|
|
69
|
+
if self.chain:
|
|
70
|
+
message += ': ' + ' -> '.join(self.chain)
|
|
51
71
|
super().__init__(message)
|
|
52
72
|
|
|
53
73
|
|