lhtml-markup 2.2.0__tar.gz → 2.3.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (32) hide show
  1. {lhtml_markup-2.2.0 → lhtml_markup-2.3.0}/PKG-INFO +48 -7
  2. {lhtml_markup-2.2.0 → lhtml_markup-2.3.0}/README.md +47 -6
  3. {lhtml_markup-2.2.0 → lhtml_markup-2.3.0}/pyproject.toml +1 -1
  4. {lhtml_markup-2.2.0 → lhtml_markup-2.3.0}/src/lhtml/__init__.py +6 -14
  5. {lhtml_markup-2.2.0 → lhtml_markup-2.3.0}/src/lhtml/cli.py +53 -14
  6. {lhtml_markup-2.2.0 → lhtml_markup-2.3.0}/src/lhtml/code.py +20 -8
  7. {lhtml_markup-2.2.0 → lhtml_markup-2.3.0}/src/lhtml/errors.py +26 -6
  8. {lhtml_markup-2.2.0 → lhtml_markup-2.3.0}/src/lhtml/export_html.py +17 -9
  9. lhtml_markup-2.3.0/src/lhtml/insert_in_text.py +30 -0
  10. lhtml_markup-2.3.0/src/lhtml/patterns.py +156 -0
  11. {lhtml_markup-2.2.0 → lhtml_markup-2.3.0}/src/lhtml/pipeline.py +50 -43
  12. lhtml_markup-2.3.0/src/lhtml/process.py +590 -0
  13. {lhtml_markup-2.2.0 → lhtml_markup-2.3.0}/src/lhtml/tag_element.lark +7 -3
  14. {lhtml_markup-2.2.0 → lhtml_markup-2.3.0}/src/lhtml/tag_parser.py +34 -36
  15. {lhtml_markup-2.2.0 → lhtml_markup-2.3.0}/src/lhtml/wrap_html.py +5 -3
  16. {lhtml_markup-2.2.0 → lhtml_markup-2.3.0}/src/lhtml_markup.egg-info/PKG-INFO +48 -7
  17. {lhtml_markup-2.2.0 → lhtml_markup-2.3.0}/src/lhtml_markup.egg-info/SOURCES.txt +0 -1
  18. lhtml_markup-2.3.0/test/test_lhtml.py +1120 -0
  19. lhtml_markup-2.2.0/src/lhtml/ast_nodes.py +0 -124
  20. lhtml_markup-2.2.0/src/lhtml/insert_in_text.py +0 -7
  21. lhtml_markup-2.2.0/src/lhtml/patterns.py +0 -92
  22. lhtml_markup-2.2.0/src/lhtml/process.py +0 -253
  23. lhtml_markup-2.2.0/test/test_lhtml.py +0 -385
  24. {lhtml_markup-2.2.0 → lhtml_markup-2.3.0}/LICENSE.md +0 -0
  25. {lhtml_markup-2.2.0 → lhtml_markup-2.3.0}/setup.cfg +0 -0
  26. {lhtml_markup-2.2.0 → lhtml_markup-2.3.0}/src/lhtml/__main__.py +0 -0
  27. {lhtml_markup-2.2.0 → lhtml_markup-2.3.0}/src/lhtml/element_extract.py +0 -0
  28. {lhtml_markup-2.2.0 → lhtml_markup-2.3.0}/src/lhtml/listing.py +0 -0
  29. {lhtml_markup-2.2.0 → lhtml_markup-2.3.0}/src/lhtml_markup.egg-info/dependency_links.txt +0 -0
  30. {lhtml_markup-2.2.0 → lhtml_markup-2.3.0}/src/lhtml_markup.egg-info/entry_points.txt +0 -0
  31. {lhtml_markup-2.2.0 → lhtml_markup-2.3.0}/src/lhtml_markup.egg-info/requires.txt +0 -0
  32. {lhtml_markup-2.2.0 → lhtml_markup-2.3.0}/src/lhtml_markup.egg-info/top_level.txt +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: lhtml-markup
3
- Version: 2.2.0
3
+ Version: 2.3.0
4
4
  Summary: Lightweight HTML markup language — simplifies HTML authoring with shorthand syntax
5
5
  License: MIT
6
6
  Project-URL: Homepage, https://github.com/drohmer/lhtml
@@ -59,9 +59,14 @@ Dependencies (`lark`, `pygments`, `pyyaml`) are installed automatically.
59
59
  lhtml input.l.html # Convert to stdout
60
60
  lhtml input.l.html -o output.html # Convert to file
61
61
  lhtml input.l.html -w # Wrap in full HTML document
62
+ lhtml input.l.html -b # Render source line breaks as <br>
63
+ lhtml a.l.html b.l.html # Several files: a.html, b.html next to sources
64
+ lhtml a.l.html b.l.html -o build/ # Several files into a directory
62
65
  python -m lhtml input.l.html # Alternative invocation
63
66
  ```
64
67
 
68
+ Files are read and written as UTF-8 (a BOM is accepted). Errors and warnings are reported on stderr with the file name, the remaining files are still processed, and the exit code is non-zero if any file failed. An input file is never overwritten (e.g. `lhtml page.html` without `-o`). Includes are looked up in the input file's directory first, then in the current directory.
69
+
65
70
  ### Python API
66
71
 
67
72
  ```python
@@ -133,6 +138,8 @@ This is <em>italic</em> text.
133
138
  This is <code class="code-inline">inline code</code> text.
134
139
  ```
135
140
 
141
+ Inside inline code, `<` and `>` are escaped (so `` `std::vector<float>` `` displays correctly), and `**`, `__`, `$`, `include::`, `code::` and `::#` are not interpreted. Existing entities such as `&lt;` are kept. Named tags such as `link::` remain active, but a bare `::` is C++ (`` `::glfwInit()` ``), not a closing tag.
142
+
136
143
 
137
144
  ### Tag Elements (the `::` system)
138
145
 
@@ -144,6 +151,12 @@ tagName::(.classes #id)[cssStyle]{htmlAttributes} content ::
144
151
 
145
152
  All bracket groups are optional. If `tagName` is omitted, defaults to `div`.
146
153
 
154
+ Tag names start with a letter and may contain letters, digits, `_` and `-` (useful for custom tags). A tag is only recognized when it is not glued to a preceding word: `std::chrono::seconds`, `obj.link::x` or the Python slice `a[::2]` are left untouched.
155
+
156
+ A closing `::` may be directly followed by punctuation or HTML (`**span::[c] x ::**`, `important ::,`). In a run of colons, `::` markers are the pairs ending the run: `::::[...]` is a closing `::` followed by `::[...]`, and `x:::nl` is `x:` followed by `::nl`. A `::` between two HTML tags (as in pre-highlighted code `<span>::</span>`) is left untouched.
157
+
158
+ A tag that is opened but never closed, or a `::` without matching opening tag, produces an `LHTMLWarning` quoting the beginning of the tag.
159
+
147
160
  #### Div / Span with Styles
148
161
 
149
162
  ```
@@ -209,6 +222,8 @@ Output:
209
222
 
210
223
  ### Links
211
224
 
225
+ The URL of `link::`, `img::`, `video::` and `videoplay::` is never modified (`__`, `**`, `$` are kept). Parentheses that are part of the URL are kept (`Mercury_(planet)`, `fig(1).png`), while a group starting with `.` or `#` is a class/id group.
226
+
212
227
  ```
213
228
  link::https://example.com[Click here]
214
229
  link::page.html(.nav)[Back to home]
@@ -252,7 +267,7 @@ def hello():
252
267
  code::[-]
253
268
  ````
254
269
 
255
- Syntax highlighting is powered by Pygments. Any language supported by Pygments can be used.
270
+ `include::file` directives inside a code block insert the file as raw code. Syntax highlighting is powered by Pygments. Any language supported by Pygments can be used. An empty language (`code::[]`) renders plain text; an unknown language renders plain text with a warning.
256
271
 
257
272
 
258
273
  ### Spacer
@@ -280,12 +295,28 @@ verbatim::[-]
280
295
  ```
281
296
 
282
297
 
298
+ ### Line breaks
299
+
300
+ By default, line breaks of the source have no visual effect (as in HTML). With the option `-b` / `--line-breaks` (or `line-breaks: true` in the YAML front matter, or `{'line-breaks': True}` in the `meta` passed to `lhtml.run()`), each line break of the text is rendered with `<br>`, blank lines included:
301
+
302
+ ```
303
+ Line 1 Line 1<br>
304
+ Line 2 => Line 2<br>
305
+ <br>
306
+ Other paragraph Other paragraph
307
+ ```
308
+
309
+ Images and videos are inline: images written one per line are stacked (write them on the same line to keep them side by side). Lines that start or end with a block element (headings, lists, `div::`, block-level HTML tags such as `<div>`, `<p>`, `<li>`, code and verbatim blocks, `<script>`, display math, ...) are structure: they never get a `<br>` and blank lines next to them are ignored. The content of code blocks, verbatim blocks, scripts, comments and math is never modified.
310
+
311
+
283
312
  ### Comments
284
313
 
285
314
  ```
286
315
  Some text ::# This comment will be removed
287
316
  ```
288
317
 
318
+ A comment starts a line or follows whitespace, so `link::#intro[...]` (link to an anchor) is not a comment.
319
+
289
320
 
290
321
  ### File Inclusion
291
322
 
@@ -294,11 +325,13 @@ include::header.html
294
325
  include::components/nav.html
295
326
  ```
296
327
 
297
- Included files are recursively processed (up to 20 levels).
328
+ Included files are recursively processed (up to 20 levels); their own YAML front matter is ignored. An included file looks for its own includes first in its own directory, then in `directory_include`. A circular include raises `LHTMLIncludeLoopError`.
298
329
 
299
330
 
300
331
  ### YAML Front Matter
301
332
 
333
+ The front matter must be at the very beginning of the file (`---` separators elsewhere are kept as text). Input text is normalized first: a leading BOM is removed and CRLF line endings become LF.
334
+
302
335
  ```
303
336
  ---
304
337
  title: "My Page"
@@ -318,6 +351,7 @@ Supported metadata keys:
318
351
  | `css` | string or list | CSS files to include |
319
352
  | `js` | string or list | JavaScript files to include |
320
353
  | `wrap-auto` | boolean | Wrap output in full HTML document |
354
+ | `line-breaks` | boolean | Render source line breaks as `<br>` |
321
355
  | `directory_include` | list | Directories to search for includes |
322
356
 
323
357
 
@@ -331,9 +365,10 @@ Register handlers for new `::` tag types:
331
365
  from lhtml.pipeline import tag_registry
332
366
 
333
367
  def handle_alert(element, tag_to_close, current_directory):
368
+ """Open <div class="alert">; the following :: closes it."""
334
369
  style = element.get('[]', '')
335
- text = element.get('text', '')
336
- return f'<div class="alert" style="{style}">{text}</div>', True
370
+ tag_to_close.append('div')
371
+ return f'<div class="alert" style="{style}">', True
337
372
 
338
373
  tag_registry.register('alert', handle_alert)
339
374
  ```
@@ -343,6 +378,12 @@ Then use in LHTML:
343
378
  alert::[background:yellow; padding:10px;] Warning message ::
344
379
  ```
345
380
 
381
+ A handler receives the parsed element (`'[]'`, `'()'`, `'{}'`, `'text'`: the
382
+ word glued after the brackets) and returns `(html, is_real_tag)`. To wrap the
383
+ content that follows, push the HTML tag name on `tag_to_close`: the next `::`
384
+ (or `::name[-]`) closes it. Returning `is_real_tag=False` leaves the source
385
+ text unchanged.
386
+
346
387
  ### Custom Code Lexers
347
388
 
348
389
  Register custom Pygments lexers for syntax highlighting:
@@ -376,6 +417,7 @@ All keys for the `meta` dict passed to `lhtml.run()`:
376
417
  ```python
377
418
  {
378
419
  'wrap-auto': False, # Wrap in HTML document
420
+ 'line-breaks': False, # Render source line breaks as <br>
379
421
  'title': 'Webpage', # Document title
380
422
  'css': [], # CSS files (string or list)
381
423
  'js': [], # JS files (string or list)
@@ -387,7 +429,7 @@ All keys for the `meta` dict passed to `lhtml.run()`:
387
429
 
388
430
  ## Design Principles
389
431
 
390
- - **HTML-first**: Raw HTML is never modified. Only LHTML syntax triggers conversions.
432
+ - **HTML-first**: Raw HTML is never modified. Only LHTML syntax triggers conversions. In particular, the following are never transformed (not even by `include::` or `::#`): HTML tags and their attributes (URLs containing `__`, quoted values containing `>`, ...), `<script>` and `<style>` blocks (CSS `::before`, ...), HTML comments, and math (`$...$`, `$$...$$`, `\(...\)`, `\[...\]`, so that `$x**2$` reaches MathJax/KaTeX intact). Text between HTML tags is still processed.
391
433
  - **Island grammar**: LHTML syntax "islands" float in a sea of opaque content (HTML, Jinja2 templates, LaTeX, etc.) that passes through untouched.
392
434
  - **Minimal**: A few symbols (`::`, `=`, `*`, `**`, `__`, `` ` ``) cover most needs. No complex configuration required.
393
435
  - **Composable**: LHTML works seamlessly with Jinja2 templates, making it suitable for static site generators.
@@ -408,7 +450,6 @@ src/lhtml/
408
450
  listing.py # List processing
409
451
  code.py # Code syntax highlighting (Pygments)
410
452
  wrap_html.py # HTML document wrapping
411
- ast_nodes.py # AST node dataclasses
412
453
  errors.py # Structured error types
413
454
  ```
414
455
 
@@ -40,9 +40,14 @@ Dependencies (`lark`, `pygments`, `pyyaml`) are installed automatically.
40
40
  lhtml input.l.html # Convert to stdout
41
41
  lhtml input.l.html -o output.html # Convert to file
42
42
  lhtml input.l.html -w # Wrap in full HTML document
43
+ lhtml input.l.html -b # Render source line breaks as <br>
44
+ lhtml a.l.html b.l.html # Several files: a.html, b.html next to sources
45
+ lhtml a.l.html b.l.html -o build/ # Several files into a directory
43
46
  python -m lhtml input.l.html # Alternative invocation
44
47
  ```
45
48
 
49
+ Files are read and written as UTF-8 (a BOM is accepted). Errors and warnings are reported on stderr with the file name, the remaining files are still processed, and the exit code is non-zero if any file failed. An input file is never overwritten (e.g. `lhtml page.html` without `-o`). Includes are looked up in the input file's directory first, then in the current directory.
50
+
46
51
  ### Python API
47
52
 
48
53
  ```python
@@ -114,6 +119,8 @@ This is <em>italic</em> text.
114
119
  This is <code class="code-inline">inline code</code> text.
115
120
  ```
116
121
 
122
+ Inside inline code, `<` and `>` are escaped (so `` `std::vector<float>` `` displays correctly), and `**`, `__`, `$`, `include::`, `code::` and `::#` are not interpreted. Existing entities such as `&lt;` are kept. Named tags such as `link::` remain active, but a bare `::` is C++ (`` `::glfwInit()` ``), not a closing tag.
123
+
117
124
 
118
125
  ### Tag Elements (the `::` system)
119
126
 
@@ -125,6 +132,12 @@ tagName::(.classes #id)[cssStyle]{htmlAttributes} content ::
125
132
 
126
133
  All bracket groups are optional. If `tagName` is omitted, defaults to `div`.
127
134
 
135
+ Tag names start with a letter and may contain letters, digits, `_` and `-` (useful for custom tags). A tag is only recognized when it is not glued to a preceding word: `std::chrono::seconds`, `obj.link::x` or the Python slice `a[::2]` are left untouched.
136
+
137
+ A closing `::` may be directly followed by punctuation or HTML (`**span::[c] x ::**`, `important ::,`). In a run of colons, `::` markers are the pairs ending the run: `::::[...]` is a closing `::` followed by `::[...]`, and `x:::nl` is `x:` followed by `::nl`. A `::` between two HTML tags (as in pre-highlighted code `<span>::</span>`) is left untouched.
138
+
139
+ A tag that is opened but never closed, or a `::` without matching opening tag, produces an `LHTMLWarning` quoting the beginning of the tag.
140
+
128
141
  #### Div / Span with Styles
129
142
 
130
143
  ```
@@ -190,6 +203,8 @@ Output:
190
203
 
191
204
  ### Links
192
205
 
206
+ The URL of `link::`, `img::`, `video::` and `videoplay::` is never modified (`__`, `**`, `$` are kept). Parentheses that are part of the URL are kept (`Mercury_(planet)`, `fig(1).png`), while a group starting with `.` or `#` is a class/id group.
207
+
193
208
  ```
194
209
  link::https://example.com[Click here]
195
210
  link::page.html(.nav)[Back to home]
@@ -233,7 +248,7 @@ def hello():
233
248
  code::[-]
234
249
  ````
235
250
 
236
- Syntax highlighting is powered by Pygments. Any language supported by Pygments can be used.
251
+ `include::file` directives inside a code block insert the file as raw code. Syntax highlighting is powered by Pygments. Any language supported by Pygments can be used. An empty language (`code::[]`) renders plain text; an unknown language renders plain text with a warning.
237
252
 
238
253
 
239
254
  ### Spacer
@@ -261,12 +276,28 @@ verbatim::[-]
261
276
  ```
262
277
 
263
278
 
279
+ ### Line breaks
280
+
281
+ By default, line breaks of the source have no visual effect (as in HTML). With the option `-b` / `--line-breaks` (or `line-breaks: true` in the YAML front matter, or `{'line-breaks': True}` in the `meta` passed to `lhtml.run()`), each line break of the text is rendered with `<br>`, blank lines included:
282
+
283
+ ```
284
+ Line 1 Line 1<br>
285
+ Line 2 => Line 2<br>
286
+ <br>
287
+ Other paragraph Other paragraph
288
+ ```
289
+
290
+ Images and videos are inline: images written one per line are stacked (write them on the same line to keep them side by side). Lines that start or end with a block element (headings, lists, `div::`, block-level HTML tags such as `<div>`, `<p>`, `<li>`, code and verbatim blocks, `<script>`, display math, ...) are structure: they never get a `<br>` and blank lines next to them are ignored. The content of code blocks, verbatim blocks, scripts, comments and math is never modified.
291
+
292
+
264
293
  ### Comments
265
294
 
266
295
  ```
267
296
  Some text ::# This comment will be removed
268
297
  ```
269
298
 
299
+ A comment starts a line or follows whitespace, so `link::#intro[...]` (link to an anchor) is not a comment.
300
+
270
301
 
271
302
  ### File Inclusion
272
303
 
@@ -275,11 +306,13 @@ include::header.html
275
306
  include::components/nav.html
276
307
  ```
277
308
 
278
- Included files are recursively processed (up to 20 levels).
309
+ Included files are recursively processed (up to 20 levels); their own YAML front matter is ignored. An included file looks for its own includes first in its own directory, then in `directory_include`. A circular include raises `LHTMLIncludeLoopError`.
279
310
 
280
311
 
281
312
  ### YAML Front Matter
282
313
 
314
+ The front matter must be at the very beginning of the file (`---` separators elsewhere are kept as text). Input text is normalized first: a leading BOM is removed and CRLF line endings become LF.
315
+
283
316
  ```
284
317
  ---
285
318
  title: "My Page"
@@ -299,6 +332,7 @@ Supported metadata keys:
299
332
  | `css` | string or list | CSS files to include |
300
333
  | `js` | string or list | JavaScript files to include |
301
334
  | `wrap-auto` | boolean | Wrap output in full HTML document |
335
+ | `line-breaks` | boolean | Render source line breaks as `<br>` |
302
336
  | `directory_include` | list | Directories to search for includes |
303
337
 
304
338
 
@@ -312,9 +346,10 @@ Register handlers for new `::` tag types:
312
346
  from lhtml.pipeline import tag_registry
313
347
 
314
348
  def handle_alert(element, tag_to_close, current_directory):
349
+ """Open <div class="alert">; the following :: closes it."""
315
350
  style = element.get('[]', '')
316
- text = element.get('text', '')
317
- return f'<div class="alert" style="{style}">{text}</div>', True
351
+ tag_to_close.append('div')
352
+ return f'<div class="alert" style="{style}">', True
318
353
 
319
354
  tag_registry.register('alert', handle_alert)
320
355
  ```
@@ -324,6 +359,12 @@ Then use in LHTML:
324
359
  alert::[background:yellow; padding:10px;] Warning message ::
325
360
  ```
326
361
 
362
+ A handler receives the parsed element (`'[]'`, `'()'`, `'{}'`, `'text'`: the
363
+ word glued after the brackets) and returns `(html, is_real_tag)`. To wrap the
364
+ content that follows, push the HTML tag name on `tag_to_close`: the next `::`
365
+ (or `::name[-]`) closes it. Returning `is_real_tag=False` leaves the source
366
+ text unchanged.
367
+
327
368
  ### Custom Code Lexers
328
369
 
329
370
  Register custom Pygments lexers for syntax highlighting:
@@ -357,6 +398,7 @@ All keys for the `meta` dict passed to `lhtml.run()`:
357
398
  ```python
358
399
  {
359
400
  'wrap-auto': False, # Wrap in HTML document
401
+ 'line-breaks': False, # Render source line breaks as <br>
360
402
  'title': 'Webpage', # Document title
361
403
  'css': [], # CSS files (string or list)
362
404
  'js': [], # JS files (string or list)
@@ -368,7 +410,7 @@ All keys for the `meta` dict passed to `lhtml.run()`:
368
410
 
369
411
  ## Design Principles
370
412
 
371
- - **HTML-first**: Raw HTML is never modified. Only LHTML syntax triggers conversions.
413
+ - **HTML-first**: Raw HTML is never modified. Only LHTML syntax triggers conversions. In particular, the following are never transformed (not even by `include::` or `::#`): HTML tags and their attributes (URLs containing `__`, quoted values containing `>`, ...), `<script>` and `<style>` blocks (CSS `::before`, ...), HTML comments, and math (`$...$`, `$$...$$`, `\(...\)`, `\[...\]`, so that `$x**2$` reaches MathJax/KaTeX intact). Text between HTML tags is still processed.
372
414
  - **Island grammar**: LHTML syntax "islands" float in a sea of opaque content (HTML, Jinja2 templates, LaTeX, etc.) that passes through untouched.
373
415
  - **Minimal**: A few symbols (`::`, `=`, `*`, `**`, `__`, `` ` ``) cover most needs. No complex configuration required.
374
416
  - **Composable**: LHTML works seamlessly with Jinja2 templates, making it suitable for static site generators.
@@ -389,7 +431,6 @@ src/lhtml/
389
431
  listing.py # List processing
390
432
  code.py # Code syntax highlighting (Pygments)
391
433
  wrap_html.py # HTML document wrapping
392
- ast_nodes.py # AST node dataclasses
393
434
  errors.py # Structured error types
394
435
  ```
395
436
 
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "lhtml-markup"
7
- version = "2.2.0"
7
+ version = "2.3.0"
8
8
  description = "Lightweight HTML markup language — simplifies HTML authoring with shorthand syntax"
9
9
  readme = "README.md"
10
10
  license = {text = "MIT"}
@@ -15,7 +15,8 @@ from .wrap_html import wrap_auto
15
15
 
16
16
  from .process import (
17
17
  process_yaml, process_verbatim_to_index, process_verbatim_back_from_index,
18
- process_remove_comment, process_include, find_file,
18
+ process_remove_comment, process_include, process_include_recursive, find_file,
19
+ process_protect, process_unprotect, process_line_breaks,
19
20
  process_bold, process_italic, process_code_inline,
20
21
  process_title, process_tag, process_code,
21
22
  process_listing,
@@ -23,13 +24,7 @@ from .process import (
23
24
 
24
25
  from .errors import (
25
26
  LHTMLError, LHTMLParseError, LHTMLFileNotFound,
26
- LHTMLTagStackError, LHTMLIncludeLoopError,
27
- )
28
-
29
- from .ast_nodes import (
30
- LHTMLDocument, TextNode, TagElement, HeadingNode,
31
- ListNode, ListItem, InlineFormat, CodeBlock,
32
- VerbatimBlock, IncludeDirective, Comment, SpacerNode, ClosingTag,
27
+ LHTMLTagStackError, LHTMLIncludeLoopError, LHTMLWarning,
33
28
  )
34
29
 
35
30
  from .pipeline import ProcessingPipeline, tag_registry, lexer_registry
@@ -79,7 +74,8 @@ __all__ = [
79
74
  # Processing functions
80
75
  'extract_bracket_elements',
81
76
  'process_yaml', 'process_verbatim_to_index', 'process_verbatim_back_from_index',
82
- 'process_remove_comment', 'process_include', 'find_file',
77
+ 'process_remove_comment', 'process_include', 'process_include_recursive', 'find_file',
78
+ 'process_protect', 'process_unprotect', 'process_line_breaks',
83
79
  'process_bold', 'process_italic', 'process_code_inline',
84
80
  'process_title', 'process_tag', 'process_code',
85
81
  'process_listing',
@@ -87,11 +83,7 @@ __all__ = [
87
83
  'wrap_auto',
88
84
  # Pipeline & plugins
89
85
  'ProcessingPipeline', 'tag_registry', 'lexer_registry',
90
- # AST
91
- 'LHTMLDocument', 'TextNode', 'TagElement', 'HeadingNode',
92
- 'ListNode', 'ListItem', 'InlineFormat', 'CodeBlock',
93
- 'VerbatimBlock', 'IncludeDirective', 'Comment', 'SpacerNode', 'ClosingTag',
94
86
  # Errors
95
87
  'LHTMLError', 'LHTMLParseError', 'LHTMLFileNotFound',
96
- 'LHTMLTagStackError', 'LHTMLIncludeLoopError',
88
+ 'LHTMLTagStackError', 'LHTMLIncludeLoopError', 'LHTMLWarning',
97
89
  ]
@@ -5,13 +5,18 @@ Usage:
5
5
  lhtml [-w] file.l.html -o output.html # Single file → output file
6
6
  lhtml [-w] a.l.html b.l.html # Multiple files → .html next to sources
7
7
  lhtml [-w] a.l.html b.l.html -o build/ # Multiple files → output directory
8
+ lhtml -b file.l.html # Source line breaks rendered as <br>
8
9
  python -m lhtml [same options]
9
10
  """
10
11
 
11
12
  import os
13
+ import sys
12
14
  import argparse
15
+ import warnings
13
16
 
17
+ from .errors import LHTMLError
14
18
  from .pipeline import ProcessingPipeline
19
+ from .process import read_source
15
20
 
16
21
 
17
22
  _pipeline = ProcessingPipeline()
@@ -49,16 +54,22 @@ def _output_path_for(input_path, output_arg):
49
54
 
50
55
 
51
56
  def _process_file(f_in, meta_base):
52
- """Process a single LHTML file and return the HTML output."""
53
- meta = dict(meta_base)
54
- dir_to_include = os.path.dirname(os.path.abspath(f_in))
55
- if dir_to_include:
56
- meta['directory_include'] = meta.get('directory_include', []) + [dir_to_include + '/']
57
+ """Process a single LHTML file and return (html, warning_messages).
57
58
 
58
- with open(f_in) as fid:
59
- txt = _ensure_trailing_newline(fid.read())
59
+ Includes are looked up first in the file's directory, then in the
60
+ current directory.
61
+ """
62
+ meta = dict(meta_base)
63
+ dir_of_file = os.path.dirname(os.path.abspath(f_in)) + '/'
64
+ meta['directory_include'] = [dir_of_file] + [
65
+ d for d in meta.get('directory_include', []) if d != dir_of_file]
66
+ meta['current_directory'] = dir_of_file
60
67
 
61
- return _ensure_trailing_newline(_pipeline.run(txt, meta))
68
+ txt = _ensure_trailing_newline(read_source(f_in))
69
+ with warnings.catch_warnings(record=True) as caught:
70
+ warnings.simplefilter('always')
71
+ html = _ensure_trailing_newline(_pipeline.run(txt, meta))
72
+ return html, [str(w.message) for w in caught]
62
73
 
63
74
 
64
75
  def main():
@@ -68,6 +79,9 @@ def main():
68
79
  parser.add_argument('-w', '--wrapAuto',
69
80
  help='Wrap content in basic HTML template',
70
81
  action='store_true')
82
+ parser.add_argument('-b', '--line-breaks',
83
+ help='Render the line breaks of the source text as <br>',
84
+ action='store_true')
71
85
  parser.add_argument('-o', '--output',
72
86
  help='Output file (single input) or directory (multiple inputs)')
73
87
  args = parser.parse_args()
@@ -75,23 +89,48 @@ def main():
75
89
  meta = {'directory_include': [os.getcwd() + '/']}
76
90
  if args.wrapAuto:
77
91
  meta['wrap-auto'] = True
92
+ if args.line_breaks:
93
+ meta['line-breaks'] = True
78
94
 
79
95
  single_file = len(args.inputFiles) == 1
80
96
  single_to_stdout = single_file and args.output is None
81
97
 
98
+ if (not single_file and args.output is not None
99
+ and not os.path.isdir(args.output) and not args.output.endswith('/')):
100
+ parser.error(f'-o must be a directory when several input files are given '
101
+ f'(got {args.output!r}; add a trailing / to create it)')
102
+
103
+ errors = 0
82
104
  for f_in in args.inputFiles:
83
105
  if not os.path.isfile(f_in):
84
- print(f'Error: file not found [{f_in}]')
106
+ print(f'lhtml: error: file not found [{f_in}]', file=sys.stderr)
107
+ errors += 1
85
108
  continue
86
109
 
87
- html = _process_file(f_in, meta)
110
+ try:
111
+ html, messages = _process_file(f_in, meta)
112
+ for message in messages:
113
+ print(f'lhtml: warning in {f_in}: {message}', file=sys.stderr)
114
+
115
+ if single_to_stdout:
116
+ sys.stdout.buffer.write(html.encode('utf-8'))
117
+ sys.stdout.flush()
118
+ continue
88
119
 
89
- if single_to_stdout:
90
- print(html)
91
- else:
92
120
  out_path = _output_path_for(f_in, args.output)
93
- with open(out_path, 'w') as f_out:
121
+ if os.path.abspath(out_path) == os.path.abspath(f_in):
122
+ raise ValueError(f'output file would overwrite the input file [{out_path}]')
123
+ out_dir = os.path.dirname(out_path)
124
+ if out_dir:
125
+ os.makedirs(out_dir, exist_ok=True)
126
+ with open(out_path, 'w', encoding='utf-8') as f_out:
94
127
  f_out.write(html)
128
+ except (LHTMLError, OSError, UnicodeError, ValueError) as e:
129
+ print(f'lhtml: error in {f_in}: {e}', file=sys.stderr)
130
+ errors += 1
131
+
132
+ if errors:
133
+ sys.exit(1)
95
134
 
96
135
 
97
136
  if __name__ == '__main__':
@@ -4,8 +4,11 @@ Uses Pygments for highlighting. Custom lexers can be registered
4
4
  via the lexer_registry in pipeline.py.
5
5
  """
6
6
 
7
+ import warnings
8
+
7
9
  from pygments import highlight
8
- from pygments.lexers import get_lexer_by_name
10
+ from pygments.lexers import get_lexer_by_name, TextLexer
11
+ from pygments.util import ClassNotFound
9
12
  from pygments.formatters import HtmlFormatter
10
13
  from pygments.lexer import words, inherit
11
14
  import pygments.lexers
@@ -51,24 +54,33 @@ _BUILTIN_LEXERS = {
51
54
 
52
55
  def export_html_code(text, language, cssclass='code'):
53
56
  """Highlight a code block and return HTML."""
54
- # Check plugin registry first, then built-in lexers
57
+ language = (language or '').strip()
58
+
59
+ # Check plugin registry first, then built-in lexers (case-insensitive)
55
60
  lexer_class = None
56
61
  try:
57
62
  from .pipeline import lexer_registry
58
- lexer_class = lexer_registry.get(language)
63
+ lexer_class = lexer_registry.get(language) or lexer_registry.get(language.lower())
59
64
  except ImportError:
60
65
  pass
61
66
 
62
67
  if lexer_class is None:
63
- lexer_class = _BUILTIN_LEXERS.get(language)
68
+ lexer_class = _BUILTIN_LEXERS.get(language.lower())
64
69
 
70
+ lexer_options = dict(stripall=False, stripnl=True, ensurenl=True,
71
+ tabsize=2, encoding='utf-8')
65
72
  if lexer_class is not None:
66
73
  lexer = lexer_class()
74
+ elif not language:
75
+ lexer = TextLexer(**lexer_options)
67
76
  else:
68
- lexer = get_lexer_by_name(
69
- language, stripall=False, stripnl=True,
70
- ensurenl='True', tabsize=2, encoding='utf-8',
71
- )
77
+ try:
78
+ lexer = get_lexer_by_name(language, **lexer_options)
79
+ except ClassNotFound:
80
+ from .errors import LHTMLWarning
81
+ warnings.warn(f'Unknown language {language!r} for code::[...] block, '
82
+ 'rendered as plain text', LHTMLWarning, stacklevel=2)
83
+ lexer = TextLexer(**lexer_options)
72
84
 
73
85
  formatter = HtmlFormatter(linenos=False, cssclass=cssclass)
74
86
  return highlight(text, lexer, formatter)
@@ -20,6 +20,10 @@ class LHTMLError(Exception):
20
20
  super().__init__(full_message)
21
21
 
22
22
 
23
+ class LHTMLWarning(UserWarning):
24
+ """Category of the warnings emitted while processing LHTML markup."""
25
+
26
+
23
27
  class LHTMLParseError(LHTMLError):
24
28
  """Error during parsing of LHTML markup (unclosed brackets, etc.)."""
25
29
  pass
@@ -36,18 +40,34 @@ class LHTMLFileNotFound(LHTMLError):
36
40
 
37
41
 
38
42
  class LHTMLTagStackError(LHTMLError):
39
- """A closing :: tag has no matching opening tag."""
40
-
41
- def __init__(self, source_pos: int = -1, source_line: int = -1):
42
- message = 'Closing tag :: has no matching opening tag'
43
+ """A closing :: tag has no matching opening tag, or a tag is never closed."""
44
+
45
+ def __init__(self, context: str | int = '', unclosed: str | None = None,
46
+ source_pos: int = -1, source_line: int = -1):
47
+ if isinstance(context, int):
48
+ # LHTML 2.2 signature: LHTMLTagStackError(source_pos, source_line)
49
+ if isinstance(unclosed, int):
50
+ source_line, unclosed = unclosed, None
51
+ source_pos, context = context, ''
52
+ self.context = context
53
+ self.unclosed = unclosed
54
+ if unclosed:
55
+ message = f'Tag <{unclosed}> is never closed'
56
+ else:
57
+ message = 'Closing tag :: has no matching opening tag'
58
+ if context:
59
+ message += f' (near {context!r})'
43
60
  super().__init__(message, source_pos, source_line)
44
61
 
45
62
 
46
63
  class LHTMLIncludeLoopError(LHTMLError):
47
64
  """Too many include iterations — likely a circular include."""
48
65
 
49
- def __init__(self, max_iterations: int = 20):
50
- message = f'Too many include iterations (>{max_iterations}), possible circular include'
66
+ def __init__(self, max_iterations: int = 20, chain: list[str] | None = None):
67
+ self.chain = chain or []
68
+ message = f'Circular or too deep include (max depth {max_iterations})'
69
+ if self.chain:
70
+ message += ': ' + ' -> '.join(self.chain)
51
71
  super().__init__(message)
52
72
 
53
73