lhtml-markup 2.3.0__tar.gz → 2.4.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {lhtml_markup-2.3.0/src/lhtml_markup.egg-info → lhtml_markup-2.4.0}/PKG-INFO +8 -6
- {lhtml_markup-2.3.0 → lhtml_markup-2.4.0}/README.md +6 -4
- {lhtml_markup-2.3.0 → lhtml_markup-2.4.0}/pyproject.toml +2 -2
- {lhtml_markup-2.3.0 → lhtml_markup-2.4.0}/src/lhtml/cli.py +4 -1
- {lhtml_markup-2.3.0 → lhtml_markup-2.4.0}/src/lhtml/export_html.py +12 -3
- {lhtml_markup-2.3.0 → lhtml_markup-2.4.0}/src/lhtml/patterns.py +23 -7
- {lhtml_markup-2.3.0 → lhtml_markup-2.4.0}/src/lhtml/pipeline.py +4 -7
- {lhtml_markup-2.3.0 → lhtml_markup-2.4.0}/src/lhtml/process.py +97 -35
- {lhtml_markup-2.3.0 → lhtml_markup-2.4.0/src/lhtml_markup.egg-info}/PKG-INFO +8 -6
- {lhtml_markup-2.3.0 → lhtml_markup-2.4.0}/src/lhtml_markup.egg-info/requires.txt +0 -1
- {lhtml_markup-2.3.0 → lhtml_markup-2.4.0}/test/test_lhtml.py +212 -0
- {lhtml_markup-2.3.0 → lhtml_markup-2.4.0}/LICENSE.md +0 -0
- {lhtml_markup-2.3.0 → lhtml_markup-2.4.0}/setup.cfg +0 -0
- {lhtml_markup-2.3.0 → lhtml_markup-2.4.0}/src/lhtml/__init__.py +0 -0
- {lhtml_markup-2.3.0 → lhtml_markup-2.4.0}/src/lhtml/__main__.py +0 -0
- {lhtml_markup-2.3.0 → lhtml_markup-2.4.0}/src/lhtml/code.py +0 -0
- {lhtml_markup-2.3.0 → lhtml_markup-2.4.0}/src/lhtml/element_extract.py +0 -0
- {lhtml_markup-2.3.0 → lhtml_markup-2.4.0}/src/lhtml/errors.py +0 -0
- {lhtml_markup-2.3.0 → lhtml_markup-2.4.0}/src/lhtml/insert_in_text.py +0 -0
- {lhtml_markup-2.3.0 → lhtml_markup-2.4.0}/src/lhtml/listing.py +0 -0
- {lhtml_markup-2.3.0 → lhtml_markup-2.4.0}/src/lhtml/tag_element.lark +0 -0
- {lhtml_markup-2.3.0 → lhtml_markup-2.4.0}/src/lhtml/tag_parser.py +0 -0
- {lhtml_markup-2.3.0 → lhtml_markup-2.4.0}/src/lhtml/wrap_html.py +0 -0
- {lhtml_markup-2.3.0 → lhtml_markup-2.4.0}/src/lhtml_markup.egg-info/SOURCES.txt +0 -0
- {lhtml_markup-2.3.0 → lhtml_markup-2.4.0}/src/lhtml_markup.egg-info/dependency_links.txt +0 -0
- {lhtml_markup-2.3.0 → lhtml_markup-2.4.0}/src/lhtml_markup.egg-info/entry_points.txt +0 -0
- {lhtml_markup-2.3.0 → lhtml_markup-2.4.0}/src/lhtml_markup.egg-info/top_level.txt +0 -0
|
@@ -1,11 +1,12 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: lhtml-markup
|
|
3
|
-
Version: 2.
|
|
3
|
+
Version: 2.4.0
|
|
4
4
|
Summary: Lightweight HTML markup language — simplifies HTML authoring with shorthand syntax
|
|
5
5
|
License: MIT
|
|
6
6
|
Project-URL: Homepage, https://github.com/drohmer/lhtml
|
|
7
7
|
Project-URL: Repository, https://github.com/drohmer/lhtml
|
|
8
8
|
Project-URL: Issues, https://github.com/drohmer/lhtml/issues
|
|
9
|
+
Project-URL: Changelog, https://github.com/drohmer/lhtml/blob/main/CHANGELOG.md
|
|
9
10
|
Requires-Python: >=3.10
|
|
10
11
|
Description-Content-Type: text/markdown
|
|
11
12
|
License-File: LICENSE.md
|
|
@@ -14,12 +15,11 @@ Requires-Dist: pygments>=2.15
|
|
|
14
15
|
Requires-Dist: pyyaml>=6.0
|
|
15
16
|
Provides-Extra: dev
|
|
16
17
|
Requires-Dist: pytest>=7.0; extra == "dev"
|
|
17
|
-
Requires-Dist: ansicolors; extra == "dev"
|
|
18
18
|
Dynamic: license-file
|
|
19
19
|
|
|
20
20
|
# LHTML — Lightweight HTML
|
|
21
21
|
|
|
22
|
-
[](https://github.com/drohmer/lhtml/actions/workflows/tests.yml)
|
|
23
23
|
[](https://pypi.org/project/lhtml-markup/)
|
|
24
24
|
|
|
25
25
|
LHTML is a markup language that simplifies HTML authoring with embedded CSS styling. It is designed to be **HTML-first**: raw HTML passes through untouched, and only a few shorthand symbols (`::`, `*`, `=`, `**`, `__`) trigger conversions.
|
|
@@ -65,7 +65,7 @@ lhtml a.l.html b.l.html -o build/ # Several files into a directory
|
|
|
65
65
|
python -m lhtml input.l.html # Alternative invocation
|
|
66
66
|
```
|
|
67
67
|
|
|
68
|
-
Files are read and written as UTF-8 (a BOM is accepted). Errors and warnings are reported on stderr with the file name, the remaining files are still processed, and the exit code is non-zero if any file failed.
|
|
68
|
+
Files are read and written as UTF-8 (a BOM is accepted). Errors and warnings are reported on stderr with the file name, the remaining files are still processed, and the exit code is non-zero if any file failed. No input file in the batch is overwritten (including through symbolic or hard links) (e.g. `lhtml page.html` without `-o`). Includes are looked up in the input file's directory first, then in the current directory.
|
|
69
69
|
|
|
70
70
|
### Python API
|
|
71
71
|
|
|
@@ -222,7 +222,7 @@ Output:
|
|
|
222
222
|
|
|
223
223
|
### Links
|
|
224
224
|
|
|
225
|
-
The URL of `link::`, `img::`, `video::` and `videoplay::` is never modified (`__`, `**`, `$` are kept). Parentheses that are part of the URL are kept (`Mercury_(planet)`, `fig(1).png`), while a group starting with `.` or `#` is a class/id group.
|
|
225
|
+
The URL of `link::`, `img::`, `video::` and `videoplay::` is never modified (`__`, `**`, `$` are kept). Parentheses that are part of the URL are kept (`Mercury_(planet)`, `fig(1).png`), while a group starting with `.` or `#` is a class/id group. The URL may contain Jinja expressions (`img::{{ base }}/photo.jpg`, `link::{{ url_for('page') }}[Home]`), which are kept unchanged.
|
|
226
226
|
|
|
227
227
|
```
|
|
228
228
|
link::https://example.com[Click here]
|
|
@@ -429,7 +429,9 @@ All keys for the `meta` dict passed to `lhtml.run()`:
|
|
|
429
429
|
|
|
430
430
|
## Design Principles
|
|
431
431
|
|
|
432
|
-
- **HTML-first**: Raw HTML is never modified. Only LHTML syntax triggers conversions. In particular, the following are never transformed (not even by `include::` or `::#`): HTML tags and their attributes (URLs containing `__`, quoted values containing `>`, ...), `<script>` and `<style>` blocks (CSS `::before`, ...), HTML comments, and math (`$...$`, `$$...$$`, `\(...\)`, `\[...\]`, so that `$x**2$` reaches MathJax/KaTeX intact). Text between HTML tags is still processed.
|
|
432
|
+
- **HTML-first**: Raw HTML is never modified. Only LHTML syntax triggers conversions. In particular, the following are never transformed (not even by `include::` or `::#`): HTML tags and their attributes, including tags spanning multiple lines (URLs containing `__`, quoted values containing `>`, ...), `<script>` and `<style>` blocks (CSS `::before`, ...), HTML comments, and math (`$...$`, `$$...$$`, `\(...\)`, `\[...\]`, so that `$x**2$` reaches MathJax/KaTeX intact). Text between HTML tags is still processed.
|
|
433
|
+
- Styles, classes/IDs and HTML attributes in LHTML tag groups are preserved without inline formatting. Link labels still support formatting.
|
|
434
|
+
- Jinja2 expressions (`{{ ... }}`), statements (`{% ... %}`) and comments (`{# ... #}`) pass through unchanged.
|
|
433
435
|
- **Island grammar**: LHTML syntax "islands" float in a sea of opaque content (HTML, Jinja2 templates, LaTeX, etc.) that passes through untouched.
|
|
434
436
|
- **Minimal**: A few symbols (`::`, `=`, `*`, `**`, `__`, `` ` ``) cover most needs. No complex configuration required.
|
|
435
437
|
- **Composable**: LHTML works seamlessly with Jinja2 templates, making it suitable for static site generators.
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
# LHTML — Lightweight HTML
|
|
2
2
|
|
|
3
|
-
[](https://github.com/drohmer/lhtml/actions/workflows/tests.yml)
|
|
4
4
|
[](https://pypi.org/project/lhtml-markup/)
|
|
5
5
|
|
|
6
6
|
LHTML is a markup language that simplifies HTML authoring with embedded CSS styling. It is designed to be **HTML-first**: raw HTML passes through untouched, and only a few shorthand symbols (`::`, `*`, `=`, `**`, `__`) trigger conversions.
|
|
@@ -46,7 +46,7 @@ lhtml a.l.html b.l.html -o build/ # Several files into a directory
|
|
|
46
46
|
python -m lhtml input.l.html # Alternative invocation
|
|
47
47
|
```
|
|
48
48
|
|
|
49
|
-
Files are read and written as UTF-8 (a BOM is accepted). Errors and warnings are reported on stderr with the file name, the remaining files are still processed, and the exit code is non-zero if any file failed.
|
|
49
|
+
Files are read and written as UTF-8 (a BOM is accepted). Errors and warnings are reported on stderr with the file name, the remaining files are still processed, and the exit code is non-zero if any file failed. No input file in the batch is overwritten (including through symbolic or hard links) (e.g. `lhtml page.html` without `-o`). Includes are looked up in the input file's directory first, then in the current directory.
|
|
50
50
|
|
|
51
51
|
### Python API
|
|
52
52
|
|
|
@@ -203,7 +203,7 @@ Output:
|
|
|
203
203
|
|
|
204
204
|
### Links
|
|
205
205
|
|
|
206
|
-
The URL of `link::`, `img::`, `video::` and `videoplay::` is never modified (`__`, `**`, `$` are kept). Parentheses that are part of the URL are kept (`Mercury_(planet)`, `fig(1).png`), while a group starting with `.` or `#` is a class/id group.
|
|
206
|
+
The URL of `link::`, `img::`, `video::` and `videoplay::` is never modified (`__`, `**`, `$` are kept). Parentheses that are part of the URL are kept (`Mercury_(planet)`, `fig(1).png`), while a group starting with `.` or `#` is a class/id group. The URL may contain Jinja expressions (`img::{{ base }}/photo.jpg`, `link::{{ url_for('page') }}[Home]`), which are kept unchanged.
|
|
207
207
|
|
|
208
208
|
```
|
|
209
209
|
link::https://example.com[Click here]
|
|
@@ -410,7 +410,9 @@ All keys for the `meta` dict passed to `lhtml.run()`:
|
|
|
410
410
|
|
|
411
411
|
## Design Principles
|
|
412
412
|
|
|
413
|
-
- **HTML-first**: Raw HTML is never modified. Only LHTML syntax triggers conversions. In particular, the following are never transformed (not even by `include::` or `::#`): HTML tags and their attributes (URLs containing `__`, quoted values containing `>`, ...), `<script>` and `<style>` blocks (CSS `::before`, ...), HTML comments, and math (`$...$`, `$$...$$`, `\(...\)`, `\[...\]`, so that `$x**2$` reaches MathJax/KaTeX intact). Text between HTML tags is still processed.
|
|
413
|
+
- **HTML-first**: Raw HTML is never modified. Only LHTML syntax triggers conversions. In particular, the following are never transformed (not even by `include::` or `::#`): HTML tags and their attributes, including tags spanning multiple lines (URLs containing `__`, quoted values containing `>`, ...), `<script>` and `<style>` blocks (CSS `::before`, ...), HTML comments, and math (`$...$`, `$$...$$`, `\(...\)`, `\[...\]`, so that `$x**2$` reaches MathJax/KaTeX intact). Text between HTML tags is still processed.
|
|
414
|
+
- Styles, classes/IDs and HTML attributes in LHTML tag groups are preserved without inline formatting. Link labels still support formatting.
|
|
415
|
+
- Jinja2 expressions (`{{ ... }}`), statements (`{% ... %}`) and comments (`{# ... #}`) pass through unchanged.
|
|
414
416
|
- **Island grammar**: LHTML syntax "islands" float in a sea of opaque content (HTML, Jinja2 templates, LaTeX, etc.) that passes through untouched.
|
|
415
417
|
- **Minimal**: A few symbols (`::`, `=`, `*`, `**`, `__`, `` ` ``) cover most needs. No complex configuration required.
|
|
416
418
|
- **Composable**: LHTML works seamlessly with Jinja2 templates, making it suitable for static site generators.
|
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "lhtml-markup"
|
|
7
|
-
version = "2.
|
|
7
|
+
version = "2.4.0"
|
|
8
8
|
description = "Lightweight HTML markup language — simplifies HTML authoring with shorthand syntax"
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
license = {text = "MIT"}
|
|
@@ -19,11 +19,11 @@ dependencies = [
|
|
|
19
19
|
Homepage = "https://github.com/drohmer/lhtml"
|
|
20
20
|
Repository = "https://github.com/drohmer/lhtml"
|
|
21
21
|
Issues = "https://github.com/drohmer/lhtml/issues"
|
|
22
|
+
Changelog = "https://github.com/drohmer/lhtml/blob/main/CHANGELOG.md"
|
|
22
23
|
|
|
23
24
|
[project.optional-dependencies]
|
|
24
25
|
dev = [
|
|
25
26
|
"pytest>=7.0",
|
|
26
|
-
"ansicolors",
|
|
27
27
|
]
|
|
28
28
|
|
|
29
29
|
[project.scripts]
|
|
@@ -118,7 +118,10 @@ def main():
|
|
|
118
118
|
continue
|
|
119
119
|
|
|
120
120
|
out_path = _output_path_for(f_in, args.output)
|
|
121
|
-
if os.path.
|
|
121
|
+
if any(os.path.realpath(out_path) == os.path.realpath(source)
|
|
122
|
+
or (os.path.exists(out_path) and os.path.exists(source)
|
|
123
|
+
and os.path.samefile(out_path, source))
|
|
124
|
+
for source in args.inputFiles):
|
|
122
125
|
raise ValueError(f'output file would overwrite the input file [{out_path}]')
|
|
123
126
|
out_dir = os.path.dirname(out_path)
|
|
124
127
|
if out_dir:
|
|
@@ -9,6 +9,8 @@ import html
|
|
|
9
9
|
import os
|
|
10
10
|
import re
|
|
11
11
|
|
|
12
|
+
from .patterns import JINJA_RE
|
|
13
|
+
|
|
12
14
|
|
|
13
15
|
# ---------------------------------------------------------------------------
|
|
14
16
|
# Attribute helpers
|
|
@@ -24,8 +26,15 @@ def _build_attrs(elements):
|
|
|
24
26
|
|
|
25
27
|
|
|
26
28
|
def _attr(value):
|
|
27
|
-
"""Escape the double quotes of a value placed inside a double-quoted HTML
|
|
28
|
-
|
|
29
|
+
"""Escape the double quotes of a value placed inside a double-quoted HTML
|
|
30
|
+
attribute (Jinja zones are kept as is, they are rendered before HTML)."""
|
|
31
|
+
parts, prev = [], 0
|
|
32
|
+
for m in JINJA_RE.finditer(value):
|
|
33
|
+
parts.append(value[prev:m.start()].replace('"', '"'))
|
|
34
|
+
parts.append(m.group(0))
|
|
35
|
+
prev = m.end()
|
|
36
|
+
parts.append(value[prev:].replace('"', '"'))
|
|
37
|
+
return ''.join(parts)
|
|
29
38
|
|
|
30
39
|
|
|
31
40
|
def export_html_element_style(text):
|
|
@@ -115,7 +124,7 @@ def export_html_video(elements, default_inline='', current_directory=''):
|
|
|
115
124
|
parts = ['<video']
|
|
116
125
|
parts.append(export_html_element_inline(elements['{}']))
|
|
117
126
|
if default_inline:
|
|
118
|
-
parts.append(f' {default_inline}
|
|
127
|
+
parts.append(f' {default_inline}')
|
|
119
128
|
parts.append(export_html_element_class_and_id(elements['()']))
|
|
120
129
|
parts.append(export_html_element_style(elements['[]']))
|
|
121
130
|
if os.path.isfile(os.path.join(current_directory, poster_candidate)):
|
|
@@ -49,16 +49,29 @@ MATH = re.compile(
|
|
|
49
49
|
r'|(?<![\\$\w])\$(?![\s$])[^$\n]*?[^\s\\$]\$(?![\w$])' # $inline$ (pandoc-like rule)
|
|
50
50
|
r'|(?<![\\$\w])\$[^\s\\$]\$(?![\w$])', # $x$ (single char)
|
|
51
51
|
re.DOTALL)
|
|
52
|
-
#
|
|
53
|
-
|
|
54
|
-
HTML_TAG
|
|
52
|
+
# HTML attributes may span lines; quoted values may contain < or >.
|
|
53
|
+
_HTML_ATTRIBUTE = r"[A-Za-z_:][\w:.-]*(?:\s*=\s*(?:\"[^\"]*\"|'[^']*'|[^\s<>\"'=\x60]+))?"
|
|
54
|
+
HTML_TAG = re.compile(r'</?[A-Za-z][\w:-]*(?:\s+' + _HTML_ATTRIBUTE + r')*\s*/?>')
|
|
55
55
|
RAW_HTML_BLOCK = re.compile(r'<!--.*?-->|<(?i:(script|style))\b[^>]*>.*?</(?i:\1)\s*>', re.DOTALL)
|
|
56
56
|
# URL of link::, img::, video::, videoplay:: (parentheses are kept when they
|
|
57
57
|
# are not a (.class #id) group, e.g. Mercury_(planet) or fig(1).png)
|
|
58
|
+
# Quoted strings can contain the Jinja closing delimiter itself. A quote can
|
|
59
|
+
# only start a string (otherwise an unclosed {{ backtracks exponentially),
|
|
60
|
+
# and a new opening delimiter ends the search (an unclosed {{ does not scan
|
|
61
|
+
# the rest of the document).
|
|
62
|
+
_JINJA_STRING = r"(?:\"(?:\\.|[^\"\\])*\"|'(?:\\.|[^'\\])*')"
|
|
63
|
+
_JINJA_EXPRESSION = r'\{\{(?:' + _JINJA_STRING + r'|(?!\}\}|\{\{)[^"\'])*?\}\}'
|
|
64
|
+
JINJA = (r'\{#.*?#\}'
|
|
65
|
+
r'|' + _JINJA_EXPRESSION +
|
|
66
|
+
r'|\{%(?:' + _JINJA_STRING + r'|(?!%\}|\{%)[^"\'])*?%\}')
|
|
67
|
+
JINJA_RE = re.compile(JINJA, re.DOTALL)
|
|
58
68
|
URL_TAGS = ('link', 'img', 'video', 'videoplay')
|
|
59
|
-
|
|
69
|
+
# A URL may contain Jinja expressions (img::{{ base }}/photo.jpg)
|
|
70
|
+
URL_TOKEN = (r'(?:' + _JINJA_EXPRESSION +
|
|
71
|
+
r'|[^\s\[\](){}<>"`\x00]|\((?![.#])[^\s()\[\]{}<>"`\x00]*\))+')
|
|
60
72
|
PROTECTED = re.compile(
|
|
61
|
-
r'(?P<
|
|
73
|
+
r'(?P<jinja>' + JINJA + r')'
|
|
74
|
+
r'|(?P<comment><!--.*?-->)'
|
|
62
75
|
r'|(?P<raw><(?i:(?P<rawtag>script|style))\b[^>]*>.*?</(?i:(?P=rawtag))\s*>)'
|
|
63
76
|
r'|`(?P<icode>[^`\n]*)`'
|
|
64
77
|
r'|(?P<urltag>(?<![\w-])(?:' + '|'.join(URL_TAGS) + r')::)(?P<url>' + URL_TOKEN + r')'
|
|
@@ -66,6 +79,9 @@ PROTECTED = re.compile(
|
|
|
66
79
|
r'|(?P<tag>' + HTML_TAG.pattern + r')',
|
|
67
80
|
re.DOTALL)
|
|
68
81
|
|
|
82
|
+
# One left-to-right pass: an outer protected zone owns its contents.
|
|
83
|
+
PROTECTED_BLOCKS = re.compile(BLOCKS.pattern + '|' + PROTECTED.pattern, re.DOTALL)
|
|
84
|
+
|
|
69
85
|
# Line breaks option: elements that make a line part of the structure
|
|
70
86
|
BLOCK_TAGS = frozenset((
|
|
71
87
|
'div p h1 h2 h3 h4 h5 h6 ul ol li dl dt dd table thead tbody tfoot tr td th '
|
|
@@ -75,9 +91,9 @@ LEADING_TAG = re.compile(r'^\s*</?([A-Za-z][\w-]*)')
|
|
|
75
91
|
TRAILING_TAG = re.compile(r'</?([A-Za-z][\w-]*)(?:[^<>"\']|"[^"]*"|\'[^\']*\')*/?>\s*$')
|
|
76
92
|
LEADING_PLACEHOLDER = re.compile(r'^\s*\x00([A-Z])(\d+)\x00')
|
|
77
93
|
TRAILING_PLACEHOLDER = re.compile(r'\x00([A-Z])(\d+)\x00\s*$')
|
|
78
|
-
JINJA_LINE = re.compile(r'\s*(\{%.*%\}|\{#.*#\})\s*')
|
|
94
|
+
JINJA_LINE = re.compile(r'\s*(\{%.*%\}|\{#.*#\})\s*', re.DOTALL)
|
|
79
95
|
# <pre> and <textarea> keep their line breaks: no <br> inside
|
|
80
|
-
PREFORMATTED_TAG = re.compile(r'<(/?)(?i:pre|textarea)\
|
|
96
|
+
PREFORMATTED_TAG = re.compile(r'<(/?)(?i:pre|textarea)(?=[\s/>])')
|
|
81
97
|
BLOCK_RAW = re.compile(r'\s*(<!--|<(?i:script|style)\b|\$\$|\\\[)')
|
|
82
98
|
|
|
83
99
|
# Placeholders for protected content. They contain no LHTML syntax
|
|
@@ -19,12 +19,9 @@ from typing import Callable
|
|
|
19
19
|
|
|
20
20
|
META_DEFAULTS = {
|
|
21
21
|
'wrap-auto': False,
|
|
22
|
-
'add_title_id': False,
|
|
23
22
|
'title': 'Webpage',
|
|
24
23
|
'css': [],
|
|
25
24
|
'js': [],
|
|
26
|
-
'wrap-custom-pre': '',
|
|
27
|
-
'wrap-custom-post': '',
|
|
28
25
|
'line-breaks': False,
|
|
29
26
|
}
|
|
30
27
|
|
|
@@ -187,7 +184,7 @@ class ProcessingPipeline:
|
|
|
187
184
|
ctx.text = process_include_recursive(ctx.text, ctx.meta['directory_include'], stores)
|
|
188
185
|
|
|
189
186
|
# Phase 3: Block-level elements
|
|
190
|
-
ctx.text = process_title(ctx.text)
|
|
187
|
+
ctx.text = process_title(ctx.text, stores)
|
|
191
188
|
ctx.text = process_listing(ctx.text)
|
|
192
189
|
|
|
193
190
|
# Phase 4: Inline elements
|
|
@@ -196,11 +193,11 @@ class ProcessingPipeline:
|
|
|
196
193
|
|
|
197
194
|
# Phase 5: Tag elements (uses tag_registry). Inside inline code,
|
|
198
195
|
# only named tags (e.g. link::) are processed.
|
|
199
|
-
def
|
|
200
|
-
return stores.restore(s, 'U')
|
|
196
|
+
def resolve_attributes(s):
|
|
197
|
+
return stores.restore(stores.restore(s, 'A'), 'U')
|
|
201
198
|
|
|
202
199
|
ctx.text = process_tag(ctx.text, ctx.current_directory, self.tag_registry,
|
|
203
|
-
resolve=
|
|
200
|
+
resolve=resolve_attributes)
|
|
204
201
|
ctx.text = stores.restore(ctx.text, 'I', lambda content: render_inline_code(
|
|
205
202
|
process_tag(content, ctx.current_directory, self.tag_registry, inline=True)))
|
|
206
203
|
|
|
@@ -28,7 +28,7 @@ from .errors import (
|
|
|
28
28
|
LHTMLParseError, LHTMLWarning,
|
|
29
29
|
)
|
|
30
30
|
from .patterns import (
|
|
31
|
-
YAML_FRONTMATTER, VERBATIM_BLOCK, CODE_BLOCK,
|
|
31
|
+
YAML_FRONTMATTER, VERBATIM_BLOCK, CODE_BLOCK, PROTECTED, PROTECTED_BLOCKS,
|
|
32
32
|
HEADING, BOLD, ITALIC, INLINE_CODE,
|
|
33
33
|
COMMENT, INCLUDE, TAG_MARKER, SPACER,
|
|
34
34
|
BLOCK_TAGS, LEADING_TAG, TRAILING_TAG, LEADING_PLACEHOLDER, TRAILING_PLACEHOLDER,
|
|
@@ -46,12 +46,14 @@ class ProtectionStores:
|
|
|
46
46
|
R: raw zones: HTML tags, comments, <script>/<style>, math
|
|
47
47
|
I: inline code (content with < and > escaped)
|
|
48
48
|
U: URL of link::/img::/video::/videoplay:: tags
|
|
49
|
+
A: LHTML styles, classes/IDs and HTML attributes
|
|
49
50
|
"""
|
|
50
51
|
V: list = field(default_factory=list)
|
|
51
52
|
C: list = field(default_factory=list)
|
|
52
53
|
R: list = field(default_factory=list)
|
|
53
54
|
I: list = field(default_factory=list)
|
|
54
55
|
U: list = field(default_factory=list)
|
|
56
|
+
A: list = field(default_factory=list) # LHTML attribute group contents
|
|
55
57
|
|
|
56
58
|
def add(self, kind, entry):
|
|
57
59
|
store = getattr(self, kind)
|
|
@@ -109,24 +111,6 @@ def process_yaml(text):
|
|
|
109
111
|
# Verbatim and code blocks (protect / restore)
|
|
110
112
|
# ---------------------------------------------------------------------------
|
|
111
113
|
|
|
112
|
-
def process_blocks_to_index(text, stores, directories=None, _stack=()):
|
|
113
|
-
"""Replace verbatim:: and code:: blocks with placeholders.
|
|
114
|
-
|
|
115
|
-
The leftmost block wins: verbatim markers inside a code block are
|
|
116
|
-
displayed as code, and a code block inside verbatim stays raw.
|
|
117
|
-
If `directories` is given, include:: directives inside code blocks
|
|
118
|
-
are expanded (the included files are inserted as raw code).
|
|
119
|
-
"""
|
|
120
|
-
def _store(m):
|
|
121
|
-
if m.group('verbatim') is not None:
|
|
122
|
-
return stores.add('V', m.group('vbody'))
|
|
123
|
-
body = m.group('body')
|
|
124
|
-
if directories is not None:
|
|
125
|
-
body = _expand_includes_raw(body, directories, _stack)
|
|
126
|
-
return stores.add('C', (m.group('header'), body))
|
|
127
|
-
return regex_transform(text, BLOCKS, _store)
|
|
128
|
-
|
|
129
|
-
|
|
130
114
|
def process_verbatim_to_index(text, verbatim_index_store):
|
|
131
115
|
"""Replace verbatim blocks with placeholders."""
|
|
132
116
|
return store_to_index(text, VERBATIM_BLOCK, 'V', verbatim_index_store,
|
|
@@ -165,26 +149,82 @@ def render_inline_code(content):
|
|
|
165
149
|
return f'<code class="code-inline">{content}</code>'
|
|
166
150
|
|
|
167
151
|
|
|
168
|
-
def process_protect(text, stores):
|
|
152
|
+
def process_protect(text, stores, directories=None, _stack=()):
|
|
169
153
|
"""Replace zones that LHTML must not transform with placeholders.
|
|
170
154
|
|
|
171
155
|
Protected (the leftmost zone wins): HTML comments, <script>/<style>
|
|
172
156
|
blocks, inline code, URLs of link::/img::/video:: tags, math
|
|
173
157
|
($...$, $$...$$, \\(...\\), \\[...\\]) and HTML tags themselves (so
|
|
174
158
|
attribute values are never modified). Text between HTML tags is
|
|
175
|
-
still processed.
|
|
159
|
+
still processed. Jinja expressions, statements and comments are opaque.
|
|
160
|
+
With directories supplied, code/verbatim blocks participate in the same
|
|
161
|
+
left-to-right scan and code includes use those directories.
|
|
176
162
|
"""
|
|
177
163
|
def _protect(m):
|
|
164
|
+
if directories is not None:
|
|
165
|
+
if m.group('verbatim') is not None:
|
|
166
|
+
return stores.add('V', m.group('vbody'))
|
|
167
|
+
if m.group('code') is not None:
|
|
168
|
+
body = _expand_includes_raw(m.group('body'), directories, _stack)
|
|
169
|
+
return stores.add('C', (m.group('header'), body))
|
|
178
170
|
if m.group('icode') is not None:
|
|
179
171
|
return stores.add('I', escape_inline_code(m.group('icode')))
|
|
180
172
|
if m.group('url') is not None:
|
|
181
173
|
return m.group('urltag') + stores.add('U', m.group('url'))
|
|
182
174
|
return stores.add('R', m.group(0))
|
|
183
|
-
|
|
175
|
+
pattern = PROTECTED if directories is None else PROTECTED_BLOCKS
|
|
176
|
+
return regex_transform(text, pattern, _protect)
|
|
177
|
+
|
|
178
|
+
|
|
179
|
+
def process_protect_attributes(text, stores):
|
|
180
|
+
"""Hide attribute groups from formatting, keeping link labels active."""
|
|
181
|
+
def _heading(m):
|
|
182
|
+
if not m.group(2):
|
|
183
|
+
return m.group(0)
|
|
184
|
+
return f'{m.group(1)}({stores.add("A", m.group(2))}) {m.group(3)}'
|
|
185
|
+
text = regex_transform(text, HEADING, _heading)
|
|
186
|
+
|
|
187
|
+
parts, prev = [], 0
|
|
188
|
+
for m in TAG_MARKER.finditer(text):
|
|
189
|
+
if m.start() < prev:
|
|
190
|
+
continue
|
|
191
|
+
try:
|
|
192
|
+
element = extract_bracket_elements(text, m.end())
|
|
193
|
+
except LHTMLParseError:
|
|
194
|
+
continue
|
|
195
|
+
tag = element['tag']
|
|
196
|
+
if tag in ('include', 'code', 'verbatim') or not _is_tag_boundary(
|
|
197
|
+
text, m.start() - len(tag), tag):
|
|
198
|
+
continue
|
|
199
|
+
end = element['index_end']
|
|
200
|
+
pos = m.end()
|
|
201
|
+
while pos < end:
|
|
202
|
+
opening = text[pos]
|
|
203
|
+
if opening not in '([{':
|
|
204
|
+
pos += 1
|
|
205
|
+
continue
|
|
206
|
+
closing = {'(': ')', '[': ']', '{': '}'}[opening]
|
|
207
|
+
start = pos
|
|
208
|
+
depth = 1
|
|
209
|
+
pos += 1
|
|
210
|
+
while pos < end and depth:
|
|
211
|
+
if text[pos] == opening:
|
|
212
|
+
depth += 1
|
|
213
|
+
elif text[pos] == closing:
|
|
214
|
+
depth -= 1
|
|
215
|
+
pos += 1
|
|
216
|
+
if depth or (tag == 'link' and opening == '['):
|
|
217
|
+
continue
|
|
218
|
+
parts.append(text[prev:start + 1])
|
|
219
|
+
parts.append(stores.add('A', text[start + 1:pos - 1]))
|
|
220
|
+
prev = pos - 1
|
|
221
|
+
parts.append(text[prev:])
|
|
222
|
+
return ''.join(parts)
|
|
184
223
|
|
|
185
224
|
|
|
186
225
|
def process_unprotect(text, stores):
|
|
187
226
|
"""Restore the raw zones and URLs protected by process_protect."""
|
|
227
|
+
text = stores.restore(text, 'A')
|
|
188
228
|
text = stores.restore(text, 'R')
|
|
189
229
|
return stores.restore(text, 'U')
|
|
190
230
|
|
|
@@ -195,7 +235,7 @@ def process_unprotect(text, stores):
|
|
|
195
235
|
|
|
196
236
|
def process_remove_comment(text):
|
|
197
237
|
"""Remove ::# comments (to the end of the line)."""
|
|
198
|
-
return regex_transform(text, COMMENT, lambda m: '
|
|
238
|
+
return regex_transform(text, COMMENT, lambda m: '')
|
|
199
239
|
|
|
200
240
|
|
|
201
241
|
# ---------------------------------------------------------------------------
|
|
@@ -270,8 +310,8 @@ def process_include_recursive(text, directories, stores, _stack=()):
|
|
|
270
310
|
its own directory, then in `directories`.
|
|
271
311
|
Raises LHTMLIncludeLoopError on circular or too deep inclusion.
|
|
272
312
|
"""
|
|
273
|
-
text =
|
|
274
|
-
text =
|
|
313
|
+
text = process_protect(text, stores, directories, _stack)
|
|
314
|
+
text = process_protect_attributes(text, stores)
|
|
275
315
|
text = process_remove_comment(text)
|
|
276
316
|
|
|
277
317
|
parts = []
|
|
@@ -318,14 +358,24 @@ def process_code_inline(text):
|
|
|
318
358
|
# Headings
|
|
319
359
|
# ---------------------------------------------------------------------------
|
|
320
360
|
|
|
321
|
-
def process_title(text):
|
|
322
|
-
"""Convert = Title or =(.class) Title to <h1>Title</h1>, etc.
|
|
361
|
+
def process_title(text, stores=None):
|
|
362
|
+
"""Convert = Title or =(.class) Title to <h1>Title</h1>, etc.
|
|
363
|
+
|
|
364
|
+
With stores, the (.class #id) group protected by process_protect_attributes
|
|
365
|
+
is resolved, and the generated attributes stay protected from inline
|
|
366
|
+
formatting until process_unprotect.
|
|
367
|
+
"""
|
|
323
368
|
def _heading(m):
|
|
324
369
|
level = str(len(m.group(1)))
|
|
325
|
-
class_id =
|
|
370
|
+
class_id = m.group(2) or ''
|
|
371
|
+
if stores is not None:
|
|
372
|
+
class_id = stores.restore(class_id, 'A')
|
|
373
|
+
class_id = class_id.strip()
|
|
326
374
|
title = m.group(3)
|
|
327
375
|
if class_id:
|
|
328
376
|
attrs = export_html_element_class_and_id(class_id)
|
|
377
|
+
if stores is not None and attrs:
|
|
378
|
+
attrs = stores.add('A', attrs)
|
|
329
379
|
return f'<h{level}{attrs}>{title}</h{level}>\n'
|
|
330
380
|
return f'<h{level}>{title}</h{level}>\n'
|
|
331
381
|
return regex_transform(text, HEADING, _heading)
|
|
@@ -483,7 +533,10 @@ def process_tag(text, current_directory='', registry=None, inline=False, resolve
|
|
|
483
533
|
continue
|
|
484
534
|
if not tag and (inline or not _adjust_bare_tag(text, m, element)):
|
|
485
535
|
continue
|
|
486
|
-
|
|
536
|
+
context = text[tag_start:element['index_end']]
|
|
537
|
+
if resolve is not None:
|
|
538
|
+
context = resolve(context)
|
|
539
|
+
element['context'] = _context(context, 0, len(context))
|
|
487
540
|
if resolve is not None:
|
|
488
541
|
for key in ('text', '[]', '()', '{}'):
|
|
489
542
|
element[key] = resolve(element[key])
|
|
@@ -520,7 +573,7 @@ def _is_block_placeholder(kind, idx, stores):
|
|
|
520
573
|
if kind != 'R':
|
|
521
574
|
return False
|
|
522
575
|
entry = stores.R[idx]
|
|
523
|
-
if BLOCK_RAW.match(entry):
|
|
576
|
+
if BLOCK_RAW.match(entry) or JINJA_LINE.fullmatch(entry):
|
|
524
577
|
return True
|
|
525
578
|
m = LEADING_TAG.match(entry)
|
|
526
579
|
return bool(m) and m.group(1).lower() in BLOCK_TAGS
|
|
@@ -541,11 +594,20 @@ def _is_structure_line(line, stores):
|
|
|
541
594
|
|
|
542
595
|
|
|
543
596
|
def _preformatted_depth_change(line, stores):
|
|
544
|
-
"""
|
|
545
|
-
tags
|
|
546
|
-
|
|
547
|
-
|
|
548
|
-
|
|
597
|
+
"""Count actual preformatted tags, never tag-like text inside raw zones."""
|
|
598
|
+
tags = []
|
|
599
|
+
for placeholder in PLACEHOLDER.finditer(line):
|
|
600
|
+
if placeholder.group(1) == 'R':
|
|
601
|
+
entry = stores.R[int(placeholder.group(2))]
|
|
602
|
+
# An actual tag starts at the beginning of its stored zone.
|
|
603
|
+
# Do not scan inside comments, scripts, attributes or Jinja.
|
|
604
|
+
tags.append(PREFORMATTED_TAG.match(entry))
|
|
605
|
+
# Tags generated by handlers (and direct calls without stores) are
|
|
606
|
+
# not placeholders. Tokenize them so quoted attributes stay opaque.
|
|
607
|
+
for zone in PROTECTED.finditer(line):
|
|
608
|
+
if zone.group('tag') is not None:
|
|
609
|
+
tags.append(PREFORMATTED_TAG.match(zone.group('tag')))
|
|
610
|
+
return sum(-1 if tag.group(1) else 1 for tag in tags if tag is not None)
|
|
549
611
|
|
|
550
612
|
|
|
551
613
|
def process_line_breaks(text, stores=None):
|
|
@@ -1,11 +1,12 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: lhtml-markup
|
|
3
|
-
Version: 2.
|
|
3
|
+
Version: 2.4.0
|
|
4
4
|
Summary: Lightweight HTML markup language — simplifies HTML authoring with shorthand syntax
|
|
5
5
|
License: MIT
|
|
6
6
|
Project-URL: Homepage, https://github.com/drohmer/lhtml
|
|
7
7
|
Project-URL: Repository, https://github.com/drohmer/lhtml
|
|
8
8
|
Project-URL: Issues, https://github.com/drohmer/lhtml/issues
|
|
9
|
+
Project-URL: Changelog, https://github.com/drohmer/lhtml/blob/main/CHANGELOG.md
|
|
9
10
|
Requires-Python: >=3.10
|
|
10
11
|
Description-Content-Type: text/markdown
|
|
11
12
|
License-File: LICENSE.md
|
|
@@ -14,12 +15,11 @@ Requires-Dist: pygments>=2.15
|
|
|
14
15
|
Requires-Dist: pyyaml>=6.0
|
|
15
16
|
Provides-Extra: dev
|
|
16
17
|
Requires-Dist: pytest>=7.0; extra == "dev"
|
|
17
|
-
Requires-Dist: ansicolors; extra == "dev"
|
|
18
18
|
Dynamic: license-file
|
|
19
19
|
|
|
20
20
|
# LHTML — Lightweight HTML
|
|
21
21
|
|
|
22
|
-
[](https://github.com/drohmer/lhtml/actions/workflows/tests.yml)
|
|
23
23
|
[](https://pypi.org/project/lhtml-markup/)
|
|
24
24
|
|
|
25
25
|
LHTML is a markup language that simplifies HTML authoring with embedded CSS styling. It is designed to be **HTML-first**: raw HTML passes through untouched, and only a few shorthand symbols (`::`, `*`, `=`, `**`, `__`) trigger conversions.
|
|
@@ -65,7 +65,7 @@ lhtml a.l.html b.l.html -o build/ # Several files into a directory
|
|
|
65
65
|
python -m lhtml input.l.html # Alternative invocation
|
|
66
66
|
```
|
|
67
67
|
|
|
68
|
-
Files are read and written as UTF-8 (a BOM is accepted). Errors and warnings are reported on stderr with the file name, the remaining files are still processed, and the exit code is non-zero if any file failed.
|
|
68
|
+
Files are read and written as UTF-8 (a BOM is accepted). Errors and warnings are reported on stderr with the file name, the remaining files are still processed, and the exit code is non-zero if any file failed. No input file in the batch is overwritten (including through symbolic or hard links) (e.g. `lhtml page.html` without `-o`). Includes are looked up in the input file's directory first, then in the current directory.
|
|
69
69
|
|
|
70
70
|
### Python API
|
|
71
71
|
|
|
@@ -222,7 +222,7 @@ Output:
|
|
|
222
222
|
|
|
223
223
|
### Links
|
|
224
224
|
|
|
225
|
-
The URL of `link::`, `img::`, `video::` and `videoplay::` is never modified (`__`, `**`, `$` are kept). Parentheses that are part of the URL are kept (`Mercury_(planet)`, `fig(1).png`), while a group starting with `.` or `#` is a class/id group.
|
|
225
|
+
The URL of `link::`, `img::`, `video::` and `videoplay::` is never modified (`__`, `**`, `$` are kept). Parentheses that are part of the URL are kept (`Mercury_(planet)`, `fig(1).png`), while a group starting with `.` or `#` is a class/id group. The URL may contain Jinja expressions (`img::{{ base }}/photo.jpg`, `link::{{ url_for('page') }}[Home]`), which are kept unchanged.
|
|
226
226
|
|
|
227
227
|
```
|
|
228
228
|
link::https://example.com[Click here]
|
|
@@ -429,7 +429,9 @@ All keys for the `meta` dict passed to `lhtml.run()`:
|
|
|
429
429
|
|
|
430
430
|
## Design Principles
|
|
431
431
|
|
|
432
|
-
- **HTML-first**: Raw HTML is never modified. Only LHTML syntax triggers conversions. In particular, the following are never transformed (not even by `include::` or `::#`): HTML tags and their attributes (URLs containing `__`, quoted values containing `>`, ...), `<script>` and `<style>` blocks (CSS `::before`, ...), HTML comments, and math (`$...$`, `$$...$$`, `\(...\)`, `\[...\]`, so that `$x**2$` reaches MathJax/KaTeX intact). Text between HTML tags is still processed.
|
|
432
|
+
- **HTML-first**: Raw HTML is never modified. Only LHTML syntax triggers conversions. In particular, the following are never transformed (not even by `include::` or `::#`): HTML tags and their attributes, including tags spanning multiple lines (URLs containing `__`, quoted values containing `>`, ...), `<script>` and `<style>` blocks (CSS `::before`, ...), HTML comments, and math (`$...$`, `$$...$$`, `\(...\)`, `\[...\]`, so that `$x**2$` reaches MathJax/KaTeX intact). Text between HTML tags is still processed.
|
|
433
|
+
- Styles, classes/IDs and HTML attributes in LHTML tag groups are preserved without inline formatting. Link labels still support formatting.
|
|
434
|
+
- Jinja2 expressions (`{{ ... }}`), statements (`{% ... %}`) and comments (`{# ... #}`) pass through unchanged.
|
|
433
435
|
- **Island grammar**: LHTML syntax "islands" float in a sea of opaque content (HTML, Jinja2 templates, LaTeX, etc.) that passes through untouched.
|
|
434
436
|
- **Minimal**: A few symbols (`::`, `=`, `*`, `**`, `__`, `` ` ``) cover most needs. No complex configuration required.
|
|
435
437
|
- **Composable**: LHTML works seamlessly with Jinja2 templates, making it suitable for static site generators.
|
|
@@ -11,6 +11,7 @@ Tests cover:
|
|
|
11
11
|
import os
|
|
12
12
|
import re
|
|
13
13
|
import glob
|
|
14
|
+
import time
|
|
14
15
|
import warnings
|
|
15
16
|
|
|
16
17
|
import pytest
|
|
@@ -1118,3 +1119,214 @@ class TestLineBreaks:
|
|
|
1118
1119
|
for flag in ('-b', '--line-breaks'):
|
|
1119
1120
|
code, out, _ = TestCli._run_cli(self, monkeypatch, capsys, flag, str(f))
|
|
1120
1121
|
assert code == 0 and out == 'a<br>\nb\n'
|
|
1122
|
+
|
|
1123
|
+
|
|
1124
|
+
class TestAuditRegressions:
|
|
1125
|
+
@pytest.mark.parametrize('source', [
|
|
1126
|
+
'<script>const s = "code::[]include::missing.txt code::[-]";</script>',
|
|
1127
|
+
'<style>p::before {content:"verbatim::[]**x**verbatim::[-]"}</style>',
|
|
1128
|
+
'<!-- code::[]include::missing.txt code::[-] -->',
|
|
1129
|
+
'<a title="code::[]include::missing.txt code::[-]">**label**</a>',
|
|
1130
|
+
'$code::[]include::missing.txt code::[-]$',
|
|
1131
|
+
])
|
|
1132
|
+
def test_outer_zone_owns_block_directives(self, source):
|
|
1133
|
+
assert lhtml.run(source) == source.replace('**label**', '<strong>label</strong>')
|
|
1134
|
+
|
|
1135
|
+
def test_inline_code_owns_block_directive_after_text(self):
|
|
1136
|
+
source = '`example code::[]include::missing.txt code::[-]`'
|
|
1137
|
+
assert lhtml.run(source) == '<code class="code-inline">' + source[1:-1] + '</code>'
|
|
1138
|
+
|
|
1139
|
+
@pytest.mark.parametrize('source', [
|
|
1140
|
+
'<a\n href="page__draft__.html">**label**</a>',
|
|
1141
|
+
'<a title="first\n**second**"\n href="x">**label**</a>',
|
|
1142
|
+
'<input\n disabled\n data-path="__draft__">',
|
|
1143
|
+
])
|
|
1144
|
+
def test_multiline_html_attributes(self, source):
|
|
1145
|
+
assert lhtml.run(source) == source.replace('**label**', '<strong>label</strong>')
|
|
1146
|
+
|
|
1147
|
+
@pytest.mark.parametrize('source', [
|
|
1148
|
+
'{{ "page__draft__.html" }}',
|
|
1149
|
+
'{{ user.__class__.__name__ }}',
|
|
1150
|
+
'{% set path = "page__draft__.html" %}',
|
|
1151
|
+
'{# code::[]include::missing.txt code::[-] #}',
|
|
1152
|
+
'{{ "}} **literal**" }}',
|
|
1153
|
+
'{% set text = "%} __literal__" %}',
|
|
1154
|
+
'{{\n "__literal__"\n }}',
|
|
1155
|
+
])
|
|
1156
|
+
def test_jinja_preserved_with_surrounding_formatting(self, source):
|
|
1157
|
+
assert lhtml.run('**before** ' + source + ' __after__') == (
|
|
1158
|
+
'<strong>before</strong> ' + source + ' <em>after</em>')
|
|
1159
|
+
|
|
1160
|
+
@pytest.mark.parametrize('opening', ['{{', '{%'])
|
|
1161
|
+
def test_unclosed_jinja_with_quotes_is_fast(self, opening):
|
|
1162
|
+
source = opening + ' ' + 'il dit "oui" puis "non ' * 200 + '**b**'
|
|
1163
|
+
start = time.perf_counter()
|
|
1164
|
+
result = lhtml.run(source)
|
|
1165
|
+
assert time.perf_counter() - start < 1
|
|
1166
|
+
assert result.endswith('<strong>b</strong>')
|
|
1167
|
+
|
|
1168
|
+
def test_styles_classes_and_inline_attributes_are_not_formatted(self):
|
|
1169
|
+
source = ('div::(.my__class__)[background:url(photo__small__x.png)]'
|
|
1170
|
+
'{data-path="a__b__"} **content** ::')
|
|
1171
|
+
assert lhtml.run(source) == (
|
|
1172
|
+
'<div class="my__class__" style="background:url(photo__small__x.png)"'
|
|
1173
|
+
' data-path="a__b__"> <strong>content</strong> </div>')
|
|
1174
|
+
|
|
1175
|
+
def test_link_label_keeps_formatting_and_inline_code(self):
|
|
1176
|
+
assert lhtml.run('link::page__draft__.html(.my__class__)[**bold** `code`]') == (
|
|
1177
|
+
'<a class="my__class__" href="page__draft__.html">'
|
|
1178
|
+
'<strong>bold</strong> <code class="code-inline">code</code></a>')
|
|
1179
|
+
|
|
1180
|
+
def test_attribute_quotes_are_escaped_after_restoring(self):
|
|
1181
|
+
assert lhtml.run('div::[font-family:"__font__"] x ::') == (
|
|
1182
|
+
'<div style="font-family:"__font__""> x </div>')
|
|
1183
|
+
|
|
1184
|
+
def test_protection_applies_in_includes(self, tmp_path):
|
|
1185
|
+
source = '<a\n href="__draft__.html">{{ "__value__" }}</a>'
|
|
1186
|
+
(tmp_path / 'part.html').write_text(source)
|
|
1187
|
+
assert lhtml.run('include::part.html', {'directory_include': [tmp_path]}) == source
|
|
1188
|
+
|
|
1189
|
+
|
|
1190
|
+
class TestCliSourceProtection:
|
|
1191
|
+
_run_cli = TestCli._run_cli
|
|
1192
|
+
|
|
1193
|
+
def test_batch_cannot_overwrite_another_input(self, tmp_path, monkeypatch, capsys):
|
|
1194
|
+
first, second, third = [tmp_path / name for name in
|
|
1195
|
+
('page.l.html', 'page.html', 'other.l.html')]
|
|
1196
|
+
first.write_text('= First\n')
|
|
1197
|
+
second.write_text('Original content\n')
|
|
1198
|
+
third.write_text('= Other\n')
|
|
1199
|
+
code, _, err = self._run_cli(monkeypatch, capsys, str(first), str(second), str(third))
|
|
1200
|
+
assert code == 1 and 'overwrite' in err
|
|
1201
|
+
assert first.read_text() == '= First\n'
|
|
1202
|
+
assert second.read_text() == 'Original content\n'
|
|
1203
|
+
assert '<h1>Other</h1>' in (tmp_path / 'other.html').read_text()
|
|
1204
|
+
|
|
1205
|
+
@pytest.mark.parametrize('alias_kind', ['symlink', 'hardlink'])
|
|
1206
|
+
def test_output_alias_cannot_overwrite_source(self, tmp_path, monkeypatch, capsys, alias_kind):
|
|
1207
|
+
source, destination = tmp_path / 'page.l.html', tmp_path / 'output.html'
|
|
1208
|
+
source.write_text('= Original\n')
|
|
1209
|
+
if alias_kind == 'symlink':
|
|
1210
|
+
destination.symlink_to(source)
|
|
1211
|
+
else:
|
|
1212
|
+
os.link(source, destination)
|
|
1213
|
+
code, _, err = self._run_cli(monkeypatch, capsys, str(source), '-o', str(destination))
|
|
1214
|
+
assert code == 1 and 'overwrite' in err
|
|
1215
|
+
assert source.read_text() == '= Original\n'
|
|
1216
|
+
|
|
1217
|
+
|
|
1218
|
+
@pytest.mark.parametrize('statement', ['{% if visible %}', '{# comment #}'])
|
|
1219
|
+
def test_jinja_statement_does_not_introduce_line_breaks(statement):
|
|
1220
|
+
source = 'before\n' + statement + '\nafter'
|
|
1221
|
+
assert lhtml.run(source, {'line-breaks': True}) == source
|
|
1222
|
+
|
|
1223
|
+
|
|
1224
|
+
class TestLineBreakRegressions:
|
|
1225
|
+
ON = {'line-breaks': True}
|
|
1226
|
+
|
|
1227
|
+
@pytest.mark.parametrize('tag', ['pre', 'textarea'])
|
|
1228
|
+
@pytest.mark.parametrize('wrapper', [
|
|
1229
|
+
'<script>const s = "<{tag}>";</script>',
|
|
1230
|
+
'<!-- <{tag}> -->',
|
|
1231
|
+
'<span title="<{tag}>">label</span>',
|
|
1232
|
+
'{{ "<{tag}>" }}',
|
|
1233
|
+
])
|
|
1234
|
+
def test_fake_preformatted_tag_does_not_disable_following_breaks(self, tag, wrapper):
|
|
1235
|
+
prefix = wrapper.replace('{tag}', tag)
|
|
1236
|
+
result = lhtml.run(prefix + '\none\ntwo', self.ON)
|
|
1237
|
+
assert result.startswith(prefix)
|
|
1238
|
+
assert result.endswith('one<br>\ntwo')
|
|
1239
|
+
|
|
1240
|
+
@pytest.mark.parametrize('tag', ['pre-view', 'textarea-widget'])
|
|
1241
|
+
def test_custom_element_is_not_preformatted(self, tag):
|
|
1242
|
+
source = f'<{tag}>one\ntwo</{tag}>\nthree\nfour'
|
|
1243
|
+
assert lhtml.run(source, self.ON) == (
|
|
1244
|
+
f'<{tag}>one<br>\ntwo</{tag}><br>\nthree<br>\nfour')
|
|
1245
|
+
|
|
1246
|
+
@pytest.mark.parametrize('tag', ['pre', 'textarea', 'PRE', 'TEXTAREA'])
|
|
1247
|
+
def test_real_preformatted_block_still_preserves_newlines(self, tag):
|
|
1248
|
+
source = f'<{tag}\n class="sample">\none\ntwo\n</{tag}>\nthree\nfour'
|
|
1249
|
+
assert lhtml.run(source, self.ON) == source.replace('three\nfour', 'three<br>\nfour')
|
|
1250
|
+
|
|
1251
|
+
def test_fake_close_in_comment_does_not_end_preformatted_block(self):
|
|
1252
|
+
source = '<pre>\n<!-- </pre> -->\none\ntwo\n</pre>\nthree\nfour'
|
|
1253
|
+
assert lhtml.run(source, self.ON) == source.replace('three\nfour', 'three<br>\nfour')
|
|
1254
|
+
|
|
1255
|
+
@pytest.mark.parametrize('source, expected', [
|
|
1256
|
+
('one ::# note\ntwo', 'one <br>\ntwo'),
|
|
1257
|
+
('one ::# note\n\ntwo', 'one <br>\n<br>\ntwo'),
|
|
1258
|
+
('one ::# note', 'one '),
|
|
1259
|
+
('one\n::# note\ntwo', 'one<br>\n<br>\ntwo'),
|
|
1260
|
+
])
|
|
1261
|
+
def test_comment_removal_keeps_only_existing_newlines(self, source, expected):
|
|
1262
|
+
assert lhtml.run(source, self.ON) == expected
|
|
1263
|
+
|
|
1264
|
+
@pytest.mark.parametrize('statement', [
|
|
1265
|
+
'{%\n if visible\n%}',
|
|
1266
|
+
'{#\n a comment\n#}',
|
|
1267
|
+
'{%-\n set value = "text"\n-%}',
|
|
1268
|
+
])
|
|
1269
|
+
def test_multiline_jinja_statement_is_structure(self, statement):
|
|
1270
|
+
source = 'one\ntwo\n' + statement + '\nthree\nfour'
|
|
1271
|
+
expected = 'one<br>\ntwo\n' + statement + '\nthree<br>\nfour'
|
|
1272
|
+
assert lhtml.run(source, self.ON) == expected
|
|
1273
|
+
|
|
1274
|
+
def test_multiline_jinja_expression_remains_inline(self):
|
|
1275
|
+
source = 'one {{\n value\n}}\ntwo'
|
|
1276
|
+
assert lhtml.run(source, self.ON) == 'one {{\n value\n}}<br>\ntwo'
|
|
1277
|
+
|
|
1278
|
+
|
|
1279
|
+
class TestJinjaInUrlsAndHeadings:
|
|
1280
|
+
@pytest.mark.parametrize('source, expected', [
|
|
1281
|
+
('img::{{ url }}', '<img src="{{ url }}" alt="{{ url }}">'),
|
|
1282
|
+
('img::{{ base }}/photo__small__.jpg[width:10px]',
|
|
1283
|
+
'<img style="width:10px" src="{{ base }}/photo__small__.jpg"'
|
|
1284
|
+
' alt="{{ base }}/photo__small__.jpg">'),
|
|
1285
|
+
('link::{{ url }}[**go**]', '<a href="{{ url }}"><strong>go</strong></a>'),
|
|
1286
|
+
('link::/posts/{{ post.slug }}.html(.nav)[x]',
|
|
1287
|
+
'<a class="nav" href="/posts/{{ post.slug }}.html">x</a>'),
|
|
1288
|
+
("link::{{ url_for('page', name='a b') }}[x]",
|
|
1289
|
+
"<a href=\"{{ url_for('page', name='a b') }}\">x</a>"),
|
|
1290
|
+
])
|
|
1291
|
+
def test_jinja_expression_in_url(self, source, expected):
|
|
1292
|
+
assert lhtml.run(source) == expected
|
|
1293
|
+
|
|
1294
|
+
def test_jinja_double_quotes_are_not_escaped_in_attributes(self):
|
|
1295
|
+
assert lhtml.run('link::{{ url_for("page") }}[x]') == (
|
|
1296
|
+
'<a href="{{ url_for("page") }}">x</a>')
|
|
1297
|
+
assert lhtml.run('div::[font-family:{{ font("a") }}; content:"x"] y ::') == (
|
|
1298
|
+
'<div style="font-family:{{ font("a") }}; content:"x""> y </div>')
|
|
1299
|
+
|
|
1300
|
+
def test_jinja_in_video_url(self):
|
|
1301
|
+
assert '<source src="{{ base }}/clip.mp4"' in lhtml.run('video::{{ base }}/clip.mp4')
|
|
1302
|
+
|
|
1303
|
+
@pytest.mark.parametrize('source, expected', [
|
|
1304
|
+
('=(.t__x__) Title', '<h1 class="t__x__">Title</h1>\n'),
|
|
1305
|
+
('==(.a**b** #id__1__) **Bold** __it__',
|
|
1306
|
+
'<h2 class="a**b**" id="id__1__"><strong>Bold</strong> <em>it</em></h2>\n'),
|
|
1307
|
+
('= Title (.not__class__)', '<h1>Title (.not<em>class</em>)</h1>\n'),
|
|
1308
|
+
])
|
|
1309
|
+
def test_heading_classes_are_not_formatted(self, source, expected):
|
|
1310
|
+
assert lhtml.run(source) == expected
|
|
1311
|
+
|
|
1312
|
+
@pytest.mark.parametrize('opening', ['{{', '{%'])
|
|
1313
|
+
def test_many_unclosed_jinja_delimiters_are_fast(self, opening):
|
|
1314
|
+
source = (opening + ' x ') * 20000 + '**b**'
|
|
1315
|
+
start = time.perf_counter()
|
|
1316
|
+
result = lhtml.run(source)
|
|
1317
|
+
assert time.perf_counter() - start < 1
|
|
1318
|
+
assert result.endswith('<strong>b</strong>')
|
|
1319
|
+
|
|
1320
|
+
def test_jinja_delimiter_in_string_still_matches(self):
|
|
1321
|
+
assert lhtml.run('{{ "{{" }} __a__') == '{{ "{{" }} <em>a</em>'
|
|
1322
|
+
assert lhtml.run('{% set x = "{%" %} __a__') == '{% set x = "{%" %} <em>a</em>'
|
|
1323
|
+
|
|
1324
|
+
|
|
1325
|
+
@pytest.mark.parametrize('source, expected', [
|
|
1326
|
+
('videoplay::v.mp4[width:400px;]', '<video autoplay loop muted style="width:400px;">'),
|
|
1327
|
+
('videoplay::v.mp4(.c)', '<video autoplay loop muted class="c">'),
|
|
1328
|
+
('videoplay::v.mp4', '<video autoplay loop muted>'),
|
|
1329
|
+
('video::v.mp4[width:400px;]', '<video style="width:400px;">'),
|
|
1330
|
+
])
|
|
1331
|
+
def test_video_opening_tag_spacing(source, expected):
|
|
1332
|
+
assert lhtml.run(source).startswith(expected + '\n')
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|