lhtml-markup 2.3.0__tar.gz → 2.4.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (27) hide show
  1. {lhtml_markup-2.3.0/src/lhtml_markup.egg-info → lhtml_markup-2.4.0}/PKG-INFO +8 -6
  2. {lhtml_markup-2.3.0 → lhtml_markup-2.4.0}/README.md +6 -4
  3. {lhtml_markup-2.3.0 → lhtml_markup-2.4.0}/pyproject.toml +2 -2
  4. {lhtml_markup-2.3.0 → lhtml_markup-2.4.0}/src/lhtml/cli.py +4 -1
  5. {lhtml_markup-2.3.0 → lhtml_markup-2.4.0}/src/lhtml/export_html.py +12 -3
  6. {lhtml_markup-2.3.0 → lhtml_markup-2.4.0}/src/lhtml/patterns.py +23 -7
  7. {lhtml_markup-2.3.0 → lhtml_markup-2.4.0}/src/lhtml/pipeline.py +4 -7
  8. {lhtml_markup-2.3.0 → lhtml_markup-2.4.0}/src/lhtml/process.py +97 -35
  9. {lhtml_markup-2.3.0 → lhtml_markup-2.4.0/src/lhtml_markup.egg-info}/PKG-INFO +8 -6
  10. {lhtml_markup-2.3.0 → lhtml_markup-2.4.0}/src/lhtml_markup.egg-info/requires.txt +0 -1
  11. {lhtml_markup-2.3.0 → lhtml_markup-2.4.0}/test/test_lhtml.py +212 -0
  12. {lhtml_markup-2.3.0 → lhtml_markup-2.4.0}/LICENSE.md +0 -0
  13. {lhtml_markup-2.3.0 → lhtml_markup-2.4.0}/setup.cfg +0 -0
  14. {lhtml_markup-2.3.0 → lhtml_markup-2.4.0}/src/lhtml/__init__.py +0 -0
  15. {lhtml_markup-2.3.0 → lhtml_markup-2.4.0}/src/lhtml/__main__.py +0 -0
  16. {lhtml_markup-2.3.0 → lhtml_markup-2.4.0}/src/lhtml/code.py +0 -0
  17. {lhtml_markup-2.3.0 → lhtml_markup-2.4.0}/src/lhtml/element_extract.py +0 -0
  18. {lhtml_markup-2.3.0 → lhtml_markup-2.4.0}/src/lhtml/errors.py +0 -0
  19. {lhtml_markup-2.3.0 → lhtml_markup-2.4.0}/src/lhtml/insert_in_text.py +0 -0
  20. {lhtml_markup-2.3.0 → lhtml_markup-2.4.0}/src/lhtml/listing.py +0 -0
  21. {lhtml_markup-2.3.0 → lhtml_markup-2.4.0}/src/lhtml/tag_element.lark +0 -0
  22. {lhtml_markup-2.3.0 → lhtml_markup-2.4.0}/src/lhtml/tag_parser.py +0 -0
  23. {lhtml_markup-2.3.0 → lhtml_markup-2.4.0}/src/lhtml/wrap_html.py +0 -0
  24. {lhtml_markup-2.3.0 → lhtml_markup-2.4.0}/src/lhtml_markup.egg-info/SOURCES.txt +0 -0
  25. {lhtml_markup-2.3.0 → lhtml_markup-2.4.0}/src/lhtml_markup.egg-info/dependency_links.txt +0 -0
  26. {lhtml_markup-2.3.0 → lhtml_markup-2.4.0}/src/lhtml_markup.egg-info/entry_points.txt +0 -0
  27. {lhtml_markup-2.3.0 → lhtml_markup-2.4.0}/src/lhtml_markup.egg-info/top_level.txt +0 -0
@@ -1,11 +1,12 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: lhtml-markup
3
- Version: 2.3.0
3
+ Version: 2.4.0
4
4
  Summary: Lightweight HTML markup language — simplifies HTML authoring with shorthand syntax
5
5
  License: MIT
6
6
  Project-URL: Homepage, https://github.com/drohmer/lhtml
7
7
  Project-URL: Repository, https://github.com/drohmer/lhtml
8
8
  Project-URL: Issues, https://github.com/drohmer/lhtml/issues
9
+ Project-URL: Changelog, https://github.com/drohmer/lhtml/blob/main/CHANGELOG.md
9
10
  Requires-Python: >=3.10
10
11
  Description-Content-Type: text/markdown
11
12
  License-File: LICENSE.md
@@ -14,12 +15,11 @@ Requires-Dist: pygments>=2.15
14
15
  Requires-Dist: pyyaml>=6.0
15
16
  Provides-Extra: dev
16
17
  Requires-Dist: pytest>=7.0; extra == "dev"
17
- Requires-Dist: ansicolors; extra == "dev"
18
18
  Dynamic: license-file
19
19
 
20
20
  # LHTML — Lightweight HTML
21
21
 
22
- [![Tests](https://github.com/drohmer/lhtml/actions/workflows/tests.yml/badge.svg?branch=feature/lhtml-v2)](https://github.com/drohmer/lhtml/actions/workflows/tests.yml)
22
+ [![Tests](https://github.com/drohmer/lhtml/actions/workflows/tests.yml/badge.svg?branch=main)](https://github.com/drohmer/lhtml/actions/workflows/tests.yml)
23
23
  [![PyPI](https://img.shields.io/pypi/v/lhtml-markup)](https://pypi.org/project/lhtml-markup/)
24
24
 
25
25
  LHTML is a markup language that simplifies HTML authoring with embedded CSS styling. It is designed to be **HTML-first**: raw HTML passes through untouched, and only a few shorthand symbols (`::`, `*`, `=`, `**`, `__`) trigger conversions.
@@ -65,7 +65,7 @@ lhtml a.l.html b.l.html -o build/ # Several files into a directory
65
65
  python -m lhtml input.l.html # Alternative invocation
66
66
  ```
67
67
 
68
- Files are read and written as UTF-8 (a BOM is accepted). Errors and warnings are reported on stderr with the file name, the remaining files are still processed, and the exit code is non-zero if any file failed. An input file is never overwritten (e.g. `lhtml page.html` without `-o`). Includes are looked up in the input file's directory first, then in the current directory.
68
+ Files are read and written as UTF-8 (a BOM is accepted). Errors and warnings are reported on stderr with the file name, the remaining files are still processed, and the exit code is non-zero if any file failed. No input file in the batch is overwritten (including through symbolic or hard links) (e.g. `lhtml page.html` without `-o`). Includes are looked up in the input file's directory first, then in the current directory.
69
69
 
70
70
  ### Python API
71
71
 
@@ -222,7 +222,7 @@ Output:
222
222
 
223
223
  ### Links
224
224
 
225
- The URL of `link::`, `img::`, `video::` and `videoplay::` is never modified (`__`, `**`, `$` are kept). Parentheses that are part of the URL are kept (`Mercury_(planet)`, `fig(1).png`), while a group starting with `.` or `#` is a class/id group.
225
+ The URL of `link::`, `img::`, `video::` and `videoplay::` is never modified (`__`, `**`, `$` are kept). Parentheses that are part of the URL are kept (`Mercury_(planet)`, `fig(1).png`), while a group starting with `.` or `#` is a class/id group. The URL may contain Jinja expressions (`img::{{ base }}/photo.jpg`, `link::{{ url_for('page') }}[Home]`), which are kept unchanged.
226
226
 
227
227
  ```
228
228
  link::https://example.com[Click here]
@@ -429,7 +429,9 @@ All keys for the `meta` dict passed to `lhtml.run()`:
429
429
 
430
430
  ## Design Principles
431
431
 
432
- - **HTML-first**: Raw HTML is never modified. Only LHTML syntax triggers conversions. In particular, the following are never transformed (not even by `include::` or `::#`): HTML tags and their attributes (URLs containing `__`, quoted values containing `>`, ...), `<script>` and `<style>` blocks (CSS `::before`, ...), HTML comments, and math (`$...$`, `$$...$$`, `\(...\)`, `\[...\]`, so that `$x**2$` reaches MathJax/KaTeX intact). Text between HTML tags is still processed.
432
+ - **HTML-first**: Raw HTML is never modified. Only LHTML syntax triggers conversions. In particular, the following are never transformed (not even by `include::` or `::#`): HTML tags and their attributes, including tags spanning multiple lines (URLs containing `__`, quoted values containing `>`, ...), `<script>` and `<style>` blocks (CSS `::before`, ...), HTML comments, and math (`$...$`, `$$...$$`, `\(...\)`, `\[...\]`, so that `$x**2$` reaches MathJax/KaTeX intact). Text between HTML tags is still processed.
433
+ - Styles, classes/IDs and HTML attributes in LHTML tag groups are preserved without inline formatting. Link labels still support formatting.
434
+ - Jinja2 expressions (`{{ ... }}`), statements (`{% ... %}`) and comments (`{# ... #}`) pass through unchanged.
433
435
  - **Island grammar**: LHTML syntax "islands" float in a sea of opaque content (HTML, Jinja2 templates, LaTeX, etc.) that passes through untouched.
434
436
  - **Minimal**: A few symbols (`::`, `=`, `*`, `**`, `__`, `` ` ``) cover most needs. No complex configuration required.
435
437
  - **Composable**: LHTML works seamlessly with Jinja2 templates, making it suitable for static site generators.
@@ -1,6 +1,6 @@
1
1
  # LHTML — Lightweight HTML
2
2
 
3
- [![Tests](https://github.com/drohmer/lhtml/actions/workflows/tests.yml/badge.svg?branch=feature/lhtml-v2)](https://github.com/drohmer/lhtml/actions/workflows/tests.yml)
3
+ [![Tests](https://github.com/drohmer/lhtml/actions/workflows/tests.yml/badge.svg?branch=main)](https://github.com/drohmer/lhtml/actions/workflows/tests.yml)
4
4
  [![PyPI](https://img.shields.io/pypi/v/lhtml-markup)](https://pypi.org/project/lhtml-markup/)
5
5
 
6
6
  LHTML is a markup language that simplifies HTML authoring with embedded CSS styling. It is designed to be **HTML-first**: raw HTML passes through untouched, and only a few shorthand symbols (`::`, `*`, `=`, `**`, `__`) trigger conversions.
@@ -46,7 +46,7 @@ lhtml a.l.html b.l.html -o build/ # Several files into a directory
46
46
  python -m lhtml input.l.html # Alternative invocation
47
47
  ```
48
48
 
49
- Files are read and written as UTF-8 (a BOM is accepted). Errors and warnings are reported on stderr with the file name, the remaining files are still processed, and the exit code is non-zero if any file failed. An input file is never overwritten (e.g. `lhtml page.html` without `-o`). Includes are looked up in the input file's directory first, then in the current directory.
49
+ Files are read and written as UTF-8 (a BOM is accepted). Errors and warnings are reported on stderr with the file name, the remaining files are still processed, and the exit code is non-zero if any file failed. No input file in the batch is overwritten (including through symbolic or hard links) (e.g. `lhtml page.html` without `-o`). Includes are looked up in the input file's directory first, then in the current directory.
50
50
 
51
51
  ### Python API
52
52
 
@@ -203,7 +203,7 @@ Output:
203
203
 
204
204
  ### Links
205
205
 
206
- The URL of `link::`, `img::`, `video::` and `videoplay::` is never modified (`__`, `**`, `$` are kept). Parentheses that are part of the URL are kept (`Mercury_(planet)`, `fig(1).png`), while a group starting with `.` or `#` is a class/id group.
206
+ The URL of `link::`, `img::`, `video::` and `videoplay::` is never modified (`__`, `**`, `$` are kept). Parentheses that are part of the URL are kept (`Mercury_(planet)`, `fig(1).png`), while a group starting with `.` or `#` is a class/id group. The URL may contain Jinja expressions (`img::{{ base }}/photo.jpg`, `link::{{ url_for('page') }}[Home]`), which are kept unchanged.
207
207
 
208
208
  ```
209
209
  link::https://example.com[Click here]
@@ -410,7 +410,9 @@ All keys for the `meta` dict passed to `lhtml.run()`:
410
410
 
411
411
  ## Design Principles
412
412
 
413
- - **HTML-first**: Raw HTML is never modified. Only LHTML syntax triggers conversions. In particular, the following are never transformed (not even by `include::` or `::#`): HTML tags and their attributes (URLs containing `__`, quoted values containing `>`, ...), `<script>` and `<style>` blocks (CSS `::before`, ...), HTML comments, and math (`$...$`, `$$...$$`, `\(...\)`, `\[...\]`, so that `$x**2$` reaches MathJax/KaTeX intact). Text between HTML tags is still processed.
413
+ - **HTML-first**: Raw HTML is never modified. Only LHTML syntax triggers conversions. In particular, the following are never transformed (not even by `include::` or `::#`): HTML tags and their attributes, including tags spanning multiple lines (URLs containing `__`, quoted values containing `>`, ...), `<script>` and `<style>` blocks (CSS `::before`, ...), HTML comments, and math (`$...$`, `$$...$$`, `\(...\)`, `\[...\]`, so that `$x**2$` reaches MathJax/KaTeX intact). Text between HTML tags is still processed.
414
+ - Styles, classes/IDs and HTML attributes in LHTML tag groups are preserved without inline formatting. Link labels still support formatting.
415
+ - Jinja2 expressions (`{{ ... }}`), statements (`{% ... %}`) and comments (`{# ... #}`) pass through unchanged.
414
416
  - **Island grammar**: LHTML syntax "islands" float in a sea of opaque content (HTML, Jinja2 templates, LaTeX, etc.) that passes through untouched.
415
417
  - **Minimal**: A few symbols (`::`, `=`, `*`, `**`, `__`, `` ` ``) cover most needs. No complex configuration required.
416
418
  - **Composable**: LHTML works seamlessly with Jinja2 templates, making it suitable for static site generators.
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "lhtml-markup"
7
- version = "2.3.0"
7
+ version = "2.4.0"
8
8
  description = "Lightweight HTML markup language — simplifies HTML authoring with shorthand syntax"
9
9
  readme = "README.md"
10
10
  license = {text = "MIT"}
@@ -19,11 +19,11 @@ dependencies = [
19
19
  Homepage = "https://github.com/drohmer/lhtml"
20
20
  Repository = "https://github.com/drohmer/lhtml"
21
21
  Issues = "https://github.com/drohmer/lhtml/issues"
22
+ Changelog = "https://github.com/drohmer/lhtml/blob/main/CHANGELOG.md"
22
23
 
23
24
  [project.optional-dependencies]
24
25
  dev = [
25
26
  "pytest>=7.0",
26
- "ansicolors",
27
27
  ]
28
28
 
29
29
  [project.scripts]
@@ -118,7 +118,10 @@ def main():
118
118
  continue
119
119
 
120
120
  out_path = _output_path_for(f_in, args.output)
121
- if os.path.abspath(out_path) == os.path.abspath(f_in):
121
+ if any(os.path.realpath(out_path) == os.path.realpath(source)
122
+ or (os.path.exists(out_path) and os.path.exists(source)
123
+ and os.path.samefile(out_path, source))
124
+ for source in args.inputFiles):
122
125
  raise ValueError(f'output file would overwrite the input file [{out_path}]')
123
126
  out_dir = os.path.dirname(out_path)
124
127
  if out_dir:
@@ -9,6 +9,8 @@ import html
9
9
  import os
10
10
  import re
11
11
 
12
+ from .patterns import JINJA_RE
13
+
12
14
 
13
15
  # ---------------------------------------------------------------------------
14
16
  # Attribute helpers
@@ -24,8 +26,15 @@ def _build_attrs(elements):
24
26
 
25
27
 
26
28
  def _attr(value):
27
- """Escape the double quotes of a value placed inside a double-quoted HTML attribute."""
28
- return value.replace('"', '&quot;')
29
+ """Escape the double quotes of a value placed inside a double-quoted HTML
30
+ attribute (Jinja zones are kept as is, they are rendered before HTML)."""
31
+ parts, prev = [], 0
32
+ for m in JINJA_RE.finditer(value):
33
+ parts.append(value[prev:m.start()].replace('"', '&quot;'))
34
+ parts.append(m.group(0))
35
+ prev = m.end()
36
+ parts.append(value[prev:].replace('"', '&quot;'))
37
+ return ''.join(parts)
29
38
 
30
39
 
31
40
  def export_html_element_style(text):
@@ -115,7 +124,7 @@ def export_html_video(elements, default_inline='', current_directory=''):
115
124
  parts = ['<video']
116
125
  parts.append(export_html_element_inline(elements['{}']))
117
126
  if default_inline:
118
- parts.append(f' {default_inline} ')
127
+ parts.append(f' {default_inline}')
119
128
  parts.append(export_html_element_class_and_id(elements['()']))
120
129
  parts.append(export_html_element_style(elements['[]']))
121
130
  if os.path.isfile(os.path.join(current_directory, poster_candidate)):
@@ -49,16 +49,29 @@ MATH = re.compile(
49
49
  r'|(?<![\\$\w])\$(?![\s$])[^$\n]*?[^\s\\$]\$(?![\w$])' # $inline$ (pandoc-like rule)
50
50
  r'|(?<![\\$\w])\$[^\s\\$]\$(?![\w$])', # $x$ (single char)
51
51
  re.DOTALL)
52
- # An HTML tag: a name followed by attributes on the same line; quoted
53
- # values may contain '<' or '>'.
54
- HTML_TAG = re.compile(r'</?[A-Za-z][\w:-]*(?=[\s/>])(?:[^<>"\'\n]|"[^"\n]*"|\'[^\'\n]*\')*>')
52
+ # HTML attributes may span lines; quoted values may contain < or >.
53
+ _HTML_ATTRIBUTE = r"[A-Za-z_:][\w:.-]*(?:\s*=\s*(?:\"[^\"]*\"|'[^']*'|[^\s<>\"'=\x60]+))?"
54
+ HTML_TAG = re.compile(r'</?[A-Za-z][\w:-]*(?:\s+' + _HTML_ATTRIBUTE + r')*\s*/?>')
55
55
  RAW_HTML_BLOCK = re.compile(r'<!--.*?-->|<(?i:(script|style))\b[^>]*>.*?</(?i:\1)\s*>', re.DOTALL)
56
56
  # URL of link::, img::, video::, videoplay:: (parentheses are kept when they
57
57
  # are not a (.class #id) group, e.g. Mercury_(planet) or fig(1).png)
58
+ # Quoted strings can contain the Jinja closing delimiter itself. A quote can
59
+ # only start a string (otherwise an unclosed {{ backtracks exponentially),
60
+ # and a new opening delimiter ends the search (an unclosed {{ does not scan
61
+ # the rest of the document).
62
+ _JINJA_STRING = r"(?:\"(?:\\.|[^\"\\])*\"|'(?:\\.|[^'\\])*')"
63
+ _JINJA_EXPRESSION = r'\{\{(?:' + _JINJA_STRING + r'|(?!\}\}|\{\{)[^"\'])*?\}\}'
64
+ JINJA = (r'\{#.*?#\}'
65
+ r'|' + _JINJA_EXPRESSION +
66
+ r'|\{%(?:' + _JINJA_STRING + r'|(?!%\}|\{%)[^"\'])*?%\}')
67
+ JINJA_RE = re.compile(JINJA, re.DOTALL)
58
68
  URL_TAGS = ('link', 'img', 'video', 'videoplay')
59
- URL_TOKEN = r'(?:[^\s\[\](){}<>"`\x00]|\((?![.#])[^\s()\[\]{}<>"`\x00]*\))+'
69
+ # A URL may contain Jinja expressions (img::{{ base }}/photo.jpg)
70
+ URL_TOKEN = (r'(?:' + _JINJA_EXPRESSION +
71
+ r'|[^\s\[\](){}<>"`\x00]|\((?![.#])[^\s()\[\]{}<>"`\x00]*\))+')
60
72
  PROTECTED = re.compile(
61
- r'(?P<comment><!--.*?-->)'
73
+ r'(?P<jinja>' + JINJA + r')'
74
+ r'|(?P<comment><!--.*?-->)'
62
75
  r'|(?P<raw><(?i:(?P<rawtag>script|style))\b[^>]*>.*?</(?i:(?P=rawtag))\s*>)'
63
76
  r'|`(?P<icode>[^`\n]*)`'
64
77
  r'|(?P<urltag>(?<![\w-])(?:' + '|'.join(URL_TAGS) + r')::)(?P<url>' + URL_TOKEN + r')'
@@ -66,6 +79,9 @@ PROTECTED = re.compile(
66
79
  r'|(?P<tag>' + HTML_TAG.pattern + r')',
67
80
  re.DOTALL)
68
81
 
82
+ # One left-to-right pass: an outer protected zone owns its contents.
83
+ PROTECTED_BLOCKS = re.compile(BLOCKS.pattern + '|' + PROTECTED.pattern, re.DOTALL)
84
+
69
85
  # Line breaks option: elements that make a line part of the structure
70
86
  BLOCK_TAGS = frozenset((
71
87
  'div p h1 h2 h3 h4 h5 h6 ul ol li dl dt dd table thead tbody tfoot tr td th '
@@ -75,9 +91,9 @@ LEADING_TAG = re.compile(r'^\s*</?([A-Za-z][\w-]*)')
75
91
  TRAILING_TAG = re.compile(r'</?([A-Za-z][\w-]*)(?:[^<>"\']|"[^"]*"|\'[^\']*\')*/?>\s*$')
76
92
  LEADING_PLACEHOLDER = re.compile(r'^\s*\x00([A-Z])(\d+)\x00')
77
93
  TRAILING_PLACEHOLDER = re.compile(r'\x00([A-Z])(\d+)\x00\s*$')
78
- JINJA_LINE = re.compile(r'\s*(\{%.*%\}|\{#.*#\})\s*')
94
+ JINJA_LINE = re.compile(r'\s*(\{%.*%\}|\{#.*#\})\s*', re.DOTALL)
79
95
  # <pre> and <textarea> keep their line breaks: no <br> inside
80
- PREFORMATTED_TAG = re.compile(r'<(/?)(?i:pre|textarea)\b')
96
+ PREFORMATTED_TAG = re.compile(r'<(/?)(?i:pre|textarea)(?=[\s/>])')
81
97
  BLOCK_RAW = re.compile(r'\s*(<!--|<(?i:script|style)\b|\$\$|\\\[)')
82
98
 
83
99
  # Placeholders for protected content. They contain no LHTML syntax
@@ -19,12 +19,9 @@ from typing import Callable
19
19
 
20
20
  META_DEFAULTS = {
21
21
  'wrap-auto': False,
22
- 'add_title_id': False,
23
22
  'title': 'Webpage',
24
23
  'css': [],
25
24
  'js': [],
26
- 'wrap-custom-pre': '',
27
- 'wrap-custom-post': '',
28
25
  'line-breaks': False,
29
26
  }
30
27
 
@@ -187,7 +184,7 @@ class ProcessingPipeline:
187
184
  ctx.text = process_include_recursive(ctx.text, ctx.meta['directory_include'], stores)
188
185
 
189
186
  # Phase 3: Block-level elements
190
- ctx.text = process_title(ctx.text)
187
+ ctx.text = process_title(ctx.text, stores)
191
188
  ctx.text = process_listing(ctx.text)
192
189
 
193
190
  # Phase 4: Inline elements
@@ -196,11 +193,11 @@ class ProcessingPipeline:
196
193
 
197
194
  # Phase 5: Tag elements (uses tag_registry). Inside inline code,
198
195
  # only named tags (e.g. link::) are processed.
199
- def resolve_urls(s):
200
- return stores.restore(s, 'U')
196
+ def resolve_attributes(s):
197
+ return stores.restore(stores.restore(s, 'A'), 'U')
201
198
 
202
199
  ctx.text = process_tag(ctx.text, ctx.current_directory, self.tag_registry,
203
- resolve=resolve_urls)
200
+ resolve=resolve_attributes)
204
201
  ctx.text = stores.restore(ctx.text, 'I', lambda content: render_inline_code(
205
202
  process_tag(content, ctx.current_directory, self.tag_registry, inline=True)))
206
203
 
@@ -28,7 +28,7 @@ from .errors import (
28
28
  LHTMLParseError, LHTMLWarning,
29
29
  )
30
30
  from .patterns import (
31
- YAML_FRONTMATTER, VERBATIM_BLOCK, CODE_BLOCK, BLOCKS, PROTECTED,
31
+ YAML_FRONTMATTER, VERBATIM_BLOCK, CODE_BLOCK, PROTECTED, PROTECTED_BLOCKS,
32
32
  HEADING, BOLD, ITALIC, INLINE_CODE,
33
33
  COMMENT, INCLUDE, TAG_MARKER, SPACER,
34
34
  BLOCK_TAGS, LEADING_TAG, TRAILING_TAG, LEADING_PLACEHOLDER, TRAILING_PLACEHOLDER,
@@ -46,12 +46,14 @@ class ProtectionStores:
46
46
  R: raw zones: HTML tags, comments, <script>/<style>, math
47
47
  I: inline code (content with < and > escaped)
48
48
  U: URL of link::/img::/video::/videoplay:: tags
49
+ A: LHTML styles, classes/IDs and HTML attributes
49
50
  """
50
51
  V: list = field(default_factory=list)
51
52
  C: list = field(default_factory=list)
52
53
  R: list = field(default_factory=list)
53
54
  I: list = field(default_factory=list)
54
55
  U: list = field(default_factory=list)
56
+ A: list = field(default_factory=list) # LHTML attribute group contents
55
57
 
56
58
  def add(self, kind, entry):
57
59
  store = getattr(self, kind)
@@ -109,24 +111,6 @@ def process_yaml(text):
109
111
  # Verbatim and code blocks (protect / restore)
110
112
  # ---------------------------------------------------------------------------
111
113
 
112
- def process_blocks_to_index(text, stores, directories=None, _stack=()):
113
- """Replace verbatim:: and code:: blocks with placeholders.
114
-
115
- The leftmost block wins: verbatim markers inside a code block are
116
- displayed as code, and a code block inside verbatim stays raw.
117
- If `directories` is given, include:: directives inside code blocks
118
- are expanded (the included files are inserted as raw code).
119
- """
120
- def _store(m):
121
- if m.group('verbatim') is not None:
122
- return stores.add('V', m.group('vbody'))
123
- body = m.group('body')
124
- if directories is not None:
125
- body = _expand_includes_raw(body, directories, _stack)
126
- return stores.add('C', (m.group('header'), body))
127
- return regex_transform(text, BLOCKS, _store)
128
-
129
-
130
114
  def process_verbatim_to_index(text, verbatim_index_store):
131
115
  """Replace verbatim blocks with placeholders."""
132
116
  return store_to_index(text, VERBATIM_BLOCK, 'V', verbatim_index_store,
@@ -165,26 +149,82 @@ def render_inline_code(content):
165
149
  return f'<code class="code-inline">{content}</code>'
166
150
 
167
151
 
168
- def process_protect(text, stores):
152
+ def process_protect(text, stores, directories=None, _stack=()):
169
153
  """Replace zones that LHTML must not transform with placeholders.
170
154
 
171
155
  Protected (the leftmost zone wins): HTML comments, <script>/<style>
172
156
  blocks, inline code, URLs of link::/img::/video:: tags, math
173
157
  ($...$, $$...$$, \\(...\\), \\[...\\]) and HTML tags themselves (so
174
158
  attribute values are never modified). Text between HTML tags is
175
- still processed.
159
+ still processed. Jinja expressions, statements and comments are opaque.
160
+ With directories supplied, code/verbatim blocks participate in the same
161
+ left-to-right scan and code includes use those directories.
176
162
  """
177
163
  def _protect(m):
164
+ if directories is not None:
165
+ if m.group('verbatim') is not None:
166
+ return stores.add('V', m.group('vbody'))
167
+ if m.group('code') is not None:
168
+ body = _expand_includes_raw(m.group('body'), directories, _stack)
169
+ return stores.add('C', (m.group('header'), body))
178
170
  if m.group('icode') is not None:
179
171
  return stores.add('I', escape_inline_code(m.group('icode')))
180
172
  if m.group('url') is not None:
181
173
  return m.group('urltag') + stores.add('U', m.group('url'))
182
174
  return stores.add('R', m.group(0))
183
- return regex_transform(text, PROTECTED, _protect)
175
+ pattern = PROTECTED if directories is None else PROTECTED_BLOCKS
176
+ return regex_transform(text, pattern, _protect)
177
+
178
+
179
+ def process_protect_attributes(text, stores):
180
+ """Hide attribute groups from formatting, keeping link labels active."""
181
+ def _heading(m):
182
+ if not m.group(2):
183
+ return m.group(0)
184
+ return f'{m.group(1)}({stores.add("A", m.group(2))}) {m.group(3)}'
185
+ text = regex_transform(text, HEADING, _heading)
186
+
187
+ parts, prev = [], 0
188
+ for m in TAG_MARKER.finditer(text):
189
+ if m.start() < prev:
190
+ continue
191
+ try:
192
+ element = extract_bracket_elements(text, m.end())
193
+ except LHTMLParseError:
194
+ continue
195
+ tag = element['tag']
196
+ if tag in ('include', 'code', 'verbatim') or not _is_tag_boundary(
197
+ text, m.start() - len(tag), tag):
198
+ continue
199
+ end = element['index_end']
200
+ pos = m.end()
201
+ while pos < end:
202
+ opening = text[pos]
203
+ if opening not in '([{':
204
+ pos += 1
205
+ continue
206
+ closing = {'(': ')', '[': ']', '{': '}'}[opening]
207
+ start = pos
208
+ depth = 1
209
+ pos += 1
210
+ while pos < end and depth:
211
+ if text[pos] == opening:
212
+ depth += 1
213
+ elif text[pos] == closing:
214
+ depth -= 1
215
+ pos += 1
216
+ if depth or (tag == 'link' and opening == '['):
217
+ continue
218
+ parts.append(text[prev:start + 1])
219
+ parts.append(stores.add('A', text[start + 1:pos - 1]))
220
+ prev = pos - 1
221
+ parts.append(text[prev:])
222
+ return ''.join(parts)
184
223
 
185
224
 
186
225
  def process_unprotect(text, stores):
187
226
  """Restore the raw zones and URLs protected by process_protect."""
227
+ text = stores.restore(text, 'A')
188
228
  text = stores.restore(text, 'R')
189
229
  return stores.restore(text, 'U')
190
230
 
@@ -195,7 +235,7 @@ def process_unprotect(text, stores):
195
235
 
196
236
  def process_remove_comment(text):
197
237
  """Remove ::# comments (to the end of the line)."""
198
- return regex_transform(text, COMMENT, lambda m: '\n')
238
+ return regex_transform(text, COMMENT, lambda m: '')
199
239
 
200
240
 
201
241
  # ---------------------------------------------------------------------------
@@ -270,8 +310,8 @@ def process_include_recursive(text, directories, stores, _stack=()):
270
310
  its own directory, then in `directories`.
271
311
  Raises LHTMLIncludeLoopError on circular or too deep inclusion.
272
312
  """
273
- text = process_blocks_to_index(text, stores, directories, _stack)
274
- text = process_protect(text, stores)
313
+ text = process_protect(text, stores, directories, _stack)
314
+ text = process_protect_attributes(text, stores)
275
315
  text = process_remove_comment(text)
276
316
 
277
317
  parts = []
@@ -318,14 +358,24 @@ def process_code_inline(text):
318
358
  # Headings
319
359
  # ---------------------------------------------------------------------------
320
360
 
321
- def process_title(text):
322
- """Convert = Title or =(.class) Title to <h1>Title</h1>, etc."""
361
+ def process_title(text, stores=None):
362
+ """Convert = Title or =(.class) Title to <h1>Title</h1>, etc.
363
+
364
+ With stores, the (.class #id) group protected by process_protect_attributes
365
+ is resolved, and the generated attributes stay protected from inline
366
+ formatting until process_unprotect.
367
+ """
323
368
  def _heading(m):
324
369
  level = str(len(m.group(1)))
325
- class_id = (m.group(2) or '').strip()
370
+ class_id = m.group(2) or ''
371
+ if stores is not None:
372
+ class_id = stores.restore(class_id, 'A')
373
+ class_id = class_id.strip()
326
374
  title = m.group(3)
327
375
  if class_id:
328
376
  attrs = export_html_element_class_and_id(class_id)
377
+ if stores is not None and attrs:
378
+ attrs = stores.add('A', attrs)
329
379
  return f'<h{level}{attrs}>{title}</h{level}>\n'
330
380
  return f'<h{level}>{title}</h{level}>\n'
331
381
  return regex_transform(text, HEADING, _heading)
@@ -483,7 +533,10 @@ def process_tag(text, current_directory='', registry=None, inline=False, resolve
483
533
  continue
484
534
  if not tag and (inline or not _adjust_bare_tag(text, m, element)):
485
535
  continue
486
- element['context'] = _context(text, tag_start, element['index_end'])
536
+ context = text[tag_start:element['index_end']]
537
+ if resolve is not None:
538
+ context = resolve(context)
539
+ element['context'] = _context(context, 0, len(context))
487
540
  if resolve is not None:
488
541
  for key in ('text', '[]', '()', '{}'):
489
542
  element[key] = resolve(element[key])
@@ -520,7 +573,7 @@ def _is_block_placeholder(kind, idx, stores):
520
573
  if kind != 'R':
521
574
  return False
522
575
  entry = stores.R[idx]
523
- if BLOCK_RAW.match(entry):
576
+ if BLOCK_RAW.match(entry) or JINJA_LINE.fullmatch(entry):
524
577
  return True
525
578
  m = LEADING_TAG.match(entry)
526
579
  return bool(m) and m.group(1).lower() in BLOCK_TAGS
@@ -541,11 +594,20 @@ def _is_structure_line(line, stores):
541
594
 
542
595
 
543
596
  def _preformatted_depth_change(line, stores):
544
- """Net number of <pre>/<textarea> elements opened by the line (raw HTML
545
- tags are still placeholders, so they are resolved)."""
546
- resolved = PLACEHOLDER.sub(
547
- lambda m: stores.R[int(m.group(2))] if m.group(1) == 'R' else '', line)
548
- return sum(-1 if m.group(1) else 1 for m in PREFORMATTED_TAG.finditer(resolved))
597
+ """Count actual preformatted tags, never tag-like text inside raw zones."""
598
+ tags = []
599
+ for placeholder in PLACEHOLDER.finditer(line):
600
+ if placeholder.group(1) == 'R':
601
+ entry = stores.R[int(placeholder.group(2))]
602
+ # An actual tag starts at the beginning of its stored zone.
603
+ # Do not scan inside comments, scripts, attributes or Jinja.
604
+ tags.append(PREFORMATTED_TAG.match(entry))
605
+ # Tags generated by handlers (and direct calls without stores) are
606
+ # not placeholders. Tokenize them so quoted attributes stay opaque.
607
+ for zone in PROTECTED.finditer(line):
608
+ if zone.group('tag') is not None:
609
+ tags.append(PREFORMATTED_TAG.match(zone.group('tag')))
610
+ return sum(-1 if tag.group(1) else 1 for tag in tags if tag is not None)
549
611
 
550
612
 
551
613
  def process_line_breaks(text, stores=None):
@@ -1,11 +1,12 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: lhtml-markup
3
- Version: 2.3.0
3
+ Version: 2.4.0
4
4
  Summary: Lightweight HTML markup language — simplifies HTML authoring with shorthand syntax
5
5
  License: MIT
6
6
  Project-URL: Homepage, https://github.com/drohmer/lhtml
7
7
  Project-URL: Repository, https://github.com/drohmer/lhtml
8
8
  Project-URL: Issues, https://github.com/drohmer/lhtml/issues
9
+ Project-URL: Changelog, https://github.com/drohmer/lhtml/blob/main/CHANGELOG.md
9
10
  Requires-Python: >=3.10
10
11
  Description-Content-Type: text/markdown
11
12
  License-File: LICENSE.md
@@ -14,12 +15,11 @@ Requires-Dist: pygments>=2.15
14
15
  Requires-Dist: pyyaml>=6.0
15
16
  Provides-Extra: dev
16
17
  Requires-Dist: pytest>=7.0; extra == "dev"
17
- Requires-Dist: ansicolors; extra == "dev"
18
18
  Dynamic: license-file
19
19
 
20
20
  # LHTML — Lightweight HTML
21
21
 
22
- [![Tests](https://github.com/drohmer/lhtml/actions/workflows/tests.yml/badge.svg?branch=feature/lhtml-v2)](https://github.com/drohmer/lhtml/actions/workflows/tests.yml)
22
+ [![Tests](https://github.com/drohmer/lhtml/actions/workflows/tests.yml/badge.svg?branch=main)](https://github.com/drohmer/lhtml/actions/workflows/tests.yml)
23
23
  [![PyPI](https://img.shields.io/pypi/v/lhtml-markup)](https://pypi.org/project/lhtml-markup/)
24
24
 
25
25
  LHTML is a markup language that simplifies HTML authoring with embedded CSS styling. It is designed to be **HTML-first**: raw HTML passes through untouched, and only a few shorthand symbols (`::`, `*`, `=`, `**`, `__`) trigger conversions.
@@ -65,7 +65,7 @@ lhtml a.l.html b.l.html -o build/ # Several files into a directory
65
65
  python -m lhtml input.l.html # Alternative invocation
66
66
  ```
67
67
 
68
- Files are read and written as UTF-8 (a BOM is accepted). Errors and warnings are reported on stderr with the file name, the remaining files are still processed, and the exit code is non-zero if any file failed. An input file is never overwritten (e.g. `lhtml page.html` without `-o`). Includes are looked up in the input file's directory first, then in the current directory.
68
+ Files are read and written as UTF-8 (a BOM is accepted). Errors and warnings are reported on stderr with the file name, the remaining files are still processed, and the exit code is non-zero if any file failed. No input file in the batch is overwritten (including through symbolic or hard links) (e.g. `lhtml page.html` without `-o`). Includes are looked up in the input file's directory first, then in the current directory.
69
69
 
70
70
  ### Python API
71
71
 
@@ -222,7 +222,7 @@ Output:
222
222
 
223
223
  ### Links
224
224
 
225
- The URL of `link::`, `img::`, `video::` and `videoplay::` is never modified (`__`, `**`, `$` are kept). Parentheses that are part of the URL are kept (`Mercury_(planet)`, `fig(1).png`), while a group starting with `.` or `#` is a class/id group.
225
+ The URL of `link::`, `img::`, `video::` and `videoplay::` is never modified (`__`, `**`, `$` are kept). Parentheses that are part of the URL are kept (`Mercury_(planet)`, `fig(1).png`), while a group starting with `.` or `#` is a class/id group. The URL may contain Jinja expressions (`img::{{ base }}/photo.jpg`, `link::{{ url_for('page') }}[Home]`), which are kept unchanged.
226
226
 
227
227
  ```
228
228
  link::https://example.com[Click here]
@@ -429,7 +429,9 @@ All keys for the `meta` dict passed to `lhtml.run()`:
429
429
 
430
430
  ## Design Principles
431
431
 
432
- - **HTML-first**: Raw HTML is never modified. Only LHTML syntax triggers conversions. In particular, the following are never transformed (not even by `include::` or `::#`): HTML tags and their attributes (URLs containing `__`, quoted values containing `>`, ...), `<script>` and `<style>` blocks (CSS `::before`, ...), HTML comments, and math (`$...$`, `$$...$$`, `\(...\)`, `\[...\]`, so that `$x**2$` reaches MathJax/KaTeX intact). Text between HTML tags is still processed.
432
+ - **HTML-first**: Raw HTML is never modified. Only LHTML syntax triggers conversions. In particular, the following are never transformed (not even by `include::` or `::#`): HTML tags and their attributes, including tags spanning multiple lines (URLs containing `__`, quoted values containing `>`, ...), `<script>` and `<style>` blocks (CSS `::before`, ...), HTML comments, and math (`$...$`, `$$...$$`, `\(...\)`, `\[...\]`, so that `$x**2$` reaches MathJax/KaTeX intact). Text between HTML tags is still processed.
433
+ - Styles, classes/IDs and HTML attributes in LHTML tag groups are preserved without inline formatting. Link labels still support formatting.
434
+ - Jinja2 expressions (`{{ ... }}`), statements (`{% ... %}`) and comments (`{# ... #}`) pass through unchanged.
433
435
  - **Island grammar**: LHTML syntax "islands" float in a sea of opaque content (HTML, Jinja2 templates, LaTeX, etc.) that passes through untouched.
434
436
  - **Minimal**: A few symbols (`::`, `=`, `*`, `**`, `__`, `` ` ``) cover most needs. No complex configuration required.
435
437
  - **Composable**: LHTML works seamlessly with Jinja2 templates, making it suitable for static site generators.
@@ -4,4 +4,3 @@ pyyaml>=6.0
4
4
 
5
5
  [dev]
6
6
  pytest>=7.0
7
- ansicolors
@@ -11,6 +11,7 @@ Tests cover:
11
11
  import os
12
12
  import re
13
13
  import glob
14
+ import time
14
15
  import warnings
15
16
 
16
17
  import pytest
@@ -1118,3 +1119,214 @@ class TestLineBreaks:
1118
1119
  for flag in ('-b', '--line-breaks'):
1119
1120
  code, out, _ = TestCli._run_cli(self, monkeypatch, capsys, flag, str(f))
1120
1121
  assert code == 0 and out == 'a<br>\nb\n'
1122
+
1123
+
1124
+ class TestAuditRegressions:
1125
+ @pytest.mark.parametrize('source', [
1126
+ '<script>const s = "code::[]include::missing.txt code::[-]";</script>',
1127
+ '<style>p::before {content:"verbatim::[]**x**verbatim::[-]"}</style>',
1128
+ '<!-- code::[]include::missing.txt code::[-] -->',
1129
+ '<a title="code::[]include::missing.txt code::[-]">**label**</a>',
1130
+ '$code::[]include::missing.txt code::[-]$',
1131
+ ])
1132
+ def test_outer_zone_owns_block_directives(self, source):
1133
+ assert lhtml.run(source) == source.replace('**label**', '<strong>label</strong>')
1134
+
1135
+ def test_inline_code_owns_block_directive_after_text(self):
1136
+ source = '`example code::[]include::missing.txt code::[-]`'
1137
+ assert lhtml.run(source) == '<code class="code-inline">' + source[1:-1] + '</code>'
1138
+
1139
+ @pytest.mark.parametrize('source', [
1140
+ '<a\n href="page__draft__.html">**label**</a>',
1141
+ '<a title="first\n**second**"\n href="x">**label**</a>',
1142
+ '<input\n disabled\n data-path="__draft__">',
1143
+ ])
1144
+ def test_multiline_html_attributes(self, source):
1145
+ assert lhtml.run(source) == source.replace('**label**', '<strong>label</strong>')
1146
+
1147
+ @pytest.mark.parametrize('source', [
1148
+ '{{ "page__draft__.html" }}',
1149
+ '{{ user.__class__.__name__ }}',
1150
+ '{% set path = "page__draft__.html" %}',
1151
+ '{# code::[]include::missing.txt code::[-] #}',
1152
+ '{{ "}} **literal**" }}',
1153
+ '{% set text = "%} __literal__" %}',
1154
+ '{{\n "__literal__"\n }}',
1155
+ ])
1156
+ def test_jinja_preserved_with_surrounding_formatting(self, source):
1157
+ assert lhtml.run('**before** ' + source + ' __after__') == (
1158
+ '<strong>before</strong> ' + source + ' <em>after</em>')
1159
+
1160
+ @pytest.mark.parametrize('opening', ['{{', '{%'])
1161
+ def test_unclosed_jinja_with_quotes_is_fast(self, opening):
1162
+ source = opening + ' ' + 'il dit "oui" puis "non ' * 200 + '**b**'
1163
+ start = time.perf_counter()
1164
+ result = lhtml.run(source)
1165
+ assert time.perf_counter() - start < 1
1166
+ assert result.endswith('<strong>b</strong>')
1167
+
1168
+ def test_styles_classes_and_inline_attributes_are_not_formatted(self):
1169
+ source = ('div::(.my__class__)[background:url(photo__small__x.png)]'
1170
+ '{data-path="a__b__"} **content** ::')
1171
+ assert lhtml.run(source) == (
1172
+ '<div class="my__class__" style="background:url(photo__small__x.png)"'
1173
+ ' data-path="a__b__"> <strong>content</strong> </div>')
1174
+
1175
+ def test_link_label_keeps_formatting_and_inline_code(self):
1176
+ assert lhtml.run('link::page__draft__.html(.my__class__)[**bold** `code`]') == (
1177
+ '<a class="my__class__" href="page__draft__.html">'
1178
+ '<strong>bold</strong> <code class="code-inline">code</code></a>')
1179
+
1180
+ def test_attribute_quotes_are_escaped_after_restoring(self):
1181
+ assert lhtml.run('div::[font-family:"__font__"] x ::') == (
1182
+ '<div style="font-family:&quot;__font__&quot;"> x </div>')
1183
+
1184
+ def test_protection_applies_in_includes(self, tmp_path):
1185
+ source = '<a\n href="__draft__.html">{{ "__value__" }}</a>'
1186
+ (tmp_path / 'part.html').write_text(source)
1187
+ assert lhtml.run('include::part.html', {'directory_include': [tmp_path]}) == source
1188
+
1189
+
1190
+ class TestCliSourceProtection:
1191
+ _run_cli = TestCli._run_cli
1192
+
1193
+ def test_batch_cannot_overwrite_another_input(self, tmp_path, monkeypatch, capsys):
1194
+ first, second, third = [tmp_path / name for name in
1195
+ ('page.l.html', 'page.html', 'other.l.html')]
1196
+ first.write_text('= First\n')
1197
+ second.write_text('Original content\n')
1198
+ third.write_text('= Other\n')
1199
+ code, _, err = self._run_cli(monkeypatch, capsys, str(first), str(second), str(third))
1200
+ assert code == 1 and 'overwrite' in err
1201
+ assert first.read_text() == '= First\n'
1202
+ assert second.read_text() == 'Original content\n'
1203
+ assert '<h1>Other</h1>' in (tmp_path / 'other.html').read_text()
1204
+
1205
+ @pytest.mark.parametrize('alias_kind', ['symlink', 'hardlink'])
1206
+ def test_output_alias_cannot_overwrite_source(self, tmp_path, monkeypatch, capsys, alias_kind):
1207
+ source, destination = tmp_path / 'page.l.html', tmp_path / 'output.html'
1208
+ source.write_text('= Original\n')
1209
+ if alias_kind == 'symlink':
1210
+ destination.symlink_to(source)
1211
+ else:
1212
+ os.link(source, destination)
1213
+ code, _, err = self._run_cli(monkeypatch, capsys, str(source), '-o', str(destination))
1214
+ assert code == 1 and 'overwrite' in err
1215
+ assert source.read_text() == '= Original\n'
1216
+
1217
+
1218
+ @pytest.mark.parametrize('statement', ['{% if visible %}', '{# comment #}'])
1219
+ def test_jinja_statement_does_not_introduce_line_breaks(statement):
1220
+ source = 'before\n' + statement + '\nafter'
1221
+ assert lhtml.run(source, {'line-breaks': True}) == source
1222
+
1223
+
1224
+ class TestLineBreakRegressions:
1225
+ ON = {'line-breaks': True}
1226
+
1227
+ @pytest.mark.parametrize('tag', ['pre', 'textarea'])
1228
+ @pytest.mark.parametrize('wrapper', [
1229
+ '<script>const s = "<{tag}>";</script>',
1230
+ '<!-- <{tag}> -->',
1231
+ '<span title="<{tag}>">label</span>',
1232
+ '{{ "<{tag}>" }}',
1233
+ ])
1234
+ def test_fake_preformatted_tag_does_not_disable_following_breaks(self, tag, wrapper):
1235
+ prefix = wrapper.replace('{tag}', tag)
1236
+ result = lhtml.run(prefix + '\none\ntwo', self.ON)
1237
+ assert result.startswith(prefix)
1238
+ assert result.endswith('one<br>\ntwo')
1239
+
1240
+ @pytest.mark.parametrize('tag', ['pre-view', 'textarea-widget'])
1241
+ def test_custom_element_is_not_preformatted(self, tag):
1242
+ source = f'<{tag}>one\ntwo</{tag}>\nthree\nfour'
1243
+ assert lhtml.run(source, self.ON) == (
1244
+ f'<{tag}>one<br>\ntwo</{tag}><br>\nthree<br>\nfour')
1245
+
1246
+ @pytest.mark.parametrize('tag', ['pre', 'textarea', 'PRE', 'TEXTAREA'])
1247
+ def test_real_preformatted_block_still_preserves_newlines(self, tag):
1248
+ source = f'<{tag}\n class="sample">\none\ntwo\n</{tag}>\nthree\nfour'
1249
+ assert lhtml.run(source, self.ON) == source.replace('three\nfour', 'three<br>\nfour')
1250
+
1251
+ def test_fake_close_in_comment_does_not_end_preformatted_block(self):
1252
+ source = '<pre>\n<!-- </pre> -->\none\ntwo\n</pre>\nthree\nfour'
1253
+ assert lhtml.run(source, self.ON) == source.replace('three\nfour', 'three<br>\nfour')
1254
+
1255
+ @pytest.mark.parametrize('source, expected', [
1256
+ ('one ::# note\ntwo', 'one <br>\ntwo'),
1257
+ ('one ::# note\n\ntwo', 'one <br>\n<br>\ntwo'),
1258
+ ('one ::# note', 'one '),
1259
+ ('one\n::# note\ntwo', 'one<br>\n<br>\ntwo'),
1260
+ ])
1261
+ def test_comment_removal_keeps_only_existing_newlines(self, source, expected):
1262
+ assert lhtml.run(source, self.ON) == expected
1263
+
1264
+ @pytest.mark.parametrize('statement', [
1265
+ '{%\n if visible\n%}',
1266
+ '{#\n a comment\n#}',
1267
+ '{%-\n set value = "text"\n-%}',
1268
+ ])
1269
+ def test_multiline_jinja_statement_is_structure(self, statement):
1270
+ source = 'one\ntwo\n' + statement + '\nthree\nfour'
1271
+ expected = 'one<br>\ntwo\n' + statement + '\nthree<br>\nfour'
1272
+ assert lhtml.run(source, self.ON) == expected
1273
+
1274
+ def test_multiline_jinja_expression_remains_inline(self):
1275
+ source = 'one {{\n value\n}}\ntwo'
1276
+ assert lhtml.run(source, self.ON) == 'one {{\n value\n}}<br>\ntwo'
1277
+
1278
+
1279
+ class TestJinjaInUrlsAndHeadings:
1280
+ @pytest.mark.parametrize('source, expected', [
1281
+ ('img::{{ url }}', '<img src="{{ url }}" alt="{{ url }}">'),
1282
+ ('img::{{ base }}/photo__small__.jpg[width:10px]',
1283
+ '<img style="width:10px" src="{{ base }}/photo__small__.jpg"'
1284
+ ' alt="{{ base }}/photo__small__.jpg">'),
1285
+ ('link::{{ url }}[**go**]', '<a href="{{ url }}"><strong>go</strong></a>'),
1286
+ ('link::/posts/{{ post.slug }}.html(.nav)[x]',
1287
+ '<a class="nav" href="/posts/{{ post.slug }}.html">x</a>'),
1288
+ ("link::{{ url_for('page', name='a b') }}[x]",
1289
+ "<a href=\"{{ url_for('page', name='a b') }}\">x</a>"),
1290
+ ])
1291
+ def test_jinja_expression_in_url(self, source, expected):
1292
+ assert lhtml.run(source) == expected
1293
+
1294
+ def test_jinja_double_quotes_are_not_escaped_in_attributes(self):
1295
+ assert lhtml.run('link::{{ url_for("page") }}[x]') == (
1296
+ '<a href="{{ url_for("page") }}">x</a>')
1297
+ assert lhtml.run('div::[font-family:{{ font("a") }}; content:"x"] y ::') == (
1298
+ '<div style="font-family:{{ font("a") }}; content:&quot;x&quot;"> y </div>')
1299
+
1300
+ def test_jinja_in_video_url(self):
1301
+ assert '<source src="{{ base }}/clip.mp4"' in lhtml.run('video::{{ base }}/clip.mp4')
1302
+
1303
+ @pytest.mark.parametrize('source, expected', [
1304
+ ('=(.t__x__) Title', '<h1 class="t__x__">Title</h1>\n'),
1305
+ ('==(.a**b** #id__1__) **Bold** __it__',
1306
+ '<h2 class="a**b**" id="id__1__"><strong>Bold</strong> <em>it</em></h2>\n'),
1307
+ ('= Title (.not__class__)', '<h1>Title (.not<em>class</em>)</h1>\n'),
1308
+ ])
1309
+ def test_heading_classes_are_not_formatted(self, source, expected):
1310
+ assert lhtml.run(source) == expected
1311
+
1312
+ @pytest.mark.parametrize('opening', ['{{', '{%'])
1313
+ def test_many_unclosed_jinja_delimiters_are_fast(self, opening):
1314
+ source = (opening + ' x ') * 20000 + '**b**'
1315
+ start = time.perf_counter()
1316
+ result = lhtml.run(source)
1317
+ assert time.perf_counter() - start < 1
1318
+ assert result.endswith('<strong>b</strong>')
1319
+
1320
+ def test_jinja_delimiter_in_string_still_matches(self):
1321
+ assert lhtml.run('{{ "{{" }} __a__') == '{{ "{{" }} <em>a</em>'
1322
+ assert lhtml.run('{% set x = "{%" %} __a__') == '{% set x = "{%" %} <em>a</em>'
1323
+
1324
+
1325
+ @pytest.mark.parametrize('source, expected', [
1326
+ ('videoplay::v.mp4[width:400px;]', '<video autoplay loop muted style="width:400px;">'),
1327
+ ('videoplay::v.mp4(.c)', '<video autoplay loop muted class="c">'),
1328
+ ('videoplay::v.mp4', '<video autoplay loop muted>'),
1329
+ ('video::v.mp4[width:400px;]', '<video style="width:400px;">'),
1330
+ ])
1331
+ def test_video_opening_tag_spacing(source, expected):
1332
+ assert lhtml.run(source).startswith(expected + '\n')
File without changes
File without changes