lhtml-markup 2.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- lhtml/__init__.py +97 -0
- lhtml/__main__.py +5 -0
- lhtml/ast_nodes.py +124 -0
- lhtml/cli.py +62 -0
- lhtml/code.py +74 -0
- lhtml/element_extract.py +21 -0
- lhtml/errors.py +58 -0
- lhtml/export_html.py +147 -0
- lhtml/insert_in_text.py +7 -0
- lhtml/listing.py +54 -0
- lhtml/patterns.py +92 -0
- lhtml/pipeline.py +215 -0
- lhtml/process.py +236 -0
- lhtml/tag_element.lark +30 -0
- lhtml/tag_parser.py +214 -0
- lhtml/wrap_html.py +50 -0
- lhtml_markup-2.0.0.dist-info/METADATA +402 -0
- lhtml_markup-2.0.0.dist-info/RECORD +22 -0
- lhtml_markup-2.0.0.dist-info/WHEEL +5 -0
- lhtml_markup-2.0.0.dist-info/entry_points.txt +2 -0
- lhtml_markup-2.0.0.dist-info/licenses/LICENSE.md +21 -0
- lhtml_markup-2.0.0.dist-info/top_level.txt +1 -0
lhtml/tag_parser.py
ADDED
|
@@ -0,0 +1,214 @@
|
|
|
1
|
+
"""Lark-based parser for the LHTML :: tag element subsystem.
|
|
2
|
+
|
|
3
|
+
The grammar is defined in tag_element.lark and handles the syntax
|
|
4
|
+
after :: in constructs like:
|
|
5
|
+
|
|
6
|
+
tag::(.class #id)[style]{attrs}text::
|
|
7
|
+
div::[color:red;]
|
|
8
|
+
link::url[link text]
|
|
9
|
+
::(.class)[style]
|
|
10
|
+
::nl
|
|
11
|
+
:: (closing tag)
|
|
12
|
+
|
|
13
|
+
Supports nested brackets: [outer[inner]still_outer]
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
import os
|
|
17
|
+
|
|
18
|
+
from lark import Lark, Transformer, UnexpectedInput
|
|
19
|
+
|
|
20
|
+
from .errors import LHTMLParseError
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
# ---------------------------------------------------------------------------
|
|
24
|
+
# Load grammar from .lark file
|
|
25
|
+
# ---------------------------------------------------------------------------
|
|
26
|
+
|
|
27
|
+
_GRAMMAR_PATH = os.path.join(os.path.dirname(__file__), 'tag_element.lark')
|
|
28
|
+
|
|
29
|
+
with open(_GRAMMAR_PATH) as _f:
|
|
30
|
+
_GRAMMAR_TEXT = _f.read()
|
|
31
|
+
|
|
32
|
+
_parser = Lark(_GRAMMAR_TEXT, parser='earley', ambiguity='resolve')
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
# ---------------------------------------------------------------------------
|
|
36
|
+
# Transformer: parse tree -> element dict
|
|
37
|
+
# ---------------------------------------------------------------------------
|
|
38
|
+
|
|
39
|
+
class _TagElementTransformer(Transformer):
|
|
40
|
+
"""Transforms Lark parse tree into element dict."""
|
|
41
|
+
|
|
42
|
+
def _flatten_content(self, items):
|
|
43
|
+
"""Recursively flatten nested bracket content to a string."""
|
|
44
|
+
parts = []
|
|
45
|
+
for item in items:
|
|
46
|
+
if hasattr(item, 'data'):
|
|
47
|
+
name = item.data
|
|
48
|
+
inner = self._flatten_content(item.children)
|
|
49
|
+
if 'paren' in name:
|
|
50
|
+
parts.append(f'({inner})')
|
|
51
|
+
elif 'square' in name:
|
|
52
|
+
parts.append(f'[{inner}]')
|
|
53
|
+
elif 'curly' in name:
|
|
54
|
+
parts.append('{' + inner + '}')
|
|
55
|
+
else:
|
|
56
|
+
parts.append(str(item))
|
|
57
|
+
return ''.join(parts)
|
|
58
|
+
|
|
59
|
+
def paren_content(self, items):
|
|
60
|
+
return self._flatten_content(items)
|
|
61
|
+
|
|
62
|
+
def square_content(self, items):
|
|
63
|
+
return self._flatten_content(items)
|
|
64
|
+
|
|
65
|
+
def curly_content(self, items):
|
|
66
|
+
return self._flatten_content(items)
|
|
67
|
+
|
|
68
|
+
def paren_group(self, items):
|
|
69
|
+
return ('()', items[0] if items else '')
|
|
70
|
+
|
|
71
|
+
def square_group(self, items):
|
|
72
|
+
return ('[]', items[0] if items else '')
|
|
73
|
+
|
|
74
|
+
def curly_group(self, items):
|
|
75
|
+
return ('{}', items[0] if items else '')
|
|
76
|
+
|
|
77
|
+
def bare_text(self, items):
|
|
78
|
+
return ('text', str(items[0]))
|
|
79
|
+
|
|
80
|
+
def bracket_and_text(self, items):
|
|
81
|
+
return items[0]
|
|
82
|
+
|
|
83
|
+
def start(self, items):
|
|
84
|
+
return list(items)
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
_transformer = _TagElementTransformer()
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
# ---------------------------------------------------------------------------
|
|
91
|
+
# Forward scanner (bracket-aware)
|
|
92
|
+
# ---------------------------------------------------------------------------
|
|
93
|
+
|
|
94
|
+
def _scan_forward_bracket_aware(text, start):
|
|
95
|
+
"""Scan forward from start, respecting bracket nesting.
|
|
96
|
+
|
|
97
|
+
Stops at space/newline outside brackets, or if text appears after
|
|
98
|
+
a bracket group that was preceded by text.
|
|
99
|
+
"""
|
|
100
|
+
idx = start
|
|
101
|
+
length = len(text)
|
|
102
|
+
bracket_pairs = {'(': ')', '[': ']', '{': '}'}
|
|
103
|
+
seen_text = False
|
|
104
|
+
seen_bracket_after_text = False
|
|
105
|
+
|
|
106
|
+
while idx < length:
|
|
107
|
+
ch = text[idx]
|
|
108
|
+
if ch == ' ' or ch == '\n':
|
|
109
|
+
break
|
|
110
|
+
if ch in bracket_pairs:
|
|
111
|
+
close = bracket_pairs[ch]
|
|
112
|
+
depth = 1
|
|
113
|
+
bracket_start = idx
|
|
114
|
+
idx += 1
|
|
115
|
+
while idx < length and depth > 0:
|
|
116
|
+
if text[idx] == ch:
|
|
117
|
+
depth += 1
|
|
118
|
+
elif text[idx] == close:
|
|
119
|
+
depth -= 1
|
|
120
|
+
idx += 1
|
|
121
|
+
if depth > 0:
|
|
122
|
+
raise LHTMLParseError(
|
|
123
|
+
f'Unclosed bracket {ch!r} (expected {close!r})',
|
|
124
|
+
source_pos=bracket_start,
|
|
125
|
+
)
|
|
126
|
+
if seen_text:
|
|
127
|
+
seen_bracket_after_text = True
|
|
128
|
+
continue
|
|
129
|
+
if seen_bracket_after_text:
|
|
130
|
+
break
|
|
131
|
+
seen_text = True
|
|
132
|
+
idx += 1
|
|
133
|
+
|
|
134
|
+
return text[start:idx]
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
# ---------------------------------------------------------------------------
|
|
138
|
+
# Public API
|
|
139
|
+
# ---------------------------------------------------------------------------
|
|
140
|
+
|
|
141
|
+
def parse_tag_after_colons(text_after_colons):
|
|
142
|
+
"""Parse the text that appears after :: in an LHTML tag element.
|
|
143
|
+
|
|
144
|
+
Returns dict with keys '[]', '()', '{}', 'text'.
|
|
145
|
+
"""
|
|
146
|
+
result = {'[]': '', '{}': '', '()': '', 'text': ''}
|
|
147
|
+
if not text_after_colons:
|
|
148
|
+
return result
|
|
149
|
+
|
|
150
|
+
try:
|
|
151
|
+
tree = _parser.parse(text_after_colons)
|
|
152
|
+
items = _transformer.transform(tree)
|
|
153
|
+
except UnexpectedInput:
|
|
154
|
+
result['text'] = text_after_colons
|
|
155
|
+
return result
|
|
156
|
+
|
|
157
|
+
text_parts = []
|
|
158
|
+
text_finished = False
|
|
159
|
+
|
|
160
|
+
for kind, content in items:
|
|
161
|
+
if kind == 'text':
|
|
162
|
+
if not text_finished:
|
|
163
|
+
text_parts.append(content)
|
|
164
|
+
else:
|
|
165
|
+
if text_parts:
|
|
166
|
+
text_finished = True
|
|
167
|
+
if result[kind]:
|
|
168
|
+
result[kind] += ' '
|
|
169
|
+
result[kind] += content
|
|
170
|
+
|
|
171
|
+
result['text'] = ''.join(text_parts)
|
|
172
|
+
return result
|
|
173
|
+
|
|
174
|
+
|
|
175
|
+
def extract_bracket_elements_lark(text, index_start):
|
|
176
|
+
"""Extract bracket elements from an LHTML :: tag expression.
|
|
177
|
+
|
|
178
|
+
Args:
|
|
179
|
+
text: Full document text.
|
|
180
|
+
index_start: Position right after :: (first char to parse).
|
|
181
|
+
|
|
182
|
+
Returns:
|
|
183
|
+
Dict with keys: '[]', '()', '{}', 'text', 'tag',
|
|
184
|
+
'index_start', 'index_end'
|
|
185
|
+
"""
|
|
186
|
+
elements = {
|
|
187
|
+
'[]': '', '{}': '', '()': '', 'text': '',
|
|
188
|
+
'tag': '', 'index_start': index_start, 'index_end': index_start,
|
|
189
|
+
}
|
|
190
|
+
|
|
191
|
+
# Backward scan for tag name
|
|
192
|
+
if index_start >= 2:
|
|
193
|
+
idx = index_start - 2
|
|
194
|
+
while idx > 0 and text[idx] != '\n' and text[idx] != ' ' and (text[idx].isalpha() or text[idx] == ':'):
|
|
195
|
+
idx -= 1
|
|
196
|
+
if idx == 0 and (text[idx].isalpha() or text[idx] == ':'):
|
|
197
|
+
tag_start = 0
|
|
198
|
+
elif text[idx] == '\n' or text[idx] == ' ' or not (text[idx].isalpha() or text[idx] == ':'):
|
|
199
|
+
tag_start = idx + 1
|
|
200
|
+
else:
|
|
201
|
+
tag_start = idx
|
|
202
|
+
elements['tag'] = text[tag_start:index_start - 2]
|
|
203
|
+
elements['index_start'] = tag_start
|
|
204
|
+
|
|
205
|
+
# Forward scan: bracket-aware extraction
|
|
206
|
+
if index_start < len(text):
|
|
207
|
+
after_colons = _scan_forward_bracket_aware(text, index_start)
|
|
208
|
+
elements['index_end'] = index_start + len(after_colons)
|
|
209
|
+
|
|
210
|
+
if after_colons:
|
|
211
|
+
parsed = parse_tag_after_colons(after_colons)
|
|
212
|
+
elements.update({k: parsed[k] for k in ('[]', '()', '{}', 'text')})
|
|
213
|
+
|
|
214
|
+
return elements
|
lhtml/wrap_html.py
ADDED
|
@@ -0,0 +1,50 @@
|
|
|
1
|
+
"""HTML document wrapping for LHTML.
|
|
2
|
+
|
|
3
|
+
Wraps LHTML output in a complete HTML5 document structure
|
|
4
|
+
with head, meta, CSS/JS includes from the YAML metadata.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
HTML_TEMPLATE = """\
|
|
8
|
+
<!DOCTYPE html>
|
|
9
|
+
|
|
10
|
+
<html lang="en">
|
|
11
|
+
|
|
12
|
+
<head>
|
|
13
|
+
\t<meta charset="utf-8">
|
|
14
|
+
\t<meta name="viewport" content="width=device-width, initial-scale=1">
|
|
15
|
+
\t<title>{title}</title>
|
|
16
|
+
{head_extras}\
|
|
17
|
+
</head>
|
|
18
|
+
|
|
19
|
+
<body>
|
|
20
|
+
{content}
|
|
21
|
+
</body>
|
|
22
|
+
|
|
23
|
+
</html>
|
|
24
|
+
"""
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def _ensure_list(value):
|
|
28
|
+
"""Normalize a string-or-list meta value to a list."""
|
|
29
|
+
if isinstance(value, str):
|
|
30
|
+
return [value]
|
|
31
|
+
if isinstance(value, list):
|
|
32
|
+
return value
|
|
33
|
+
return []
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def wrap_auto(html_in, meta):
|
|
37
|
+
"""Wrap content in a full HTML5 document using meta configuration."""
|
|
38
|
+
head_parts = []
|
|
39
|
+
|
|
40
|
+
for css in _ensure_list(meta.get('css', [])):
|
|
41
|
+
head_parts.append(f'\t<link rel="stylesheet" type="text/css" href="{css}">\n')
|
|
42
|
+
|
|
43
|
+
for js in _ensure_list(meta.get('js', [])):
|
|
44
|
+
head_parts.append(f'\t<script src="{js}" defer></script>\n')
|
|
45
|
+
|
|
46
|
+
return HTML_TEMPLATE.format(
|
|
47
|
+
title=meta.get('title', 'Webpage'),
|
|
48
|
+
head_extras=''.join(head_parts),
|
|
49
|
+
content=html_in,
|
|
50
|
+
)
|
|
@@ -0,0 +1,402 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: lhtml-markup
|
|
3
|
+
Version: 2.0.0
|
|
4
|
+
Summary: Lightweight HTML markup language — simplifies HTML authoring with shorthand syntax
|
|
5
|
+
License: MIT
|
|
6
|
+
Requires-Python: >=3.10
|
|
7
|
+
Description-Content-Type: text/markdown
|
|
8
|
+
License-File: LICENSE.md
|
|
9
|
+
Requires-Dist: lark>=1.1
|
|
10
|
+
Requires-Dist: pygments>=2.15
|
|
11
|
+
Requires-Dist: pyyaml>=6.0
|
|
12
|
+
Provides-Extra: dev
|
|
13
|
+
Requires-Dist: pytest>=7.0; extra == "dev"
|
|
14
|
+
Requires-Dist: ansicolors; extra == "dev"
|
|
15
|
+
Dynamic: license-file
|
|
16
|
+
|
|
17
|
+
# LHTML — Lightweight HTML
|
|
18
|
+
|
|
19
|
+
LHTML is a markup language that simplifies HTML authoring with embedded CSS styling. It is designed to be **HTML-first**: raw HTML passes through untouched, and only a few shorthand symbols (`::`, `*`, `=`, `**`, `__`) trigger conversions.
|
|
20
|
+
|
|
21
|
+
LHTML is used to build static websites and presentation slides, typically combined with Jinja2 templates.
|
|
22
|
+
|
|
23
|
+
## Installation
|
|
24
|
+
|
|
25
|
+
```bash
|
|
26
|
+
pip install .
|
|
27
|
+
```
|
|
28
|
+
|
|
29
|
+
Or in development mode:
|
|
30
|
+
|
|
31
|
+
```bash
|
|
32
|
+
pip install -e .
|
|
33
|
+
```
|
|
34
|
+
|
|
35
|
+
**Dependencies**: `lark`, `pygments`, `pyyaml` (installed automatically).
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
## Quick Start
|
|
39
|
+
|
|
40
|
+
### Command Line
|
|
41
|
+
|
|
42
|
+
```bash
|
|
43
|
+
lhtml input.l.html # Convert to stdout
|
|
44
|
+
lhtml input.l.html -o output.html # Convert to file
|
|
45
|
+
lhtml input.l.html -w # Wrap in full HTML document
|
|
46
|
+
python -m lhtml input.l.html # Alternative invocation
|
|
47
|
+
```
|
|
48
|
+
|
|
49
|
+
### Python API
|
|
50
|
+
|
|
51
|
+
```python
|
|
52
|
+
import lhtml
|
|
53
|
+
|
|
54
|
+
html = lhtml.run('= Hello World\nSome **bold** text.\n')
|
|
55
|
+
|
|
56
|
+
html = lhtml.run(text, {
|
|
57
|
+
'wrap-auto': True,
|
|
58
|
+
'title': 'My Page',
|
|
59
|
+
'css': ['style.css'],
|
|
60
|
+
'js': ['script.js'],
|
|
61
|
+
})
|
|
62
|
+
```
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
## Syntax Reference
|
|
66
|
+
|
|
67
|
+
### Headings
|
|
68
|
+
|
|
69
|
+
```
|
|
70
|
+
= Main Title
|
|
71
|
+
== Subtitle
|
|
72
|
+
=== Level 3
|
|
73
|
+
```
|
|
74
|
+
|
|
75
|
+
Output:
|
|
76
|
+
```html
|
|
77
|
+
<h1>Main Title</h1>
|
|
78
|
+
<h2>Subtitle</h2>
|
|
79
|
+
<h3>Level 3</h3>
|
|
80
|
+
```
|
|
81
|
+
|
|
82
|
+
With classes/IDs:
|
|
83
|
+
```
|
|
84
|
+
=(.highlight #intro) Styled Title
|
|
85
|
+
```
|
|
86
|
+
```html
|
|
87
|
+
<h1 class="highlight" id="intro">Styled Title</h1>
|
|
88
|
+
```
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
### Lists
|
|
92
|
+
|
|
93
|
+
```
|
|
94
|
+
* First item
|
|
95
|
+
* Second item
|
|
96
|
+
** Nested item A
|
|
97
|
+
** Nested item B
|
|
98
|
+
*** Deep nested
|
|
99
|
+
* Back to top level
|
|
100
|
+
```
|
|
101
|
+
|
|
102
|
+
Produces nested `<ul><li>` structures.
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
### Inline Formatting
|
|
106
|
+
|
|
107
|
+
```
|
|
108
|
+
This is **bold** text.
|
|
109
|
+
This is __italic__ text.
|
|
110
|
+
This is `inline code` text.
|
|
111
|
+
```
|
|
112
|
+
|
|
113
|
+
Output:
|
|
114
|
+
```html
|
|
115
|
+
This is <strong>bold</strong> text.
|
|
116
|
+
This is <em>italic</em> text.
|
|
117
|
+
This is <code class="code-inline">inline code</code> text.
|
|
118
|
+
```
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
### Tag Elements (the `::` system)
|
|
122
|
+
|
|
123
|
+
The core of LHTML. The general syntax is:
|
|
124
|
+
|
|
125
|
+
```
|
|
126
|
+
tagName::(.classes #id)[cssStyle]{htmlAttributes} content ::
|
|
127
|
+
```
|
|
128
|
+
|
|
129
|
+
All bracket groups are optional. If `tagName` is omitted, defaults to `div`.
|
|
130
|
+
|
|
131
|
+
#### Div / Span with Styles
|
|
132
|
+
|
|
133
|
+
```
|
|
134
|
+
div::[color:red; font-size:120%;]
|
|
135
|
+
This text is big and red.
|
|
136
|
+
::
|
|
137
|
+
|
|
138
|
+
span::(.highlight)[font-weight:bold;] inline content ::
|
|
139
|
+
```
|
|
140
|
+
|
|
141
|
+
Output:
|
|
142
|
+
```html
|
|
143
|
+
<div style="color:red; font-size:120%;">
|
|
144
|
+
This text is big and red.
|
|
145
|
+
</div>
|
|
146
|
+
|
|
147
|
+
<span class="highlight" style="font-weight:bold;"> inline content </span>
|
|
148
|
+
```
|
|
149
|
+
|
|
150
|
+
#### Anonymous Div (no tag name)
|
|
151
|
+
|
|
152
|
+
```
|
|
153
|
+
::[padding:10px; background:#eee;]
|
|
154
|
+
Content in a styled div.
|
|
155
|
+
::
|
|
156
|
+
```
|
|
157
|
+
|
|
158
|
+
Output:
|
|
159
|
+
```html
|
|
160
|
+
<div style="padding:10px; background:#eee;">
|
|
161
|
+
Content in a styled div.
|
|
162
|
+
</div>
|
|
163
|
+
```
|
|
164
|
+
|
|
165
|
+
#### Classes, IDs, and Inline Attributes
|
|
166
|
+
|
|
167
|
+
```
|
|
168
|
+
::(.classA .classB #myId)[margin:10px;]{data-role="main"}
|
|
169
|
+
Content
|
|
170
|
+
::
|
|
171
|
+
```
|
|
172
|
+
|
|
173
|
+
Output:
|
|
174
|
+
```html
|
|
175
|
+
<div class="classA classB" id="myId" style="margin:10px;" data-role="main">
|
|
176
|
+
Content
|
|
177
|
+
</div>
|
|
178
|
+
```
|
|
179
|
+
|
|
180
|
+
#### Self-Closing (inline)
|
|
181
|
+
|
|
182
|
+
End the content with `::` on the same line:
|
|
183
|
+
|
|
184
|
+
```
|
|
185
|
+
div::[color:blue;] short text ::
|
|
186
|
+
```
|
|
187
|
+
|
|
188
|
+
Output:
|
|
189
|
+
```html
|
|
190
|
+
<div style="color:blue;"> short text </div>
|
|
191
|
+
```
|
|
192
|
+
|
|
193
|
+
|
|
194
|
+
### Links
|
|
195
|
+
|
|
196
|
+
```
|
|
197
|
+
link::https://example.com[Click here]
|
|
198
|
+
link::page.html(.nav)[Back to home]
|
|
199
|
+
```
|
|
200
|
+
|
|
201
|
+
Output:
|
|
202
|
+
```html
|
|
203
|
+
<a href="https://example.com">Click here</a>
|
|
204
|
+
<a class="nav" href="page.html">Back to home</a>
|
|
205
|
+
```
|
|
206
|
+
|
|
207
|
+
|
|
208
|
+
### Images
|
|
209
|
+
|
|
210
|
+
```
|
|
211
|
+
img::photo.jpg[width:400px;]
|
|
212
|
+
```
|
|
213
|
+
|
|
214
|
+
Output:
|
|
215
|
+
```html
|
|
216
|
+
<img style="width:400px;" src="photo.jpg" alt="photo.jpg">
|
|
217
|
+
```
|
|
218
|
+
|
|
219
|
+
|
|
220
|
+
### Videos
|
|
221
|
+
|
|
222
|
+
```
|
|
223
|
+
video::assets/clip.mp4[width:600px;]
|
|
224
|
+
videoplay::assets/clip.mp4[width:600px;]
|
|
225
|
+
```
|
|
226
|
+
|
|
227
|
+
`videoplay` adds `autoplay loop muted` attributes. The parser automatically detects transcoded codec variants (`-vp9.webm`, `-h265.mp4`, `-h264.mp4`) and poster images (`-poster.jpg`).
|
|
228
|
+
|
|
229
|
+
|
|
230
|
+
### Code Blocks
|
|
231
|
+
|
|
232
|
+
````
|
|
233
|
+
code::[python]
|
|
234
|
+
def hello():
|
|
235
|
+
print("Hello, world!")
|
|
236
|
+
code::[-]
|
|
237
|
+
````
|
|
238
|
+
|
|
239
|
+
Syntax highlighting is powered by Pygments. Any language supported by Pygments can be used.
|
|
240
|
+
|
|
241
|
+
|
|
242
|
+
### Spacer
|
|
243
|
+
|
|
244
|
+
```
|
|
245
|
+
::nl
|
|
246
|
+
```
|
|
247
|
+
|
|
248
|
+
Output:
|
|
249
|
+
```html
|
|
250
|
+
<div style="height:1em;"></div>
|
|
251
|
+
```
|
|
252
|
+
|
|
253
|
+
|
|
254
|
+
### Verbatim (raw passthrough)
|
|
255
|
+
|
|
256
|
+
Content inside verbatim blocks is preserved exactly as-is, with no LHTML processing:
|
|
257
|
+
|
|
258
|
+
```
|
|
259
|
+
verbatim::[]
|
|
260
|
+
This = is not a title
|
|
261
|
+
**not bold** __not italic__
|
|
262
|
+
div::[not a tag]
|
|
263
|
+
verbatim::[-]
|
|
264
|
+
```
|
|
265
|
+
|
|
266
|
+
|
|
267
|
+
### Comments
|
|
268
|
+
|
|
269
|
+
```
|
|
270
|
+
Some text ::# This comment will be removed
|
|
271
|
+
```
|
|
272
|
+
|
|
273
|
+
|
|
274
|
+
### File Inclusion
|
|
275
|
+
|
|
276
|
+
```
|
|
277
|
+
include::header.html
|
|
278
|
+
include::components/nav.html
|
|
279
|
+
```
|
|
280
|
+
|
|
281
|
+
Included files are recursively processed (up to 20 levels).
|
|
282
|
+
|
|
283
|
+
|
|
284
|
+
### YAML Front Matter
|
|
285
|
+
|
|
286
|
+
```
|
|
287
|
+
---
|
|
288
|
+
title: "My Page"
|
|
289
|
+
css: ["style.css", "theme.css"]
|
|
290
|
+
js: "app.js"
|
|
291
|
+
wrap-auto: true
|
|
292
|
+
---
|
|
293
|
+
|
|
294
|
+
= Page content starts here
|
|
295
|
+
```
|
|
296
|
+
|
|
297
|
+
Supported metadata keys:
|
|
298
|
+
|
|
299
|
+
| Key | Type | Description |
|
|
300
|
+
|-----|------|-------------|
|
|
301
|
+
| `title` | string | Page title (used in HTML wrapper) |
|
|
302
|
+
| `css` | string or list | CSS files to include |
|
|
303
|
+
| `js` | string or list | JavaScript files to include |
|
|
304
|
+
| `wrap-auto` | boolean | Wrap output in full HTML document |
|
|
305
|
+
| `directory_include` | list | Directories to search for includes |
|
|
306
|
+
|
|
307
|
+
|
|
308
|
+
## Plugin System
|
|
309
|
+
|
|
310
|
+
### Custom Tag Handlers
|
|
311
|
+
|
|
312
|
+
Register handlers for new `::` tag types:
|
|
313
|
+
|
|
314
|
+
```python
|
|
315
|
+
from lhtml.pipeline import tag_registry
|
|
316
|
+
|
|
317
|
+
def handle_alert(element, tag_to_close, current_directory):
|
|
318
|
+
style = element.get('[]', '')
|
|
319
|
+
text = element.get('text', '')
|
|
320
|
+
return f'<div class="alert" style="{style}">{text}</div>', True
|
|
321
|
+
|
|
322
|
+
tag_registry.register('alert', handle_alert)
|
|
323
|
+
```
|
|
324
|
+
|
|
325
|
+
Then use in LHTML:
|
|
326
|
+
```
|
|
327
|
+
alert::[background:yellow; padding:10px;] Warning message ::
|
|
328
|
+
```
|
|
329
|
+
|
|
330
|
+
### Custom Code Lexers
|
|
331
|
+
|
|
332
|
+
Register custom Pygments lexers for syntax highlighting:
|
|
333
|
+
|
|
334
|
+
```python
|
|
335
|
+
from lhtml.pipeline import lexer_registry
|
|
336
|
+
from pygments.lexers import PythonLexer
|
|
337
|
+
|
|
338
|
+
lexer_registry.register('mypython', PythonLexer)
|
|
339
|
+
```
|
|
340
|
+
|
|
341
|
+
### Custom Pipeline
|
|
342
|
+
|
|
343
|
+
Create an isolated pipeline with its own tag registry:
|
|
344
|
+
|
|
345
|
+
```python
|
|
346
|
+
from lhtml.pipeline import ProcessingPipeline, TagRegistry
|
|
347
|
+
|
|
348
|
+
registry = TagRegistry()
|
|
349
|
+
registry.register('note', my_note_handler)
|
|
350
|
+
|
|
351
|
+
pipeline = ProcessingPipeline(registry=registry)
|
|
352
|
+
html = pipeline.run(text, {'wrap-auto': True})
|
|
353
|
+
```
|
|
354
|
+
|
|
355
|
+
|
|
356
|
+
## Configuration Reference
|
|
357
|
+
|
|
358
|
+
All keys for the `meta` dict passed to `lhtml.run()`:
|
|
359
|
+
|
|
360
|
+
```python
|
|
361
|
+
{
|
|
362
|
+
'wrap-auto': False, # Wrap in HTML document
|
|
363
|
+
'title': 'Webpage', # Document title
|
|
364
|
+
'css': [], # CSS files (string or list)
|
|
365
|
+
'js': [], # JS files (string or list)
|
|
366
|
+
'directory_include': [], # Search paths for include::
|
|
367
|
+
'current_directory': '', # Base directory for video codec detection
|
|
368
|
+
}
|
|
369
|
+
```
|
|
370
|
+
|
|
371
|
+
|
|
372
|
+
## Design Principles
|
|
373
|
+
|
|
374
|
+
- **HTML-first**: Raw HTML is never modified. Only LHTML syntax triggers conversions.
|
|
375
|
+
- **Island grammar**: LHTML syntax "islands" float in a sea of opaque content (HTML, Jinja2 templates, LaTeX, etc.) that passes through untouched.
|
|
376
|
+
- **Minimal**: A few symbols (`::`, `=`, `*`, `**`, `__`, `` ` ``) cover most needs. No complex configuration required.
|
|
377
|
+
- **Composable**: LHTML works seamlessly with Jinja2 templates, making it suitable for static site generators.
|
|
378
|
+
|
|
379
|
+
|
|
380
|
+
## Project Structure
|
|
381
|
+
|
|
382
|
+
```
|
|
383
|
+
src/lhtml/
|
|
384
|
+
__init__.py # Public API: run(), analyse_tag(), read_yaml()
|
|
385
|
+
cli.py # Command-line interface
|
|
386
|
+
pipeline.py # ProcessingPipeline, TagRegistry, LexerRegistry
|
|
387
|
+
process.py # Core transformation functions
|
|
388
|
+
patterns.py # Centralized regex patterns and utilities
|
|
389
|
+
tag_parser.py # Lark-based parser for :: bracket syntax
|
|
390
|
+
tag_element.lark # Lark grammar definition
|
|
391
|
+
export_html.py # HTML generation for tag elements
|
|
392
|
+
listing.py # List processing
|
|
393
|
+
code.py # Code syntax highlighting (Pygments)
|
|
394
|
+
wrap_html.py # HTML document wrapping
|
|
395
|
+
ast_nodes.py # AST node dataclasses
|
|
396
|
+
errors.py # Structured error types
|
|
397
|
+
```
|
|
398
|
+
|
|
399
|
+
|
|
400
|
+
## License
|
|
401
|
+
|
|
402
|
+
MIT
|
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
lhtml/__init__.py,sha256=u7ANa7U-nJiGJz_d9cXOpQOfreHuZY31ao50xvpDE5I,3064
|
|
2
|
+
lhtml/__main__.py,sha256=1R7j15YAM6-cJ2dF6-xg5ix3npRb1McOx2OYa1IoZbk,86
|
|
3
|
+
lhtml/ast_nodes.py,sha256=gEbiEL5D-V1q_k1YCqpb8Ko4i6hj564-Q2QBHnFtJXw,3384
|
|
4
|
+
lhtml/cli.py,sha256=Gl4TMX_r0AsIFFwRr1uh71L_338cYKAU-cLYl4Ir6-g,1579
|
|
5
|
+
lhtml/code.py,sha256=UIdKHJ9HKlwZp4jpb86auK28GD4QP34KZaPRxOeyLyM,2513
|
|
6
|
+
lhtml/element_extract.py,sha256=rN9NPUOaJOYrWopD241Xdwdg5X0ZvxvrGubTes-dHMc,619
|
|
7
|
+
lhtml/errors.py,sha256=fQ7zwzdwywqeUQY2VWnl0ih-feXRa2pnWj7s--bWpyo,1999
|
|
8
|
+
lhtml/export_html.py,sha256=3fRlAT9mKn0fBCstgjlqYU9uJK302wFtjXFWFxzc6dg,4951
|
|
9
|
+
lhtml/insert_in_text.py,sha256=TSg_oTYh-kyHc6HABerx4C9pzuIfdE1IoZXnm5YcYzo,247
|
|
10
|
+
lhtml/listing.py,sha256=CfX6u4uat4WoBnI0GuvSNj_pLWACGdk87SH1Y3lDGW4,1281
|
|
11
|
+
lhtml/patterns.py,sha256=C9ShYSnZZZrHZ3LGI5EzgpOZnE2XDP0lMLGQPVI56Zs,3322
|
|
12
|
+
lhtml/pipeline.py,sha256=_EcA9xSaumAx0IbMN1Dygl6eA1fuXCgCcRj-CZP1mTc,7337
|
|
13
|
+
lhtml/process.py,sha256=uQ16Bw6rfOhV2ywPLy1ReAlGXvlquVmjxWLFrgkBt0s,8293
|
|
14
|
+
lhtml/tag_element.lark,sha256=z6z77vnHzY7Ywo4hm5zrYyS4MiArA9QulDrlIpAPuVY,870
|
|
15
|
+
lhtml/tag_parser.py,sha256=Wz_DGcRD0E8DuoO2nxFp6Xm-rM70NMn5nmMjFgKC6rk,6529
|
|
16
|
+
lhtml/wrap_html.py,sha256=0V9Zr1SvFKUdp1PbrDZ1BtlUcBVjtRtHFujsKwhh8po,1143
|
|
17
|
+
lhtml_markup-2.0.0.dist-info/licenses/LICENSE.md,sha256=vtub9GYnfalS9yrZ2HblcyHqspQ3GS_J4od_zcT1mAE,1103
|
|
18
|
+
lhtml_markup-2.0.0.dist-info/METADATA,sha256=-lmq7oVvSBJpPO9Y4i5OWdtHlEFGoEbpJd7aw_OegiQ,8007
|
|
19
|
+
lhtml_markup-2.0.0.dist-info/WHEEL,sha256=aeYiig01lYGDzBgS8HxWXOg3uV61G9ijOsup-k9o1sk,91
|
|
20
|
+
lhtml_markup-2.0.0.dist-info/entry_points.txt,sha256=Go5qk4IiwAhHwFlaQqH_rJ_wOR8mG97BAt0LeemuK7M,37
|
|
21
|
+
lhtml_markup-2.0.0.dist-info/top_level.txt,sha256=qx9Rgy0hxw3DSS-CtSK-0faRdTHw3fKbw44QD4qRS7s,6
|
|
22
|
+
lhtml_markup-2.0.0.dist-info/RECORD,,
|