lhtml-markup 2.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- lhtml/__init__.py +97 -0
- lhtml/__main__.py +5 -0
- lhtml/ast_nodes.py +124 -0
- lhtml/cli.py +62 -0
- lhtml/code.py +74 -0
- lhtml/element_extract.py +21 -0
- lhtml/errors.py +58 -0
- lhtml/export_html.py +147 -0
- lhtml/insert_in_text.py +7 -0
- lhtml/listing.py +54 -0
- lhtml/patterns.py +92 -0
- lhtml/pipeline.py +215 -0
- lhtml/process.py +236 -0
- lhtml/tag_element.lark +30 -0
- lhtml/tag_parser.py +214 -0
- lhtml/wrap_html.py +50 -0
- lhtml_markup-2.0.0.dist-info/METADATA +402 -0
- lhtml_markup-2.0.0.dist-info/RECORD +22 -0
- lhtml_markup-2.0.0.dist-info/WHEEL +5 -0
- lhtml_markup-2.0.0.dist-info/entry_points.txt +2 -0
- lhtml_markup-2.0.0.dist-info/licenses/LICENSE.md +21 -0
- lhtml_markup-2.0.0.dist-info/top_level.txt +1 -0
lhtml/__init__.py
ADDED
|
@@ -0,0 +1,97 @@
|
|
|
1
|
+
"""LHTML — Lightweight HTML markup language.
|
|
2
|
+
|
|
3
|
+
A simple markup language that extends HTML with shorthand syntax
|
|
4
|
+
for styling, layout, and content formatting.
|
|
5
|
+
|
|
6
|
+
Usage:
|
|
7
|
+
import lhtml
|
|
8
|
+
html = lhtml.run(text)
|
|
9
|
+
html = lhtml.run(text, {'wrap-auto': True, 'title': 'My Page'})
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from .element_extract import extract_bracket_elements
|
|
13
|
+
from .insert_in_text import insert_element_from_index, remove_element_to_index
|
|
14
|
+
from .wrap_html import wrap_auto
|
|
15
|
+
|
|
16
|
+
from .process import (
|
|
17
|
+
process_yaml, process_verbatim_to_index, process_verbatim_back_from_index,
|
|
18
|
+
process_remove_comment, process_include, find_file,
|
|
19
|
+
process_bold, process_italic, process_code_inline,
|
|
20
|
+
process_title, process_tag, process_code,
|
|
21
|
+
process_listing,
|
|
22
|
+
)
|
|
23
|
+
|
|
24
|
+
from .errors import (
|
|
25
|
+
LHTMLError, LHTMLParseError, LHTMLFileNotFound,
|
|
26
|
+
LHTMLTagStackError, LHTMLIncludeLoopError,
|
|
27
|
+
)
|
|
28
|
+
|
|
29
|
+
from .ast_nodes import (
|
|
30
|
+
LHTMLDocument, TextNode, TagElement, HeadingNode,
|
|
31
|
+
ListNode, ListItem, InlineFormat, CodeBlock,
|
|
32
|
+
VerbatimBlock, IncludeDirective, Comment, SpacerNode, ClosingTag,
|
|
33
|
+
)
|
|
34
|
+
|
|
35
|
+
from .pipeline import ProcessingPipeline, tag_registry, lexer_registry
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
# ---------------------------------------------------------------------------
|
|
39
|
+
# Public API (backward-compatible with the old lhtml.py module)
|
|
40
|
+
# ---------------------------------------------------------------------------
|
|
41
|
+
|
|
42
|
+
_pipeline = ProcessingPipeline()
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def run(text, meta_arg=None):
|
|
46
|
+
"""Process LHTML text and return HTML.
|
|
47
|
+
|
|
48
|
+
Args:
|
|
49
|
+
text: LHTML markup string.
|
|
50
|
+
meta_arg: Optional dict overriding default configuration.
|
|
51
|
+
Keys: wrap-auto, title, css, js, directory_include, etc.
|
|
52
|
+
|
|
53
|
+
Returns:
|
|
54
|
+
HTML string.
|
|
55
|
+
"""
|
|
56
|
+
return _pipeline.run(text, meta_arg)
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def analyse_tag(text_in):
|
|
60
|
+
"""Parse a single :: tag expression (utility)."""
|
|
61
|
+
return extract_bracket_elements(text_in, text_in.find('::') + 2)
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def read_yaml(text_in):
|
|
65
|
+
"""Extract YAML front matter from text (utility)."""
|
|
66
|
+
_, yaml_parameter = process_yaml(text_in)
|
|
67
|
+
return yaml_parameter
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def main():
|
|
71
|
+
"""CLI entry point (delegates to cli.main)."""
|
|
72
|
+
from .cli import main as _cli_main
|
|
73
|
+
_cli_main()
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
__all__ = [
|
|
77
|
+
# Core API
|
|
78
|
+
'run', 'analyse_tag', 'read_yaml', 'main',
|
|
79
|
+
# Processing functions
|
|
80
|
+
'extract_bracket_elements',
|
|
81
|
+
'process_yaml', 'process_verbatim_to_index', 'process_verbatim_back_from_index',
|
|
82
|
+
'process_remove_comment', 'process_include', 'find_file',
|
|
83
|
+
'process_bold', 'process_italic', 'process_code_inline',
|
|
84
|
+
'process_title', 'process_tag', 'process_code',
|
|
85
|
+
'process_listing',
|
|
86
|
+
'insert_element_from_index', 'remove_element_to_index',
|
|
87
|
+
'wrap_auto',
|
|
88
|
+
# Pipeline & plugins
|
|
89
|
+
'ProcessingPipeline', 'tag_registry', 'lexer_registry',
|
|
90
|
+
# AST
|
|
91
|
+
'LHTMLDocument', 'TextNode', 'TagElement', 'HeadingNode',
|
|
92
|
+
'ListNode', 'ListItem', 'InlineFormat', 'CodeBlock',
|
|
93
|
+
'VerbatimBlock', 'IncludeDirective', 'Comment', 'SpacerNode', 'ClosingTag',
|
|
94
|
+
# Errors
|
|
95
|
+
'LHTMLError', 'LHTMLParseError', 'LHTMLFileNotFound',
|
|
96
|
+
'LHTMLTagStackError', 'LHTMLIncludeLoopError',
|
|
97
|
+
]
|
lhtml/__main__.py
ADDED
lhtml/ast_nodes.py
ADDED
|
@@ -0,0 +1,124 @@
|
|
|
1
|
+
"""AST node definitions for LHTML.
|
|
2
|
+
|
|
3
|
+
These dataclasses represent the intermediate representation produced
|
|
4
|
+
by the LHTML parser before HTML emission. Each node corresponds to
|
|
5
|
+
an LHTML language construct as defined in grammar.ebnf.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
from dataclasses import dataclass, field
|
|
10
|
+
from typing import Union
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
# Type alias for any LHTML node
|
|
14
|
+
LHTMLNode = Union[
|
|
15
|
+
'TextNode', 'TagElement', 'HeadingNode', 'ListNode',
|
|
16
|
+
'InlineFormat', 'CodeBlock', 'VerbatimBlock',
|
|
17
|
+
'IncludeDirective', 'Comment', 'SpacerNode', 'ClosingTag'
|
|
18
|
+
]
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
@dataclass
|
|
22
|
+
class LHTMLDocument:
|
|
23
|
+
"""Root node representing a full LHTML document."""
|
|
24
|
+
meta: dict = field(default_factory=dict)
|
|
25
|
+
children: list[LHTMLNode] = field(default_factory=list)
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
@dataclass
|
|
29
|
+
class TextNode:
|
|
30
|
+
"""Passthrough text (raw HTML, Jinja2 templates, prose).
|
|
31
|
+
Not transformed by LHTML — emitted as-is."""
|
|
32
|
+
content: str
|
|
33
|
+
source_span: tuple[int, int] = (0, 0)
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
@dataclass
|
|
37
|
+
class TagElement:
|
|
38
|
+
"""An LHTML tag element: tag::(.class #id)[style]{attrs} content ::
|
|
39
|
+
|
|
40
|
+
For special tags (link, img, video), the 'text' field holds the
|
|
41
|
+
URL or file path that appears between :: and the first bracket.
|
|
42
|
+
"""
|
|
43
|
+
tag: str # "div", "span", "link", "img", "video", "videoplay", ""
|
|
44
|
+
classes: list[str] = field(default_factory=list)
|
|
45
|
+
id: str | None = None
|
|
46
|
+
style: str = ''
|
|
47
|
+
inline_attrs: str = ''
|
|
48
|
+
text: str = '' # URL for link, path for img/video, content for self-closing
|
|
49
|
+
self_closing: bool = False # True if content ends with ::
|
|
50
|
+
source_span: tuple[int, int] = (0, 0)
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
@dataclass
|
|
54
|
+
class HeadingNode:
|
|
55
|
+
"""A heading: =() Title, ==() Subtitle, etc."""
|
|
56
|
+
level: int # 1-6
|
|
57
|
+
text: str = ''
|
|
58
|
+
classes: list[str] = field(default_factory=list)
|
|
59
|
+
id: str | None = None
|
|
60
|
+
source_span: tuple[int, int] = (0, 0)
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
@dataclass
|
|
64
|
+
class ListItem:
|
|
65
|
+
"""A single list item with its nesting level."""
|
|
66
|
+
level: int # 1 for *, 2 for **, etc.
|
|
67
|
+
content: str = ''
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
@dataclass
|
|
71
|
+
class ListNode:
|
|
72
|
+
"""A sequence of list items (* item, ** subitem)."""
|
|
73
|
+
items: list[ListItem] = field(default_factory=list)
|
|
74
|
+
source_span: tuple[int, int] = (0, 0)
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
@dataclass
|
|
78
|
+
class InlineFormat:
|
|
79
|
+
"""Inline formatting: **bold**, __italic__, `code`."""
|
|
80
|
+
kind: str # "bold" | "italic" | "code"
|
|
81
|
+
content: str = ''
|
|
82
|
+
source_span: tuple[int, int] = (0, 0)
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
@dataclass
|
|
86
|
+
class CodeBlock:
|
|
87
|
+
"""A code block: code::[language] ... code::[-]."""
|
|
88
|
+
language: str = ''
|
|
89
|
+
content: str = ''
|
|
90
|
+
source_span: tuple[int, int] = (0, 0)
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
@dataclass
|
|
94
|
+
class VerbatimBlock:
|
|
95
|
+
"""A verbatim block: verbatim::[] ... verbatim::[-].
|
|
96
|
+
Content is preserved without any LHTML processing."""
|
|
97
|
+
content: str = ''
|
|
98
|
+
source_span: tuple[int, int] = (0, 0)
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
@dataclass
|
|
102
|
+
class IncludeDirective:
|
|
103
|
+
"""An include directive: include::filename."""
|
|
104
|
+
filename: str = ''
|
|
105
|
+
source_span: tuple[int, int] = (0, 0)
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
@dataclass
|
|
109
|
+
class Comment:
|
|
110
|
+
"""An LHTML comment: ::#comment text."""
|
|
111
|
+
text: str = ''
|
|
112
|
+
source_span: tuple[int, int] = (0, 0)
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
@dataclass
|
|
116
|
+
class SpacerNode:
|
|
117
|
+
"""A spacer element: ::nl."""
|
|
118
|
+
source_span: tuple[int, int] = (0, 0)
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
@dataclass
|
|
122
|
+
class ClosingTag:
|
|
123
|
+
"""A bare closing tag: :: (closes the most recent open tag)."""
|
|
124
|
+
source_span: tuple[int, int] = (0, 0)
|
lhtml/cli.py
ADDED
|
@@ -0,0 +1,62 @@
|
|
|
1
|
+
"""LHTML command-line interface.
|
|
2
|
+
|
|
3
|
+
Usage:
|
|
4
|
+
lhtml [-w] inputFile [-o outputFile]
|
|
5
|
+
python -m lhtml [-w] inputFile [-o outputFile]
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
import os
|
|
9
|
+
import argparse
|
|
10
|
+
|
|
11
|
+
from .pipeline import ProcessingPipeline
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
_pipeline = ProcessingPipeline()
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def _ensure_trailing_newline(text):
|
|
18
|
+
if text and text[-1] != '\n':
|
|
19
|
+
return text + '\n'
|
|
20
|
+
return text
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def main():
|
|
24
|
+
"""CLI entry point."""
|
|
25
|
+
parser = argparse.ArgumentParser(description='Lightweight HTML')
|
|
26
|
+
parser.add_argument('inputFile', help='Input filepath')
|
|
27
|
+
parser.add_argument('-w', '--wrapAuto',
|
|
28
|
+
help='Wrap content in basic HTML template',
|
|
29
|
+
action='store_true')
|
|
30
|
+
parser.add_argument('-o', '--outputFile', help='Output filepath')
|
|
31
|
+
args = parser.parse_args()
|
|
32
|
+
|
|
33
|
+
meta = {}
|
|
34
|
+
if args.wrapAuto:
|
|
35
|
+
meta['wrap-auto'] = True
|
|
36
|
+
|
|
37
|
+
meta['directory_include'] = [os.getcwd() + '/']
|
|
38
|
+
f_in = args.inputFile
|
|
39
|
+
|
|
40
|
+
if os.path.isfile(f_in):
|
|
41
|
+
dir_to_include = os.path.dirname(f_in)
|
|
42
|
+
if dir_to_include:
|
|
43
|
+
meta['directory_include'].append(
|
|
44
|
+
os.path.join(os.getcwd(), dir_to_include) + '/'
|
|
45
|
+
)
|
|
46
|
+
|
|
47
|
+
with open(f_in) as fid:
|
|
48
|
+
txt = _ensure_trailing_newline(fid.read())
|
|
49
|
+
|
|
50
|
+
html = _ensure_trailing_newline(_pipeline.run(txt, meta))
|
|
51
|
+
|
|
52
|
+
if args.outputFile:
|
|
53
|
+
with open(args.outputFile, 'w') as f_out:
|
|
54
|
+
f_out.write(html)
|
|
55
|
+
else:
|
|
56
|
+
print(html)
|
|
57
|
+
else:
|
|
58
|
+
print(f'Error: unrecognized input file [{f_in}]')
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
if __name__ == '__main__':
|
|
62
|
+
main()
|
lhtml/code.py
ADDED
|
@@ -0,0 +1,74 @@
|
|
|
1
|
+
"""Code block syntax highlighting for LHTML.
|
|
2
|
+
|
|
3
|
+
Uses Pygments for highlighting. Custom lexers can be registered
|
|
4
|
+
via the lexer_registry in pipeline.py.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from pygments import highlight
|
|
8
|
+
from pygments.lexers import get_lexer_by_name
|
|
9
|
+
from pygments.formatters import HtmlFormatter
|
|
10
|
+
from pygments.lexer import words, inherit
|
|
11
|
+
import pygments.lexers
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
# ---------------------------------------------------------------------------
|
|
15
|
+
# Built-in custom lexer: C++ with CGP library types
|
|
16
|
+
# ---------------------------------------------------------------------------
|
|
17
|
+
|
|
18
|
+
class CppCgpLexer(pygments.lexers.CppLexer):
|
|
19
|
+
"""Extended C++ lexer with CGP library types and functions."""
|
|
20
|
+
name = 'C++ (CGP)'
|
|
21
|
+
tokens = {
|
|
22
|
+
'statements': [
|
|
23
|
+
(words((
|
|
24
|
+
'vec2', 'vec3', 'vec4', 'mat2', 'mat3', 'mat4',
|
|
25
|
+
'numarray_stack', 'numarray', 'mesh', 'mesh_drawable',
|
|
26
|
+
'rotation_transform', 'affine_rt', 'affine_rts', 'affine',
|
|
27
|
+
'quaternion', 'string', 'ostream',
|
|
28
|
+
), suffix=r'\b'), pygments.token.Keyword.Type),
|
|
29
|
+
(words((
|
|
30
|
+
'dot', 'cross', 'norm', 'normalize',
|
|
31
|
+
'draw', 'draw_wireframe', 'transpose', 'det', 'inverse',
|
|
32
|
+
), suffix=r'\b'), pygments.token.Keyword.Function),
|
|
33
|
+
(r'gl\w*', pygments.token.Keyword.Function),
|
|
34
|
+
(r'const\&', pygments.token.Number),
|
|
35
|
+
(r'std::', pygments.token.Number),
|
|
36
|
+
(r'cgp::', pygments.token.Number),
|
|
37
|
+
inherit,
|
|
38
|
+
]
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
# Default custom lexer mapping
|
|
43
|
+
_BUILTIN_LEXERS = {
|
|
44
|
+
'c++': CppCgpLexer,
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
# ---------------------------------------------------------------------------
|
|
49
|
+
# Public API
|
|
50
|
+
# ---------------------------------------------------------------------------
|
|
51
|
+
|
|
52
|
+
def export_html_code(text, language, cssclass='code'):
|
|
53
|
+
"""Highlight a code block and return HTML."""
|
|
54
|
+
# Check plugin registry first, then built-in lexers
|
|
55
|
+
lexer_class = None
|
|
56
|
+
try:
|
|
57
|
+
from .pipeline import lexer_registry
|
|
58
|
+
lexer_class = lexer_registry.get(language)
|
|
59
|
+
except ImportError:
|
|
60
|
+
pass
|
|
61
|
+
|
|
62
|
+
if lexer_class is None:
|
|
63
|
+
lexer_class = _BUILTIN_LEXERS.get(language)
|
|
64
|
+
|
|
65
|
+
if lexer_class is not None:
|
|
66
|
+
lexer = lexer_class()
|
|
67
|
+
else:
|
|
68
|
+
lexer = get_lexer_by_name(
|
|
69
|
+
language, stripall=False, stripnl=True,
|
|
70
|
+
ensurenl='True', tabsize=2, encoding='utf-8',
|
|
71
|
+
)
|
|
72
|
+
|
|
73
|
+
formatter = HtmlFormatter(linenos=False, cssclass=cssclass)
|
|
74
|
+
return highlight(text, lexer, formatter)
|
lhtml/element_extract.py
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
"""LHTML element extraction from :: tag syntax.
|
|
2
|
+
|
|
3
|
+
Delegates to the Lark-based parser in tag_parser.py for robust
|
|
4
|
+
bracket parsing with proper nesting support.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from .tag_parser import extract_bracket_elements_lark
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
def extract_bracket_elements(text, index_start):
|
|
11
|
+
"""Extract bracket elements from an LHTML :: tag expression.
|
|
12
|
+
|
|
13
|
+
Args:
|
|
14
|
+
text: Full document text.
|
|
15
|
+
index_start: Position right after :: (first char to parse).
|
|
16
|
+
|
|
17
|
+
Returns:
|
|
18
|
+
Dict with keys: '[]', '()', '{}', 'text', 'tag',
|
|
19
|
+
'index_start', 'index_end'
|
|
20
|
+
"""
|
|
21
|
+
return extract_bracket_elements_lark(text, index_start)
|
lhtml/errors.py
ADDED
|
@@ -0,0 +1,58 @@
|
|
|
1
|
+
"""Structured error types for LHTML parsing.
|
|
2
|
+
|
|
3
|
+
These exceptions carry source location information to help
|
|
4
|
+
users diagnose problems in their LHTML markup.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
class LHTMLError(Exception):
|
|
9
|
+
"""Base class for all LHTML errors."""
|
|
10
|
+
|
|
11
|
+
def __init__(self, message: str, source_pos: int = -1, source_line: int = -1):
|
|
12
|
+
self.source_pos = source_pos
|
|
13
|
+
self.source_line = source_line
|
|
14
|
+
if source_line >= 0:
|
|
15
|
+
full_message = f'Line {source_line}: {message}'
|
|
16
|
+
elif source_pos >= 0:
|
|
17
|
+
full_message = f'Position {source_pos}: {message}'
|
|
18
|
+
else:
|
|
19
|
+
full_message = message
|
|
20
|
+
super().__init__(full_message)
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
class LHTMLParseError(LHTMLError):
|
|
24
|
+
"""Error during parsing of LHTML markup (unclosed brackets, etc.)."""
|
|
25
|
+
pass
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
class LHTMLFileNotFound(LHTMLError):
|
|
29
|
+
"""An include:: directive references a file that cannot be found."""
|
|
30
|
+
|
|
31
|
+
def __init__(self, filename: str, directories: list[str], source_pos: int = -1, source_line: int = -1):
|
|
32
|
+
self.filename = filename
|
|
33
|
+
self.directories = directories
|
|
34
|
+
message = f'Could not find file [{filename}] in directories {directories}'
|
|
35
|
+
super().__init__(message, source_pos, source_line)
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
class LHTMLTagStackError(LHTMLError):
|
|
39
|
+
"""A closing :: tag has no matching opening tag."""
|
|
40
|
+
|
|
41
|
+
def __init__(self, source_pos: int = -1, source_line: int = -1):
|
|
42
|
+
message = 'Closing tag :: has no matching opening tag'
|
|
43
|
+
super().__init__(message, source_pos, source_line)
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
class LHTMLIncludeLoopError(LHTMLError):
|
|
47
|
+
"""Too many include iterations — likely a circular include."""
|
|
48
|
+
|
|
49
|
+
def __init__(self, max_iterations: int = 20):
|
|
50
|
+
message = f'Too many include iterations (>{max_iterations}), possible circular include'
|
|
51
|
+
super().__init__(message)
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def pos_to_line(text: str, pos: int) -> int:
|
|
55
|
+
"""Convert a character position to a 1-based line number."""
|
|
56
|
+
if pos < 0 or pos > len(text):
|
|
57
|
+
return -1
|
|
58
|
+
return text[:pos].count('\n') + 1
|
lhtml/export_html.py
ADDED
|
@@ -0,0 +1,147 @@
|
|
|
1
|
+
"""HTML generation functions for LHTML tag elements.
|
|
2
|
+
|
|
3
|
+
Each export_html_* function converts a parsed element dict into
|
|
4
|
+
an HTML string. The element dict has keys: '[]' (style), '()'
|
|
5
|
+
(class/id), '{}' (inline attrs), 'text', 'tag'.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
import os
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
# ---------------------------------------------------------------------------
|
|
12
|
+
# Attribute helpers
|
|
13
|
+
# ---------------------------------------------------------------------------
|
|
14
|
+
|
|
15
|
+
def _build_attrs(elements):
|
|
16
|
+
"""Build the common HTML attribute string from an element dict."""
|
|
17
|
+
parts = []
|
|
18
|
+
parts.append(export_html_element_class_and_id(elements['()']))
|
|
19
|
+
parts.append(export_html_element_style(elements['[]']))
|
|
20
|
+
parts.append(export_html_element_inline(elements['{}']))
|
|
21
|
+
return ''.join(parts)
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def export_html_element_style(text):
|
|
25
|
+
if not text:
|
|
26
|
+
return ''
|
|
27
|
+
return f' style="{text}"'
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def export_html_element_class_and_id(text):
|
|
31
|
+
"""Parse '.class1 .class2 #id' into class="..." id="..." attributes."""
|
|
32
|
+
if not text:
|
|
33
|
+
return ''
|
|
34
|
+
classes = []
|
|
35
|
+
ids = []
|
|
36
|
+
for token in text.split():
|
|
37
|
+
if token.startswith('.'):
|
|
38
|
+
classes.append(token[1:])
|
|
39
|
+
elif token.startswith('#'):
|
|
40
|
+
ids.append(token[1:])
|
|
41
|
+
parts = []
|
|
42
|
+
if classes:
|
|
43
|
+
parts.append(f' class="{" ".join(classes)}"')
|
|
44
|
+
if ids:
|
|
45
|
+
parts.append(f' id="{" ".join(ids)}"')
|
|
46
|
+
return ''.join(parts)
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def export_html_element_inline(text):
|
|
50
|
+
if not text:
|
|
51
|
+
return ''
|
|
52
|
+
return ' ' + text
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
# ---------------------------------------------------------------------------
|
|
56
|
+
# Tag element renderers
|
|
57
|
+
# ---------------------------------------------------------------------------
|
|
58
|
+
|
|
59
|
+
def export_html_generic(elements, tag, tag_to_close):
|
|
60
|
+
"""Render a generic div/span element."""
|
|
61
|
+
html = f'<{tag}{_build_attrs(elements)}>'
|
|
62
|
+
|
|
63
|
+
text = elements['text']
|
|
64
|
+
if text:
|
|
65
|
+
if text.endswith('::'):
|
|
66
|
+
text = text[:-2]
|
|
67
|
+
html += text + f'</{tag}>'
|
|
68
|
+
return html
|
|
69
|
+
html += text
|
|
70
|
+
|
|
71
|
+
tag_to_close.append(tag)
|
|
72
|
+
return html
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def export_html_link(elements):
|
|
76
|
+
"""Render a link:: element as <a href="...">text</a>."""
|
|
77
|
+
attrs = export_html_element_class_and_id(elements['()'])
|
|
78
|
+
attrs += export_html_element_inline(elements['{}'])
|
|
79
|
+
return f'<a{attrs} href="{elements["text"]}">{elements["[]"]}</a>'
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def export_html_img(elements):
|
|
83
|
+
"""Render an img:: element as <img src="..." alt="...">."""
|
|
84
|
+
src = elements['text']
|
|
85
|
+
return f'<img{_build_attrs(elements)} src="{src}" alt="{src}">'
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
# ---------------------------------------------------------------------------
|
|
89
|
+
# Video rendering with codec detection
|
|
90
|
+
# ---------------------------------------------------------------------------
|
|
91
|
+
|
|
92
|
+
CACHE_VIDEO_DIR = 'cache_video_codecs/'
|
|
93
|
+
|
|
94
|
+
VIDEO_CODECS = [
|
|
95
|
+
('-vp9.webm', 'video/webm'),
|
|
96
|
+
('-h265.mp4', 'video/mp4'),
|
|
97
|
+
('-h264.mp4', 'video/mp4'),
|
|
98
|
+
]
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def export_html_video(elements, default_inline='', current_directory=''):
|
|
102
|
+
"""Render a video:: or videoplay:: element with codec variants."""
|
|
103
|
+
source = elements['text']
|
|
104
|
+
extension = source.rsplit('.', 1)[-1]
|
|
105
|
+
poster_candidate = source.rsplit(f'.{extension}', 1)[0] + '-poster.jpg'
|
|
106
|
+
|
|
107
|
+
parts = ['<video']
|
|
108
|
+
parts.append(export_html_element_inline(elements['{}']))
|
|
109
|
+
if default_inline:
|
|
110
|
+
parts.append(f' {default_inline} ')
|
|
111
|
+
parts.append(export_html_element_class_and_id(elements['()']))
|
|
112
|
+
parts.append(export_html_element_style(elements['[]']))
|
|
113
|
+
if os.path.isfile(poster_candidate):
|
|
114
|
+
parts.append(f' poster="{poster_candidate}"')
|
|
115
|
+
parts.append('>\n')
|
|
116
|
+
|
|
117
|
+
# Look for transcoded codec variants
|
|
118
|
+
source_name, _ = os.path.splitext(source)
|
|
119
|
+
source_basename = os.path.basename(source_name)
|
|
120
|
+
source_dirname = os.path.dirname(source_name)
|
|
121
|
+
|
|
122
|
+
found_codecs = False
|
|
123
|
+
if current_directory and os.path.basename(source_dirname) == 'assets':
|
|
124
|
+
codec_dir = os.path.join(source_dirname, CACHE_VIDEO_DIR)
|
|
125
|
+
full_codec_dir = os.path.join(current_directory, codec_dir)
|
|
126
|
+
if os.path.isdir(full_codec_dir):
|
|
127
|
+
for suffix, mime in VIDEO_CODECS:
|
|
128
|
+
candidate = os.path.join(codec_dir, source_basename + suffix)
|
|
129
|
+
if os.path.isfile(os.path.join(current_directory, candidate)):
|
|
130
|
+
parts.append(f'\t<source src="{candidate}" type="{mime}">\n')
|
|
131
|
+
found_codecs = True
|
|
132
|
+
|
|
133
|
+
if not found_codecs:
|
|
134
|
+
parts.append(f'\t<source src="{source}" type="video/{extension}">\n')
|
|
135
|
+
|
|
136
|
+
parts.append(f'\t Cannot play video {source}\n')
|
|
137
|
+
parts.append('</video>')
|
|
138
|
+
return ''.join(parts)
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
# ---------------------------------------------------------------------------
|
|
142
|
+
# Closing tag detection
|
|
143
|
+
# ---------------------------------------------------------------------------
|
|
144
|
+
|
|
145
|
+
def check_is_closing_tag(elements):
|
|
146
|
+
"""A bare :: (empty tag, span of exactly 2) is a closing tag."""
|
|
147
|
+
return elements['tag'] == '' and elements['index_end'] - elements['index_start'] == 2
|
lhtml/insert_in_text.py
ADDED
lhtml/listing.py
ADDED
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
"""List processing for LHTML.
|
|
2
|
+
|
|
3
|
+
Converts * item / ** subitem syntax to nested <ul><li> HTML.
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
import io
|
|
7
|
+
import re
|
|
8
|
+
|
|
9
|
+
_LIST_ITEM = re.compile(r'^(\*+) (.*)')
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def _analyse_line(line):
|
|
13
|
+
"""Extract nesting level and content from a line."""
|
|
14
|
+
m = _LIST_ITEM.search(line)
|
|
15
|
+
if m:
|
|
16
|
+
return len(m.group(1)), m.group(2) + '\n'
|
|
17
|
+
return 0, line
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def process_listing(text):
|
|
21
|
+
"""Convert * / ** / *** list items to nested <ul><li> HTML."""
|
|
22
|
+
lines = []
|
|
23
|
+
buf = io.StringIO(text)
|
|
24
|
+
for line in buf:
|
|
25
|
+
lines.append(_analyse_line(line))
|
|
26
|
+
|
|
27
|
+
parts = []
|
|
28
|
+
level = 0
|
|
29
|
+
|
|
30
|
+
for k, (line_level, content) in enumerate(lines):
|
|
31
|
+
# Open deeper levels
|
|
32
|
+
while level < line_level:
|
|
33
|
+
level += 1
|
|
34
|
+
if level > 1:
|
|
35
|
+
parts.append('<li>\n')
|
|
36
|
+
parts.append('<ul>\n')
|
|
37
|
+
|
|
38
|
+
if level > 0:
|
|
39
|
+
parts.append('<li>\n')
|
|
40
|
+
|
|
41
|
+
parts.append(content)
|
|
42
|
+
|
|
43
|
+
if level > 0:
|
|
44
|
+
parts.append('</li>\n')
|
|
45
|
+
|
|
46
|
+
# Close shallower levels (peek at next line)
|
|
47
|
+
next_level = lines[k + 1][0] if k < len(lines) - 1 else 0
|
|
48
|
+
while level > next_level:
|
|
49
|
+
level -= 1
|
|
50
|
+
parts.append('</ul>\n')
|
|
51
|
+
if level >= 1:
|
|
52
|
+
parts.append('</li>\n')
|
|
53
|
+
|
|
54
|
+
return ''.join(parts)
|