lhtml-markup 2.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
lhtml/patterns.py ADDED
@@ -0,0 +1,92 @@
1
+ """Centralized regex patterns and text transformation utilities for LHTML."""
2
+
3
+ import re
4
+ from typing import Callable
5
+
6
+
7
+ # ---------------------------------------------------------------------------
8
+ # Compiled regex patterns
9
+ # ---------------------------------------------------------------------------
10
+
11
+ YAML_FRONTMATTER = re.compile(r'^---\n(.*?)\n---$', re.DOTALL | re.MULTILINE)
12
+ VERBATIM_BLOCK = re.compile(r'verbatim::\[\](.*?)verbatim::\[-\]', re.DOTALL | re.MULTILINE)
13
+ VERBATIM_INDEX = re.compile(r'verbatim::\[(.*?)\]')
14
+ CODE_BLOCK = re.compile(r'code::(.*?)code::\[-\]', re.DOTALL | re.MULTILINE)
15
+ CODE_INDEX = re.compile(r'code::\[(.*?)\]')
16
+ HEADING = re.compile(r'^(=+)(?:\((.*?)\))? (.*?)$', re.MULTILINE)
17
+ BOLD = re.compile(r'\*\*(.*?)\*\*')
18
+ ITALIC = re.compile(r'__(.*?)__')
19
+ INLINE_CODE = re.compile(r'`(.*?)`')
20
+ COMMENT = re.compile(r'::#(.*?)$', re.MULTILINE)
21
+ INCLUDE = re.compile(r'include::')
22
+ TAG_MARKER = re.compile(r'::')
23
+ LIST_ITEM = re.compile(r'^(\*+) (.*)')
24
+
25
+ # String constants
26
+ VERBATIM_OPEN = 'verbatim::[]'
27
+ VERBATIM_CLOSE = 'verbatim::[-]'
28
+ CODE_CLOSE = 'code::[-]'
29
+ SPACER_TAG = 'nl'
30
+
31
+ MAX_INCLUDE_ITERATIONS = 20
32
+
33
+
34
+ # ---------------------------------------------------------------------------
35
+ # Generic regex transform utility
36
+ # ---------------------------------------------------------------------------
37
+
38
+ def store_to_index(text: str, pattern: re.Pattern | str, name: str, store: list) -> str:
39
+ """Extract regex matches into a store, replacing with indexed placeholders.
40
+
41
+ Each match is stored in `store` and replaced with `name::[index]`.
42
+ Used to protect code/verbatim blocks from further processing.
43
+ """
44
+ if isinstance(pattern, str):
45
+ pattern = re.compile(pattern, re.DOTALL | re.MULTILINE)
46
+ parts = []
47
+ prev = 0
48
+ for m in pattern.finditer(text):
49
+ idx = len(store)
50
+ store.append(m.group(0))
51
+ parts.append(text[prev:m.start()])
52
+ parts.append(f'{name}::[{idx}]')
53
+ prev = m.end()
54
+ parts.append(text[prev:])
55
+ return ''.join(parts)
56
+
57
+
58
+ def restore_from_index(text: str, pattern: re.Pattern | str, store: list) -> str:
59
+ """Restore indexed placeholders from a store.
60
+
61
+ Matches `name::[index]` patterns and replaces with the stored content.
62
+ """
63
+ if isinstance(pattern, str):
64
+ pattern = re.compile(pattern)
65
+ parts = []
66
+ prev = 0
67
+ for m in pattern.finditer(text):
68
+ idx = int(m.group(1))
69
+ parts.append(text[prev:m.start()])
70
+ parts.append(store[idx])
71
+ prev = m.end()
72
+ parts.append(text[prev:])
73
+ return ''.join(parts)
74
+
75
+
76
+ def regex_transform(text: str, pattern: re.Pattern, transform_fn: Callable[[re.Match], str]) -> str:
77
+ """Apply a regex-based transformation across text.
78
+
79
+ For each match of `pattern` in `text`, calls `transform_fn(match)`
80
+ to produce the replacement string. Non-matching text passes through.
81
+
82
+ This eliminates the boilerplate loop duplicated across process_bold,
83
+ process_italic, process_code_inline, process_title, etc.
84
+ """
85
+ parts = []
86
+ prev = 0
87
+ for m in pattern.finditer(text):
88
+ parts.append(text[prev:m.start()])
89
+ parts.append(transform_fn(m))
90
+ prev = m.end()
91
+ parts.append(text[prev:])
92
+ return ''.join(parts)
lhtml/pipeline.py ADDED
@@ -0,0 +1,215 @@
1
+ """Processing pipeline and tag handler registry for LHTML.
2
+
3
+ The pipeline orchestrates the multi-pass LHTML-to-HTML transformation.
4
+ Tag handlers are registered in a plugin registry, making it easy to
5
+ add custom element types (e.g., new :: tag names) without modifying
6
+ the core processing code.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import os
12
+ import warnings
13
+ from dataclasses import dataclass, field
14
+ from typing import Callable
15
+
16
+ from .errors import LHTMLIncludeLoopError
17
+ from .patterns import MAX_INCLUDE_ITERATIONS
18
+
19
+
20
+ # ---------------------------------------------------------------------------
21
+ # Configuration
22
+ # ---------------------------------------------------------------------------
23
+
24
+ META_DEFAULTS = {
25
+ 'wrap-auto': False,
26
+ 'add_title_id': False,
27
+ 'title': 'Webpage',
28
+ 'css': [],
29
+ 'js': [],
30
+ 'wrap-custom-pre': '',
31
+ 'wrap-custom-post': '',
32
+ 'directory_include': [os.getcwd() + '/'],
33
+ }
34
+
35
+
36
+ @dataclass
37
+ class ProcessingContext:
38
+ """Mutable state carried through the pipeline."""
39
+ text: str
40
+ meta: dict = field(default_factory=dict)
41
+ verbatim_store: list = field(default_factory=list)
42
+ code_store: list = field(default_factory=list)
43
+ current_directory: str = ''
44
+
45
+
46
+ # ---------------------------------------------------------------------------
47
+ # Tag handler registry (plugin system)
48
+ # ---------------------------------------------------------------------------
49
+
50
+ # A tag handler receives (element_dict, tag_to_close, current_directory)
51
+ # and returns (html_string, is_real_tag).
52
+ TagHandler = Callable[[dict, list, str], tuple[str, bool]]
53
+
54
+
55
+ class TagRegistry:
56
+ """Registry for LHTML :: tag element handlers.
57
+
58
+ Built-in tags (div, span, link, img, video, videoplay) are
59
+ registered at import time. Users can register custom tags:
60
+
61
+ from lhtml.pipeline import tag_registry
62
+ tag_registry.register('custom', my_handler)
63
+ """
64
+
65
+ def __init__(self):
66
+ self._handlers: dict[str, TagHandler] = {}
67
+
68
+ def register(self, tag_name: str, handler: TagHandler):
69
+ """Register a handler for a tag name."""
70
+ self._handlers[tag_name] = handler
71
+
72
+ def get(self, tag_name: str) -> TagHandler | None:
73
+ """Look up the handler for a tag name."""
74
+ return self._handlers.get(tag_name)
75
+
76
+ def has(self, tag_name: str) -> bool:
77
+ return tag_name in self._handlers
78
+
79
+ def registered_tags(self) -> list[str]:
80
+ return list(self._handlers.keys())
81
+
82
+
83
+ # Global registry instance
84
+ tag_registry = TagRegistry()
85
+
86
+
87
+ def _register_builtin_tags():
88
+ """Register the built-in LHTML tag handlers."""
89
+ from .export_html import (
90
+ export_html_generic, export_html_link,
91
+ export_html_img, export_html_video,
92
+ )
93
+
94
+ def _handle_generic(tag_name):
95
+ def handler(element, tag_to_close, current_directory):
96
+ return export_html_generic(element, tag_name, tag_to_close), True
97
+ return handler
98
+
99
+ tag_registry.register('div', _handle_generic('div'))
100
+ tag_registry.register('span', _handle_generic('span'))
101
+
102
+ def _handle_link(element, tag_to_close, current_directory):
103
+ return export_html_link(element), True
104
+ tag_registry.register('link', _handle_link)
105
+
106
+ def _handle_img(element, tag_to_close, current_directory):
107
+ return export_html_img(element), True
108
+ tag_registry.register('img', _handle_img)
109
+
110
+ def _handle_video(element, tag_to_close, current_directory):
111
+ return export_html_video(element, '', current_directory), True
112
+ tag_registry.register('video', _handle_video)
113
+
114
+ def _handle_videoplay(element, tag_to_close, current_directory):
115
+ return export_html_video(element, 'autoplay loop muted', current_directory), True
116
+ tag_registry.register('videoplay', _handle_videoplay)
117
+
118
+
119
+ # ---------------------------------------------------------------------------
120
+ # Code lexer registry (plugin system for syntax highlighting)
121
+ # ---------------------------------------------------------------------------
122
+
123
+ class LexerRegistry:
124
+ """Registry for custom Pygments lexers by language name."""
125
+
126
+ def __init__(self):
127
+ self._lexers: dict[str, type] = {}
128
+
129
+ def register(self, language: str, lexer_class: type):
130
+ self._lexers[language] = lexer_class
131
+
132
+ def get(self, language: str) -> type | None:
133
+ return self._lexers.get(language)
134
+
135
+
136
+ lexer_registry = LexerRegistry()
137
+
138
+
139
+ # ---------------------------------------------------------------------------
140
+ # Pipeline
141
+ # ---------------------------------------------------------------------------
142
+
143
+ class ProcessingPipeline:
144
+ """Orchestrates the LHTML-to-HTML transformation pipeline.
145
+
146
+ Usage:
147
+ pipeline = ProcessingPipeline()
148
+ html = pipeline.run(lhtml_text)
149
+ html = pipeline.run(lhtml_text, meta={'wrap-auto': True})
150
+ """
151
+
152
+ def __init__(self, registry: TagRegistry | None = None):
153
+ self.tag_registry = registry or tag_registry
154
+
155
+ def run(self, text: str, meta_arg: dict | None = None) -> str:
156
+ from .process import (
157
+ process_yaml, process_verbatim_to_index,
158
+ process_remove_comment, process_include,
159
+ process_title, process_listing,
160
+ process_bold, process_italic, process_code_inline,
161
+ process_tag, process_code,
162
+ process_verbatim_back_from_index,
163
+ )
164
+ from .wrap_html import wrap_auto
165
+ from .patterns import CODE_BLOCK, CODE_INDEX, store_to_index, restore_from_index
166
+
167
+ ctx = ProcessingContext(
168
+ text=text,
169
+ meta={**META_DEFAULTS, **(meta_arg or {})},
170
+ )
171
+
172
+ # Phase 1: YAML front matter
173
+ ctx.text, meta_yaml = process_yaml(ctx.text)
174
+ ctx.meta = {**ctx.meta, **meta_yaml}
175
+ ctx.current_directory = ctx.meta.get('current_directory', '')
176
+
177
+ # Phase 2: Verbatim / comments / includes (iterative)
178
+ found_include = True
179
+ iteration = 0
180
+ while found_include:
181
+ ctx.text = process_verbatim_to_index(ctx.text, ctx.verbatim_store)
182
+ ctx.text = process_remove_comment(ctx.text)
183
+ ctx.text, found_include = process_include(ctx.text, ctx.meta['directory_include'])
184
+ iteration += 1
185
+ if iteration > MAX_INCLUDE_ITERATIONS:
186
+ warnings.warn(str(LHTMLIncludeLoopError(MAX_INCLUDE_ITERATIONS)))
187
+ break
188
+
189
+ # Phase 3: Protect code blocks
190
+ ctx.text = store_to_index(ctx.text, CODE_BLOCK, 'code', ctx.code_store)
191
+
192
+ # Phase 4: Block-level elements
193
+ ctx.text = process_title(ctx.text)
194
+ ctx.text = process_listing(ctx.text)
195
+
196
+ # Phase 5: Inline elements
197
+ ctx.text = process_bold(ctx.text)
198
+ ctx.text = process_italic(ctx.text)
199
+ ctx.text = process_code_inline(ctx.text)
200
+
201
+ # Phase 6: Tag elements (uses tag_registry)
202
+ ctx.text = process_tag(ctx.text, ctx.current_directory, self.tag_registry)
203
+
204
+ # Phase 7: Restore code blocks with highlighting
205
+ ctx.text = restore_from_index(ctx.text, CODE_INDEX, ctx.code_store)
206
+ ctx.text = process_code(ctx.text)
207
+
208
+ # Phase 8: Restore verbatim blocks
209
+ ctx.text = process_verbatim_back_from_index(ctx.text, ctx.verbatim_store)
210
+
211
+ # Phase 9: Optional HTML wrapping
212
+ if ctx.meta.get('wrap-auto') is True:
213
+ ctx.text = wrap_auto(ctx.text, ctx.meta)
214
+
215
+ return ctx.text
lhtml/process.py ADDED
@@ -0,0 +1,236 @@
1
+ """Core LHTML processing functions.
2
+
3
+ Each function transforms LHTML markup into HTML for one language
4
+ feature (headings, bold, lists, tags, etc.). The pipeline execution
5
+ order is managed by pipeline.py.
6
+ """
7
+
8
+ import os
9
+ import re
10
+ import warnings
11
+
12
+ import yaml
13
+
14
+ from .element_extract import extract_bracket_elements
15
+ from .export_html import (
16
+ export_html_element_class_and_id, export_html_generic,
17
+ export_html_link, export_html_img, export_html_video,
18
+ check_is_closing_tag,
19
+ )
20
+ from .listing import process_listing # noqa: F401 — re-exported
21
+ from .code import export_html_code
22
+ from .errors import LHTMLFileNotFound, LHTMLTagStackError
23
+ from .patterns import (
24
+ YAML_FRONTMATTER, VERBATIM_BLOCK, VERBATIM_INDEX,
25
+ CODE_BLOCK, HEADING, BOLD, ITALIC, INLINE_CODE,
26
+ COMMENT, INCLUDE, TAG_MARKER,
27
+ VERBATIM_OPEN, VERBATIM_CLOSE, CODE_CLOSE, SPACER_TAG,
28
+ regex_transform,
29
+ )
30
+
31
+
32
+ # ---------------------------------------------------------------------------
33
+ # YAML front matter
34
+ # ---------------------------------------------------------------------------
35
+
36
+ def process_yaml(text):
37
+ """Extract YAML front matter and return (remaining_text, meta_dict)."""
38
+ match = YAML_FRONTMATTER.search(text)
39
+ if match:
40
+ new_text = text[:match.start()] + text[match.end():]
41
+ try:
42
+ yaml_content = yaml.load(match.group(1), Loader=yaml.FullLoader) or {}
43
+ except yaml.YAMLError:
44
+ yaml_content = {}
45
+ return new_text, yaml_content
46
+ return text, {}
47
+
48
+
49
+ # ---------------------------------------------------------------------------
50
+ # Verbatim blocks (protect / restore)
51
+ # ---------------------------------------------------------------------------
52
+
53
+ def process_verbatim_to_index(text, verbatim_index_store):
54
+ """Replace verbatim blocks with indexed placeholders."""
55
+ def _store(m):
56
+ content = m.group(0)[len(VERBATIM_OPEN):-len(VERBATIM_CLOSE)]
57
+ idx = len(verbatim_index_store)
58
+ verbatim_index_store.append(content)
59
+ return f'verbatim::[{idx}]'
60
+ return regex_transform(text, VERBATIM_BLOCK, _store)
61
+
62
+
63
+ def process_verbatim_back_from_index(text, verbatim_index_store):
64
+ """Restore verbatim blocks from indexed placeholders."""
65
+ def _restore(m):
66
+ return verbatim_index_store[int(m.group(1))]
67
+ return regex_transform(text, VERBATIM_INDEX, _restore)
68
+
69
+
70
+ # ---------------------------------------------------------------------------
71
+ # Comments
72
+ # ---------------------------------------------------------------------------
73
+
74
+ def process_remove_comment(text):
75
+ """Remove ::#comment lines."""
76
+ return regex_transform(text, COMMENT, lambda m: '\n')
77
+
78
+
79
+ # ---------------------------------------------------------------------------
80
+ # File inclusion
81
+ # ---------------------------------------------------------------------------
82
+
83
+ def find_file(directories, filename):
84
+ """Locate a file in the given directory list."""
85
+ for d in directories:
86
+ pathname = os.path.join(d, filename)
87
+ if os.path.isfile(pathname):
88
+ return pathname
89
+ raise LHTMLFileNotFound(filename, directories)
90
+
91
+
92
+ def process_include(text, directory):
93
+ """Expand include:: directives. Returns (new_text, found_any)."""
94
+ found_include = False
95
+ parts = []
96
+ prev = 0
97
+
98
+ for m in INCLUDE.finditer(text):
99
+ found_include = True
100
+ element = extract_bracket_elements(text, m.end())
101
+ filename = find_file(directory, element['text'])
102
+
103
+ with open(filename, 'r') as fid:
104
+ included = fid.read()
105
+
106
+ parts.append(text[prev:m.start()])
107
+ parts.append(included)
108
+ prev = element['index_end']
109
+
110
+ parts.append(text[prev:])
111
+ return ''.join(parts), found_include
112
+
113
+
114
+ # ---------------------------------------------------------------------------
115
+ # Inline formatting
116
+ # ---------------------------------------------------------------------------
117
+
118
+ def process_bold(text):
119
+ """Convert **text** to <strong>text</strong>."""
120
+ return regex_transform(text, BOLD, lambda m: f'<strong>{m.group(1)}</strong>')
121
+
122
+
123
+ def process_italic(text):
124
+ """Convert __text__ to <em>text</em>."""
125
+ return regex_transform(text, ITALIC, lambda m: f'<em>{m.group(1)}</em>')
126
+
127
+
128
+ def process_code_inline(text):
129
+ """Convert `text` to <code class="code-inline">text</code>."""
130
+ return regex_transform(text, INLINE_CODE,
131
+ lambda m: f'<code class="code-inline">{m.group(1)}</code>')
132
+
133
+
134
+ # ---------------------------------------------------------------------------
135
+ # Headings
136
+ # ---------------------------------------------------------------------------
137
+
138
+ def process_title(text):
139
+ """Convert = Title or =(.class) Title to <h1>Title</h1>, etc."""
140
+ def _heading(m):
141
+ level = str(len(m.group(1)))
142
+ class_id = (m.group(2) or '').strip()
143
+ title = m.group(3)
144
+ if class_id:
145
+ attrs = export_html_element_class_and_id(class_id)
146
+ return f'<h{level}{attrs}>{title}</h{level}>\n'
147
+ return f'<h{level}>{title}</h{level}>\n'
148
+ return regex_transform(text, HEADING, _heading)
149
+
150
+
151
+ # ---------------------------------------------------------------------------
152
+ # Code blocks
153
+ # ---------------------------------------------------------------------------
154
+
155
+ def process_code(text):
156
+ """Parse and highlight code:: blocks."""
157
+ def _highlight(m):
158
+ element = extract_bracket_elements(text, m.start() + len('code::'))
159
+ code = text[element['index_end']:m.end() - len(CODE_CLOSE)]
160
+ language = element['[]']
161
+ return export_html_code(code, language)
162
+ return regex_transform(text, CODE_BLOCK, _highlight)
163
+
164
+
165
+ # ---------------------------------------------------------------------------
166
+ # Tag elements (the :: system)
167
+ # ---------------------------------------------------------------------------
168
+
169
+ def _dispatch_tag(element, tag_to_close, current_directory, registry=None):
170
+ """Dispatch a parsed tag element to the appropriate handler.
171
+
172
+ Uses the tag_registry if provided, otherwise falls back to
173
+ built-in dispatch for backward compatibility.
174
+ """
175
+ tag = element['tag']
176
+
177
+ # Try plugin registry first
178
+ if registry is not None:
179
+ handler = registry.get(tag)
180
+ if handler is not None:
181
+ return handler(element, tag_to_close, current_directory)
182
+
183
+ # Built-in tag dispatch (used when no registry, or tag not in registry)
184
+ if tag == 'div' or tag == 'span':
185
+ return export_html_generic(element, tag, tag_to_close), True
186
+ elif tag == 'link':
187
+ return export_html_link(element), True
188
+ elif tag == 'img':
189
+ return export_html_img(element), True
190
+ elif tag == 'video':
191
+ return export_html_video(element, '', current_directory), True
192
+ elif tag == 'videoplay':
193
+ return export_html_video(element, 'autoplay loop muted', current_directory), True
194
+ elif tag == '':
195
+ if check_is_closing_tag(element):
196
+ if not tag_to_close:
197
+ warnings.warn(
198
+ str(LHTMLTagStackError(source_pos=element.get('index_start', -1))),
199
+ stacklevel=3,
200
+ )
201
+ return '::??ERROR', True
202
+ return '</' + tag_to_close.pop() + '>', True
203
+ if element['text'] == SPACER_TAG:
204
+ return '<div style="height:1em;"></div>', True
205
+ if element['[]'] or element['()'] or element['{}'] or element['text']:
206
+ return export_html_generic(element, 'div', tag_to_close), True
207
+ return '', False
208
+
209
+ # Unrecognized tag — pass through
210
+ return '', False
211
+
212
+
213
+ def process_tag(text, current_directory='', registry=None):
214
+ """Process all :: tag elements in text."""
215
+ parts = []
216
+ prev = 0
217
+ tag_to_close = []
218
+
219
+ for m in TAG_MARKER.finditer(text):
220
+ if m.start() <= prev:
221
+ continue # skip overlapping matches
222
+
223
+ element = extract_bracket_elements(text, m.end())
224
+ html, is_real_tag = _dispatch_tag(element, tag_to_close, current_directory, registry)
225
+ tag = element['tag']
226
+ index_end = element['index_end']
227
+
228
+ if is_real_tag:
229
+ parts.append(text[prev:m.start() - len(tag)])
230
+ parts.append(html)
231
+ else:
232
+ parts.append(text[prev:index_end])
233
+ prev = index_end
234
+
235
+ parts.append(text[prev:])
236
+ return ''.join(parts)
lhtml/tag_element.lark ADDED
@@ -0,0 +1,30 @@
1
+ // LHTML tag element grammar for the :: bracket subsystem.
2
+ //
3
+ // Parses the text after :: in constructs like:
4
+ // tag::(.class #id)[style]{attrs}text::
5
+ // div::[color:red;]
6
+ // link::url[link text]
7
+ // ::(.class)[style]
8
+ //
9
+ // Supports nested brackets: [outer[inner]still_outer]
10
+
11
+ start: bracket_and_text*
12
+
13
+ bracket_and_text: paren_group
14
+ | square_group
15
+ | curly_group
16
+ | bare_text
17
+
18
+ paren_group: "(" paren_content ")"
19
+ square_group: "[" square_content "]"
20
+ curly_group: "{" curly_content "}"
21
+
22
+ // Bracket content allows nested brackets of the same type
23
+ paren_content: (/[^()]+/ | "(" paren_content ")")*
24
+ square_content: (/[^\[\]]+/ | "[" square_content "]")*
25
+ curly_content: (/[^{}]+/ | "{" curly_content "}")*
26
+
27
+ // Bare text: any non-whitespace, non-bracket characters
28
+ bare_text: /[^\s\[\](){}]+/
29
+
30
+ %import common.WS_INLINE