doc2dict 0.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
doc2dict-0.1/PKG-INFO ADDED
@@ -0,0 +1,4 @@
1
+ Metadata-Version: 2.1
2
+ Name: doc2dict
3
+ Version: 0.1
4
+ Requires-Python: >=3.8
@@ -0,0 +1,3 @@
1
+ from .html_parser import html_reduction
2
+ from .visualizer import format_list_html
3
+ from .xml_parser import xml_parser
@@ -0,0 +1,302 @@
1
+ # Core functionality (original)
2
+ def standardize_css(style):
3
+ if not style:
4
+ return {}
5
+ return {
6
+ k.strip(): v.strip().split()
7
+ for k, v in (pair.split(':') for pair in style.split(';') if ':' in pair)
8
+ if k.strip() and v.strip()
9
+ }
10
+
11
+ def get_node_styles(node):
12
+ style = node.attributes.get('style')
13
+ if not style:
14
+ return {}
15
+ return standardize_css(style)
16
+
17
+ def get_text_style(node, styles, inherited_style=None):
18
+ text_style = inherited_style.copy() if inherited_style else {
19
+ 'bold': False,
20
+ 'italic': False,
21
+ 'underline': False
22
+ }
23
+
24
+ # Check HTML tags
25
+ if node.tag in {'b', 'strong'}:
26
+ text_style['bold'] = True
27
+ if node.tag in {'i', 'em'}:
28
+ text_style['italic'] = True
29
+ if node.tag == 'u':
30
+ text_style['underline'] = True
31
+
32
+ # Check CSS styles
33
+ font_weight = styles.get('font-weight', [''])[0]
34
+ if font_weight in {'bold', '700', '800', '900'}:
35
+ text_style['bold'] = True
36
+
37
+ font_style = styles.get('font-style', [''])[0]
38
+ if font_style == 'italic':
39
+ text_style['italic'] = True
40
+
41
+ text_decoration = styles.get('text-decoration', [''])[0]
42
+ if 'underline' in text_decoration:
43
+ text_style['underline'] = True
44
+
45
+ return text_style
46
+
47
+ def get_indent(styles):
48
+ # Handle direct indentation
49
+ margin_left = styles.get('margin-left', ['0'])[0]
50
+ text_indent = styles.get('text-indent', ['0'])[0]
51
+ text_align = styles.get('text-align', ['left'])[0] # default to left
52
+
53
+ def convert_to_pixels(value):
54
+ try:
55
+ if value.endswith('in'):
56
+ return float(value.rstrip('in')) * 96
57
+ elif value.endswith('cm'):
58
+ return float(value.rstrip('cm')) * 37.8
59
+ elif value.endswith('mm'):
60
+ return float(value.rstrip('mm')) * 3.78
61
+ elif value.endswith('pt'):
62
+ return float(value.rstrip('pt')) * 1.333
63
+ elif value.endswith('px'):
64
+ return float(value.rstrip('px'))
65
+ elif value.endswith('%'):
66
+ return float(value.rstrip('%'))
67
+ else:
68
+ return float(value)
69
+ except ValueError:
70
+ return 0
71
+
72
+ # Calculate physical indentation in pixels
73
+ margin = convert_to_pixels(margin_left)
74
+ indent = convert_to_pixels(text_indent)
75
+
76
+ # Convert alignment to positions on 0-100 scale
77
+ # left = 0, center = 50, right = 100
78
+ if text_align == 'center':
79
+ return 50
80
+ elif text_align == 'right':
81
+ return 100
82
+ else: # left or justify
83
+ # Convert physical indent to percentage (assuming standard page width of ~700px)
84
+ # This is approximate - might need to adjust the scaling factor
85
+ page_width = 700 # assumed page width in pixels
86
+ position = ((margin + indent) / page_width) * 100
87
+ return min(100, position) # cap at 100
88
+
89
+ def is_inline(node, styles):
90
+ if node.tag in {'span', 'a', 'strong', 'em', 'b', 'i', 'u'}:
91
+ return True
92
+ display = styles.get('display', [''])[0]
93
+ return display == 'inline'
94
+
95
+ def iterwalk(root, include_text):
96
+ def walk(node, inherited_styles=None, inherited_text_style=None):
97
+ current_styles = dict(inherited_styles or {})
98
+ node_styles = get_node_styles(node)
99
+ current_styles.update(node_styles)
100
+
101
+ if current_styles.get('display', [''])[0] == 'none':
102
+ return
103
+
104
+ text_content = node.text_content
105
+ if text_content:
106
+ text_transform = current_styles.get('text-transform', [''])[0]
107
+ if text_transform == 'uppercase':
108
+ text_content = text_content.upper()
109
+ elif text_transform == 'lowercase':
110
+ text_content = text_content.lower()
111
+ elif text_transform == 'capitalize':
112
+ text_content = text_content.capitalize()
113
+
114
+ text_style = get_text_style(node, current_styles, inherited_text_style)
115
+
116
+ yield ("start", node, current_styles, text_content, text_style)
117
+ for child in node.iter(include_text=include_text):
118
+ if child is not node:
119
+ yield from walk(child, current_styles, text_style)
120
+ yield ("end", node, current_styles, text_content, text_style)
121
+
122
+ for node in root.iter(include_text=include_text):
123
+ if node is not root:
124
+ yield from walk(node)
125
+
126
+ # New height calculation functionality (add/modify these sections)
127
+ def parse_font_size(size_str, base_size):
128
+ """Parse font size with precise point-to-pixel conversion"""
129
+ try:
130
+ if size_str.endswith('pt'):
131
+ # Convert points to pixels (1pt = 1.333px)
132
+ return float(size_str[:-2]) * 1.333
133
+ elif size_str.endswith('px'):
134
+ return float(size_str[:-2])
135
+ elif size_str.endswith('%'):
136
+ return base_size * float(size_str[:-1]) / 100
137
+ elif size_str.endswith('em'):
138
+ return base_size * float(size_str[:-2])
139
+ elif size_str.endswith('rem'):
140
+ return 16 * float(size_str[:-3]) # rem is relative to root (usually 16px)
141
+ return float(size_str) # Assume pixels if no unit
142
+ except ValueError:
143
+ return base_size # Return base size if parsing fails
144
+
145
+ def get_effective_style(node, parent_styles=None):
146
+ """Get combined styles considering element and parent context"""
147
+ styles = {}
148
+
149
+ # Start with parent styles
150
+ if parent_styles:
151
+ styles.update(parent_styles)
152
+
153
+ # Add element's own styles
154
+ element_styles = get_node_styles(node)
155
+ for key, value in element_styles.items():
156
+ if key in {'font-size', 'line-height', 'margin-top', 'margin-bottom',
157
+ 'padding-top', 'padding-bottom', 'border-top', 'border-bottom'}:
158
+ styles[key] = value
159
+
160
+ return styles
161
+
162
+ def calculate_compound_height(node, inherited_size=None):
163
+ """Calculate exact height based on all style factors"""
164
+ base_size = inherited_size or 16
165
+ styles = get_effective_style(node)
166
+
167
+ # Get font size
168
+ if 'font-size' in styles:
169
+ font_size = parse_font_size(styles['font-size'][0], base_size)
170
+ else:
171
+ font_size = base_size
172
+
173
+ # Calculate line height (default 1.2 if not specified)
174
+ line_height_mult = 1.2
175
+ if 'line-height' in styles:
176
+ try:
177
+ line_height_mult = float(styles['line-height'][0])
178
+ except ValueError:
179
+ pass
180
+
181
+ content_height = font_size * line_height_mult
182
+
183
+ # Add margins, padding, borders
184
+ spacing = 0
185
+ for prop in ['margin-top', 'margin-bottom', 'padding-top', 'padding-bottom']:
186
+ if prop in styles:
187
+ spacing += parse_font_size(styles[prop][0], base_size)
188
+
189
+ # Handle borders
190
+ if 'border-top' in styles:
191
+ border_value = styles['border-top'][0].split()[0]
192
+ spacing += parse_font_size(border_value, base_size)
193
+ if 'border-bottom' in styles:
194
+ border_value = styles['border-bottom'][0].split()[0]
195
+ spacing += parse_font_size(border_value, base_size)
196
+
197
+ # round to 2 decimal places
198
+ font_size = round(font_size, 2)
199
+ content_height = round(content_height, 2)
200
+ spacing = round(spacing, 2)
201
+
202
+ return {
203
+ 'font_size': font_size,
204
+ 'content_height': content_height,
205
+ 'total_height': content_height + spacing
206
+ }
207
+
208
+ def html_reduction(tree):
209
+ """Main processing function"""
210
+ blocks = {'p', 'div', 'h1', 'h2', 'h3', 'h4', 'h5', 'h6', 'article', 'section', 'li'}
211
+ skip_elements = {'script', 'style', 'noscript'}
212
+ lines = []
213
+ current_line = []
214
+ row_content = []
215
+ in_row = False
216
+ inherited_size = 16 # Base font size
217
+
218
+ def flush_line():
219
+ if current_line:
220
+ items = []
221
+ for text, style in current_line:
222
+ if text.strip():
223
+ height_metrics = calculate_compound_height(current_node, inherited_size)
224
+ items.append({
225
+ "text": text.strip(),
226
+ "indent": get_indent(current_styles),
227
+ "bold": style.get('bold', False),
228
+ "italic": style.get('italic', False),
229
+ "underline": style.get('underline', False),
230
+ "height": height_metrics['total_height'],
231
+ "font_size": height_metrics['font_size']
232
+ })
233
+ if items:
234
+ lines.append(items)
235
+ current_line.clear()
236
+
237
+ for event, node, styles, text_content, text_style in iterwalk(tree.body, include_text=True):
238
+ if node.tag in skip_elements:
239
+ continue
240
+
241
+ current_node = node
242
+ current_styles = styles
243
+ current_text_style = text_style
244
+
245
+ if event == "start":
246
+ # Update inherited size based on parent styles
247
+ if 'font-size' in styles:
248
+ inherited_size = parse_font_size(styles['font-size'][0], inherited_size)
249
+
250
+ # Process content based on node type
251
+ if node.tag == 'tr':
252
+ if event == "start":
253
+ in_row = True
254
+ row_content = []
255
+ elif event == "end":
256
+ if row_content:
257
+ height_metrics = calculate_compound_height(node, inherited_size)
258
+ lines.append([{
259
+ "text": text,
260
+ "indent": 0,
261
+ "bold": style['bold'], # Use captured style
262
+ "italic": style['italic'], # Use captured style
263
+ "underline": style['underline'], # Use captured style
264
+ "height": height_metrics['total_height'],
265
+ "font_size": height_metrics['font_size']
266
+ } for text, style in row_content]) # Unpack both text and style
267
+ in_row = False
268
+ row_content = []
269
+ continue
270
+
271
+ if in_row and event == "start" and text_content:
272
+ text = text_content.strip()
273
+ if text:
274
+ # Store both text and style
275
+ row_content.append((text, text_style))
276
+ continue
277
+
278
+ if (node.tag in blocks or styles.get('display', [''])[0] == 'flex') and not is_inline(node, styles):
279
+ if event == "start":
280
+ flush_line()
281
+ if text_content:
282
+ text = text_content.strip()
283
+ if text:
284
+ height_metrics = calculate_compound_height(node, inherited_size)
285
+ lines.append([{
286
+ "text": text,
287
+ "indent": get_indent(styles),
288
+ "bold": text_style['bold'],
289
+ "italic": text_style['italic'],
290
+ "underline": text_style['underline'],
291
+ "height": height_metrics['total_height'],
292
+ "font_size": height_metrics['font_size']
293
+ }])
294
+ elif event == "end":
295
+ flush_line()
296
+ elif event == "start" and text_content:
297
+ text = text_content.strip()
298
+ if text:
299
+ current_line.append((text, text_style)) # Store tuple of (text, style)
300
+
301
+ flush_line()
302
+ return lines
@@ -0,0 +1,125 @@
1
+ import webbrowser
2
+ import os
3
+
4
+ def format_list_html(nested_list):
5
+ # Simplified color scheme
6
+ single_line_color = '#E6F3FF' # Blue hue
7
+ multi_first_color = '#E6FFE6' # Green hue
8
+ multi_rest_color = '#FFE6E6' # Red hue
9
+
10
+ html_content = """
11
+ <html>
12
+ <head>
13
+ <style>
14
+ body {
15
+ font-family: Arial, sans-serif;
16
+ line-height: 1.6;
17
+ margin: 20px;
18
+ max-width: 1200px; /* Set a max-width for consistent scaling */
19
+ }
20
+ .row {
21
+ margin-bottom: 20px;
22
+ width: 100%;
23
+ position: relative; /* For absolute positioning of items */
24
+ }
25
+ .item-container {
26
+ display: inline-block;
27
+ position: relative; /* For proper indent positioning */
28
+ }
29
+ .item {
30
+ padding: 10px 15px;
31
+ border-radius: 5px;
32
+ margin-right: 15px;
33
+ margin-bottom: 10px;
34
+ display: inline-block;
35
+ }
36
+ </style>
37
+ </head>
38
+ <body>
39
+ """
40
+
41
+ def get_styled_text(item):
42
+ """Helper function to apply text styling"""
43
+ if not isinstance(item, dict):
44
+ return str(item)
45
+
46
+ height = str(item.get('height', ''))
47
+ text = item.get('text', '')
48
+ style_properties = []
49
+
50
+ if item.get('bold', False):
51
+ style_properties.append('font-weight: bold')
52
+ if item.get('italic', False):
53
+ style_properties.append('font-style: italic')
54
+ if item.get('underline', False):
55
+ style_properties.append('text-decoration: underline')
56
+ if height:
57
+ style_properties.append(f'font-size: {height}')
58
+
59
+ if style_properties:
60
+ return f"<span style='{'; '.join(style_properties)}'>{text}</span>"
61
+ return text
62
+
63
+ def calculate_position(indent):
64
+ """Convert normalized indent (0-100) to percentage position"""
65
+ if indent is None:
66
+ return 0
67
+ # Ensure indent is within 0-100 range
68
+ indent = max(0, min(100, indent))
69
+ # Convert to percentage for positioning
70
+ return indent
71
+
72
+ for items in nested_list:
73
+ if isinstance(items, list):
74
+ # Get indent from first item if it's a dictionary
75
+ first_indent = items[0].get('indent', 0) if isinstance(items[0], dict) else 0
76
+ position = calculate_position(first_indent)
77
+
78
+ html_content += f"<div class='row'>\n"
79
+ html_content += f" <div class='item-container' style='margin-left: {position}%'>\n"
80
+
81
+ for i, item in enumerate(items):
82
+ # Single item -> blue
83
+ if len(items) == 1:
84
+ color = single_line_color
85
+ # Multiple items: first -> green, rest -> red
86
+ else:
87
+ color = multi_first_color if i == 0 else multi_rest_color
88
+
89
+ styled_text = get_styled_text(item)
90
+ html_content += f" <span class='item' style='background-color: {color}'>{styled_text}</span>\n"
91
+
92
+ html_content += " </div>\n</div>\n"
93
+ else:
94
+ # Handle single items
95
+ indent = items.get('indent', 0) if isinstance(items, dict) else 0
96
+ position = calculate_position(indent)
97
+
98
+ html_content += f"<div class='row'>\n"
99
+ html_content += f" <div class='item-container' style='margin-left: {position}%'>\n"
100
+
101
+ if isinstance(items, dict):
102
+ styled_text = get_styled_text(items)
103
+ height = items.get('height', '')
104
+ height_style = f'height: {height};' if height else ''
105
+ html_content += f" <span class='item' style='background-color: {single_line_color}; {height_style}'>{styled_text}</span>\n"
106
+ else:
107
+ styled_text = str(items)
108
+ html_content += f" <span class='item' style='background-color: {single_line_color}'>{styled_text}</span>\n"
109
+
110
+ html_content += " </div>\n</div>\n"
111
+
112
+ html_content += """
113
+ </body>
114
+ </html>
115
+ """
116
+
117
+ # Write HTML content to a temporary file
118
+ with open('temp.html', 'w', encoding='utf-8') as f:
119
+ f.write(html_content)
120
+
121
+ # Get the absolute path of the file
122
+ file_path = os.path.abspath('temp.html')
123
+
124
+ # Open the HTML file in the default web browser
125
+ webbrowser.open('file://' + file_path)
@@ -0,0 +1,6 @@
1
+ import xmltodict
2
+ import json
3
+
4
+ # Modify to add mapping dict
5
+ def parse_xml(content):
6
+ return xmltodict.parse(content)
@@ -0,0 +1,4 @@
1
+ Metadata-Version: 2.1
2
+ Name: doc2dict
3
+ Version: 0.1
4
+ Requires-Python: >=3.8
@@ -0,0 +1,10 @@
1
+ setup.py
2
+ doc2dict/__init__.py
3
+ doc2dict/html_parser.py
4
+ doc2dict/visualizer.py
5
+ doc2dict/xml_parser.py
6
+ doc2dict.egg-info/PKG-INFO
7
+ doc2dict.egg-info/SOURCES.txt
8
+ doc2dict.egg-info/dependency_links.txt
9
+ doc2dict.egg-info/requires.txt
10
+ doc2dict.egg-info/top_level.txt
@@ -0,0 +1 @@
1
+ selectolax
@@ -0,0 +1 @@
1
+ doc2dict
doc2dict-0.1/setup.cfg ADDED
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+
doc2dict-0.1/setup.py ADDED
@@ -0,0 +1,10 @@
1
+ from setuptools import setup, find_packages
2
+
3
+ setup(
4
+ name="doc2dict",
5
+ version="0.01",
6
+ packages=find_packages(),
7
+ install_requires=['selectolax'
8
+ ],
9
+ python_requires=">=3.8"
10
+ )