doc2dict 0.2.2__tar.gz → 0.2.4__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {doc2dict-0.2.2 → doc2dict-0.2.4}/PKG-INFO +1 -1
- {doc2dict-0.2.2 → doc2dict-0.2.4}/doc2dict/mapping.py +30 -38
- {doc2dict-0.2.2 → doc2dict-0.2.4}/doc2dict/xml/parser.py +11 -2
- {doc2dict-0.2.2 → doc2dict-0.2.4}/doc2dict.egg-info/PKG-INFO +1 -1
- {doc2dict-0.2.2 → doc2dict-0.2.4}/setup.py +1 -1
- {doc2dict-0.2.2 → doc2dict-0.2.4}/doc2dict/__init__.py +0 -0
- {doc2dict-0.2.2 → doc2dict-0.2.4}/doc2dict/dict2dict.py +0 -0
- {doc2dict-0.2.2 → doc2dict-0.2.4}/doc2dict/html/__init__.py +0 -0
- {doc2dict-0.2.2 → doc2dict-0.2.4}/doc2dict/html/html_parser.py +0 -0
- {doc2dict-0.2.2 → doc2dict-0.2.4}/doc2dict/html/visualizer.py +0 -0
- {doc2dict-0.2.2 → doc2dict-0.2.4}/doc2dict/txt/__init__.py +0 -0
- {doc2dict-0.2.2 → doc2dict-0.2.4}/doc2dict/txt/parser.py +0 -0
- {doc2dict-0.2.2 → doc2dict-0.2.4}/doc2dict/xml/__init__.py +0 -0
- {doc2dict-0.2.2 → doc2dict-0.2.4}/doc2dict.egg-info/SOURCES.txt +0 -0
- {doc2dict-0.2.2 → doc2dict-0.2.4}/doc2dict.egg-info/dependency_links.txt +0 -0
- {doc2dict-0.2.2 → doc2dict-0.2.4}/doc2dict.egg-info/requires.txt +0 -0
- {doc2dict-0.2.2 → doc2dict-0.2.4}/doc2dict.egg-info/top_level.txt +0 -0
- {doc2dict-0.2.2 → doc2dict-0.2.4}/setup.cfg +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import re
|
|
2
2
|
|
|
3
|
-
def flatten_hierarchy(content,sep='\n'):
|
|
3
|
+
def flatten_hierarchy(content, sep='\n'):
|
|
4
4
|
result = []
|
|
5
5
|
|
|
6
6
|
def process_node(node):
|
|
@@ -50,12 +50,26 @@ class JSONTransformer:
|
|
|
50
50
|
|
|
51
51
|
return matches
|
|
52
52
|
|
|
53
|
+
def _extract_ref_ids(self, ref_data, search_id):
|
|
54
|
+
"""Extract reference IDs from either dict or list data."""
|
|
55
|
+
if isinstance(ref_data, dict):
|
|
56
|
+
ref_id = ref_data.get(search_id)
|
|
57
|
+
return [ref_id] if ref_id is not None else []
|
|
58
|
+
elif isinstance(ref_data, list):
|
|
59
|
+
ids = []
|
|
60
|
+
for item in ref_data:
|
|
61
|
+
if isinstance(item, dict):
|
|
62
|
+
ref_id = item.get(search_id)
|
|
63
|
+
if ref_id is not None:
|
|
64
|
+
ids.append(ref_id)
|
|
65
|
+
return ids
|
|
66
|
+
return []
|
|
67
|
+
|
|
53
68
|
def _find_content(self, data, match_identifier, match_content):
|
|
54
69
|
"""Find all content entries in the data that match the identifier and content pattern."""
|
|
55
70
|
matches = []
|
|
56
71
|
|
|
57
72
|
if isinstance(data, dict):
|
|
58
|
-
# Check if this dict has both the identifier and content keys
|
|
59
73
|
if match_identifier in data and match_content in data:
|
|
60
74
|
matches.append(data)
|
|
61
75
|
for value in data.values():
|
|
@@ -85,18 +99,15 @@ class JSONTransformer:
|
|
|
85
99
|
if isinstance(data, dict):
|
|
86
100
|
id_key = match_rule['identifier']
|
|
87
101
|
|
|
88
|
-
# If this is a match we used (only need to check id), remove it
|
|
89
102
|
if id_key in data and data.get(id_key) in self.used_matches:
|
|
90
103
|
return None
|
|
91
104
|
|
|
92
|
-
# Process remaining dict entries
|
|
93
105
|
result = {}
|
|
94
106
|
for k, v in data.items():
|
|
95
107
|
processed = self._remove_used_content(v, match_rule)
|
|
96
108
|
if processed is not None:
|
|
97
109
|
result[k] = processed
|
|
98
110
|
|
|
99
|
-
# If dict is empty after processing, remove it
|
|
100
111
|
return result if result else None
|
|
101
112
|
|
|
102
113
|
elif isinstance(data, list):
|
|
@@ -105,7 +116,7 @@ class JSONTransformer:
|
|
|
105
116
|
return result if result else None
|
|
106
117
|
|
|
107
118
|
return data
|
|
108
|
-
|
|
119
|
+
|
|
109
120
|
def _apply_standardization(self, data, transformation):
|
|
110
121
|
"""Apply standardization rules to transform text based on regex pattern."""
|
|
111
122
|
if isinstance(data, dict):
|
|
@@ -117,7 +128,6 @@ class JSONTransformer:
|
|
|
117
128
|
output_field = transformation['output'].get('field', 'text')
|
|
118
129
|
data[output_field] = transformation['output']['format'].format(value.lower())
|
|
119
130
|
|
|
120
|
-
# Process all nested structures
|
|
121
131
|
for value in data.values():
|
|
122
132
|
if isinstance(value, (dict, list)):
|
|
123
133
|
self._apply_standardization(value, transformation)
|
|
@@ -127,7 +137,6 @@ class JSONTransformer:
|
|
|
127
137
|
if isinstance(item, (dict, list)):
|
|
128
138
|
self._apply_standardization(item, transformation)
|
|
129
139
|
|
|
130
|
-
|
|
131
140
|
def _apply_trim(self, data, transformation):
|
|
132
141
|
if not isinstance(data, dict) or 'content' not in data:
|
|
133
142
|
return data
|
|
@@ -136,7 +145,6 @@ class JSONTransformer:
|
|
|
136
145
|
expected = transformation['match'].get('expected')
|
|
137
146
|
output_type = transformation['output']['type']
|
|
138
147
|
|
|
139
|
-
# Find matches at any level
|
|
140
148
|
matches = []
|
|
141
149
|
def find_matches(content, current_path=[]):
|
|
142
150
|
for i, item in enumerate(content):
|
|
@@ -153,25 +161,21 @@ class JSONTransformer:
|
|
|
153
161
|
if not matches:
|
|
154
162
|
return data
|
|
155
163
|
|
|
156
|
-
# Group matches by their text to find duplicates
|
|
157
164
|
text_groups = {}
|
|
158
165
|
for match in matches:
|
|
159
166
|
text = match['text']
|
|
160
167
|
if text not in text_groups:
|
|
161
168
|
text_groups[text] = []
|
|
162
|
-
text_groups[text].append(match['path'])
|
|
169
|
+
text_groups[text].append(match['path'])
|
|
163
170
|
|
|
164
|
-
# Find first duplicate pair
|
|
165
171
|
result = {'type': output_type}
|
|
166
172
|
for text, paths in text_groups.items():
|
|
167
173
|
if len(paths) > expected:
|
|
168
174
|
if expected == 0:
|
|
169
|
-
# If expected is 0, everything goes into introduction
|
|
170
175
|
result['content'] = [flatten_hierarchy(data['content'])]
|
|
171
176
|
data['content'] = [result]
|
|
172
177
|
else:
|
|
173
|
-
|
|
174
|
-
split_path = paths[expected] # This is the key change
|
|
178
|
+
split_path = paths[expected]
|
|
175
179
|
split_idx = split_path[0]
|
|
176
180
|
before_content = data['content'][:split_idx]
|
|
177
181
|
result['content'] = [flatten_hierarchy(before_content)]
|
|
@@ -192,11 +196,9 @@ class JSONTransformer:
|
|
|
192
196
|
if (isinstance(item, dict) and
|
|
193
197
|
item.get('type') in transformation['match']['types'] and
|
|
194
198
|
'text' in item):
|
|
195
|
-
# If we have a matching previous section
|
|
196
199
|
if (current_section and
|
|
197
200
|
current_section['type'] == item['type'] and
|
|
198
201
|
current_section['text'] == item['text']):
|
|
199
|
-
# Merge content
|
|
200
202
|
current_section['content'].extend(item['content'])
|
|
201
203
|
else:
|
|
202
204
|
if current_section:
|
|
@@ -213,7 +215,6 @@ class JSONTransformer:
|
|
|
213
215
|
|
|
214
216
|
data['content'] = new_content
|
|
215
217
|
|
|
216
|
-
# Process nested structures
|
|
217
218
|
for value in data.values():
|
|
218
219
|
if isinstance(value, (dict, list)):
|
|
219
220
|
self._apply_consecutive_merge(value, transformation)
|
|
@@ -245,10 +246,17 @@ class JSONTransformer:
|
|
|
245
246
|
refs = self._find_refs(result, search_key)
|
|
246
247
|
|
|
247
248
|
for ref in refs:
|
|
248
|
-
|
|
249
|
-
if
|
|
250
|
-
|
|
251
|
-
|
|
249
|
+
ref_ids = self._extract_ref_ids(ref[search_key], search_id)
|
|
250
|
+
if ref_ids:
|
|
251
|
+
# Create a list of referenced content
|
|
252
|
+
referenced_content = [
|
|
253
|
+
self.id_to_text[ref_id]
|
|
254
|
+
for ref_id in ref_ids
|
|
255
|
+
if ref_id in self.id_to_text
|
|
256
|
+
]
|
|
257
|
+
if referenced_content:
|
|
258
|
+
ref[output_key] = referenced_content
|
|
259
|
+
del ref[search_key]
|
|
252
260
|
|
|
253
261
|
if transformation['match'].get('remove_after_use', False):
|
|
254
262
|
result = self._remove_used_content(result, transformation['match'])
|
|
@@ -282,16 +290,13 @@ class RuleProcessor:
|
|
|
282
290
|
if isinstance(item, str):
|
|
283
291
|
current_strings.append(item)
|
|
284
292
|
else:
|
|
285
|
-
# If we have collected strings, join them and add to result
|
|
286
293
|
if current_strings:
|
|
287
294
|
result.append(self.rules.get('join_text').join(current_strings))
|
|
288
295
|
current_strings = []
|
|
289
|
-
# Process nested structure
|
|
290
296
|
if isinstance(item, dict) and 'content' in item:
|
|
291
297
|
item['content'] = self._join_consecutive_strings(item['content'])
|
|
292
298
|
result.append(item)
|
|
293
299
|
|
|
294
|
-
# Don't forget strings at the end
|
|
295
300
|
if current_strings:
|
|
296
301
|
result.append(self.rules.get('join_text').join(current_strings))
|
|
297
302
|
|
|
@@ -305,10 +310,8 @@ class RuleProcessor:
|
|
|
305
310
|
for i in range(start_idx + 1, len(lines)):
|
|
306
311
|
line = lines[i]
|
|
307
312
|
|
|
308
|
-
# Check for nested start patterns
|
|
309
313
|
if pattern_name and re.match(pattern_name, line):
|
|
310
314
|
nesting_level += 1
|
|
311
|
-
# Check for end pattern
|
|
312
315
|
elif re.match(end_pattern, line):
|
|
313
316
|
nesting_level -= 1
|
|
314
317
|
if nesting_level == 0:
|
|
@@ -325,7 +328,6 @@ class RuleProcessor:
|
|
|
325
328
|
if rule.get('end'):
|
|
326
329
|
end_idx = self._find_matching_end(lines, start_idx, rule['end'])
|
|
327
330
|
else:
|
|
328
|
-
# If no end pattern, collect until next hierarchy match
|
|
329
331
|
for i in range(start_idx + 1, len(lines)):
|
|
330
332
|
if any(re.match(r['pattern'], lines[i])
|
|
331
333
|
for r in mappings if r.get('hierarchy') is not None):
|
|
@@ -334,12 +336,10 @@ class RuleProcessor:
|
|
|
334
336
|
if end_idx is None:
|
|
335
337
|
end_idx = len(lines) - 1
|
|
336
338
|
|
|
337
|
-
# Process content between start and end
|
|
338
339
|
while current_idx < end_idx:
|
|
339
340
|
line = lines[current_idx]
|
|
340
341
|
matched = False
|
|
341
342
|
|
|
342
|
-
# Check for nested patterns
|
|
343
343
|
for nested_rule in mappings:
|
|
344
344
|
if re.match(nested_rule['pattern'], line):
|
|
345
345
|
nested_content, next_idx = self._process_block(
|
|
@@ -355,7 +355,6 @@ class RuleProcessor:
|
|
|
355
355
|
content.append(line)
|
|
356
356
|
current_idx += 1
|
|
357
357
|
|
|
358
|
-
# Include end pattern line if keep_end is True
|
|
359
358
|
if rule.get('keep_end', False) and end_idx < len(lines):
|
|
360
359
|
content.append(lines[end_idx])
|
|
361
360
|
|
|
@@ -371,7 +370,6 @@ class RuleProcessor:
|
|
|
371
370
|
result = {'content': []}
|
|
372
371
|
hierarchy_stack = [result]
|
|
373
372
|
|
|
374
|
-
# Sort mappings by hierarchy level
|
|
375
373
|
mappings = sorted(
|
|
376
374
|
self.rules['mappings'],
|
|
377
375
|
key=lambda x: x.get('hierarchy', float('inf'))
|
|
@@ -385,28 +383,23 @@ class RuleProcessor:
|
|
|
385
383
|
for rule in mappings:
|
|
386
384
|
if re.match(rule['pattern'], line):
|
|
387
385
|
if rule.get('hierarchy') is not None:
|
|
388
|
-
# Create new section
|
|
389
386
|
new_section = {
|
|
390
387
|
'type': rule['name'],
|
|
391
388
|
'text': line,
|
|
392
389
|
'content': []
|
|
393
390
|
}
|
|
394
391
|
|
|
395
|
-
# Find correct parent based on hierarchy level
|
|
396
392
|
while len(hierarchy_stack) > rule['hierarchy'] + 1:
|
|
397
393
|
hierarchy_stack.pop()
|
|
398
394
|
|
|
399
|
-
# Add to parent's content
|
|
400
395
|
parent = hierarchy_stack[-1]
|
|
401
396
|
if isinstance(parent.get('content'), list):
|
|
402
397
|
parent['content'].append(new_section)
|
|
403
398
|
|
|
404
|
-
# Update hierarchy stack
|
|
405
399
|
hierarchy_stack.append(new_section)
|
|
406
400
|
i += 1
|
|
407
401
|
|
|
408
402
|
else:
|
|
409
|
-
# Handle block with potential nesting
|
|
410
403
|
block, end_idx = self._process_block(lines, i, rule, mappings)
|
|
411
404
|
parent = hierarchy_stack[-1]
|
|
412
405
|
if isinstance(parent.get('content'), list):
|
|
@@ -422,7 +415,6 @@ class RuleProcessor:
|
|
|
422
415
|
parent['content'].append(line)
|
|
423
416
|
i += 1
|
|
424
417
|
|
|
425
|
-
# Join text content if specified
|
|
426
418
|
if self.rules.get('join_text') is not None:
|
|
427
419
|
result['content'] = self._join_consecutive_strings(result['content'])
|
|
428
420
|
|
|
@@ -1,11 +1,20 @@
|
|
|
1
1
|
import xmltodict
|
|
2
2
|
from ..mapping import JSONTransformer
|
|
3
3
|
|
|
4
|
+
def remove_namespace(path, key, value):
|
|
5
|
+
# Remove namespace from keys
|
|
6
|
+
if ':' in key:
|
|
7
|
+
# Keep only the part after the last colon
|
|
8
|
+
return key.split(':')[-1], value
|
|
9
|
+
return key, value
|
|
10
|
+
|
|
4
11
|
def xml2dict(content, mapping_dict=None):
|
|
5
|
-
|
|
6
|
-
|
|
12
|
+
|
|
13
|
+
data = xmltodict.parse(content,postprocessor=remove_namespace)
|
|
14
|
+
|
|
7
15
|
if mapping_dict is None:
|
|
8
16
|
return data
|
|
17
|
+
|
|
9
18
|
|
|
10
19
|
transformer = JSONTransformer(mapping_dict)
|
|
11
20
|
transformed_data = transformer.transform(data)
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|