doc2dict 0.2.2__tar.gz → 0.2.4__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,4 +1,4 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: doc2dict
3
- Version: 0.2.2
3
+ Version: 0.2.4
4
4
  Requires-Python: >=3.8
@@ -1,6 +1,6 @@
1
1
  import re
2
2
 
3
- def flatten_hierarchy(content,sep='\n'):
3
+ def flatten_hierarchy(content, sep='\n'):
4
4
  result = []
5
5
 
6
6
  def process_node(node):
@@ -50,12 +50,26 @@ class JSONTransformer:
50
50
 
51
51
  return matches
52
52
 
53
+ def _extract_ref_ids(self, ref_data, search_id):
54
+ """Extract reference IDs from either dict or list data."""
55
+ if isinstance(ref_data, dict):
56
+ ref_id = ref_data.get(search_id)
57
+ return [ref_id] if ref_id is not None else []
58
+ elif isinstance(ref_data, list):
59
+ ids = []
60
+ for item in ref_data:
61
+ if isinstance(item, dict):
62
+ ref_id = item.get(search_id)
63
+ if ref_id is not None:
64
+ ids.append(ref_id)
65
+ return ids
66
+ return []
67
+
53
68
  def _find_content(self, data, match_identifier, match_content):
54
69
  """Find all content entries in the data that match the identifier and content pattern."""
55
70
  matches = []
56
71
 
57
72
  if isinstance(data, dict):
58
- # Check if this dict has both the identifier and content keys
59
73
  if match_identifier in data and match_content in data:
60
74
  matches.append(data)
61
75
  for value in data.values():
@@ -85,18 +99,15 @@ class JSONTransformer:
85
99
  if isinstance(data, dict):
86
100
  id_key = match_rule['identifier']
87
101
 
88
- # If this is a match we used (only need to check id), remove it
89
102
  if id_key in data and data.get(id_key) in self.used_matches:
90
103
  return None
91
104
 
92
- # Process remaining dict entries
93
105
  result = {}
94
106
  for k, v in data.items():
95
107
  processed = self._remove_used_content(v, match_rule)
96
108
  if processed is not None:
97
109
  result[k] = processed
98
110
 
99
- # If dict is empty after processing, remove it
100
111
  return result if result else None
101
112
 
102
113
  elif isinstance(data, list):
@@ -105,7 +116,7 @@ class JSONTransformer:
105
116
  return result if result else None
106
117
 
107
118
  return data
108
-
119
+
109
120
  def _apply_standardization(self, data, transformation):
110
121
  """Apply standardization rules to transform text based on regex pattern."""
111
122
  if isinstance(data, dict):
@@ -117,7 +128,6 @@ class JSONTransformer:
117
128
  output_field = transformation['output'].get('field', 'text')
118
129
  data[output_field] = transformation['output']['format'].format(value.lower())
119
130
 
120
- # Process all nested structures
121
131
  for value in data.values():
122
132
  if isinstance(value, (dict, list)):
123
133
  self._apply_standardization(value, transformation)
@@ -127,7 +137,6 @@ class JSONTransformer:
127
137
  if isinstance(item, (dict, list)):
128
138
  self._apply_standardization(item, transformation)
129
139
 
130
-
131
140
  def _apply_trim(self, data, transformation):
132
141
  if not isinstance(data, dict) or 'content' not in data:
133
142
  return data
@@ -136,7 +145,6 @@ class JSONTransformer:
136
145
  expected = transformation['match'].get('expected')
137
146
  output_type = transformation['output']['type']
138
147
 
139
- # Find matches at any level
140
148
  matches = []
141
149
  def find_matches(content, current_path=[]):
142
150
  for i, item in enumerate(content):
@@ -153,25 +161,21 @@ class JSONTransformer:
153
161
  if not matches:
154
162
  return data
155
163
 
156
- # Group matches by their text to find duplicates
157
164
  text_groups = {}
158
165
  for match in matches:
159
166
  text = match['text']
160
167
  if text not in text_groups:
161
168
  text_groups[text] = []
162
- text_groups[text].append(match['path']) # Now appending path instead of index
169
+ text_groups[text].append(match['path'])
163
170
 
164
- # Find first duplicate pair
165
171
  result = {'type': output_type}
166
172
  for text, paths in text_groups.items():
167
173
  if len(paths) > expected:
168
174
  if expected == 0:
169
- # If expected is 0, everything goes into introduction
170
175
  result['content'] = [flatten_hierarchy(data['content'])]
171
176
  data['content'] = [result]
172
177
  else:
173
- # Take everything before the (expected + 1)th occurrence
174
- split_path = paths[expected] # This is the key change
178
+ split_path = paths[expected]
175
179
  split_idx = split_path[0]
176
180
  before_content = data['content'][:split_idx]
177
181
  result['content'] = [flatten_hierarchy(before_content)]
@@ -192,11 +196,9 @@ class JSONTransformer:
192
196
  if (isinstance(item, dict) and
193
197
  item.get('type') in transformation['match']['types'] and
194
198
  'text' in item):
195
- # If we have a matching previous section
196
199
  if (current_section and
197
200
  current_section['type'] == item['type'] and
198
201
  current_section['text'] == item['text']):
199
- # Merge content
200
202
  current_section['content'].extend(item['content'])
201
203
  else:
202
204
  if current_section:
@@ -213,7 +215,6 @@ class JSONTransformer:
213
215
 
214
216
  data['content'] = new_content
215
217
 
216
- # Process nested structures
217
218
  for value in data.values():
218
219
  if isinstance(value, (dict, list)):
219
220
  self._apply_consecutive_merge(value, transformation)
@@ -245,10 +246,17 @@ class JSONTransformer:
245
246
  refs = self._find_refs(result, search_key)
246
247
 
247
248
  for ref in refs:
248
- ref_id = ref[search_key].get(search_id)
249
- if ref_id in self.id_to_text:
250
- ref[output_key] = self.id_to_text[ref_id]
251
- del ref[search_key]
249
+ ref_ids = self._extract_ref_ids(ref[search_key], search_id)
250
+ if ref_ids:
251
+ # Create a list of referenced content
252
+ referenced_content = [
253
+ self.id_to_text[ref_id]
254
+ for ref_id in ref_ids
255
+ if ref_id in self.id_to_text
256
+ ]
257
+ if referenced_content:
258
+ ref[output_key] = referenced_content
259
+ del ref[search_key]
252
260
 
253
261
  if transformation['match'].get('remove_after_use', False):
254
262
  result = self._remove_used_content(result, transformation['match'])
@@ -282,16 +290,13 @@ class RuleProcessor:
282
290
  if isinstance(item, str):
283
291
  current_strings.append(item)
284
292
  else:
285
- # If we have collected strings, join them and add to result
286
293
  if current_strings:
287
294
  result.append(self.rules.get('join_text').join(current_strings))
288
295
  current_strings = []
289
- # Process nested structure
290
296
  if isinstance(item, dict) and 'content' in item:
291
297
  item['content'] = self._join_consecutive_strings(item['content'])
292
298
  result.append(item)
293
299
 
294
- # Don't forget strings at the end
295
300
  if current_strings:
296
301
  result.append(self.rules.get('join_text').join(current_strings))
297
302
 
@@ -305,10 +310,8 @@ class RuleProcessor:
305
310
  for i in range(start_idx + 1, len(lines)):
306
311
  line = lines[i]
307
312
 
308
- # Check for nested start patterns
309
313
  if pattern_name and re.match(pattern_name, line):
310
314
  nesting_level += 1
311
- # Check for end pattern
312
315
  elif re.match(end_pattern, line):
313
316
  nesting_level -= 1
314
317
  if nesting_level == 0:
@@ -325,7 +328,6 @@ class RuleProcessor:
325
328
  if rule.get('end'):
326
329
  end_idx = self._find_matching_end(lines, start_idx, rule['end'])
327
330
  else:
328
- # If no end pattern, collect until next hierarchy match
329
331
  for i in range(start_idx + 1, len(lines)):
330
332
  if any(re.match(r['pattern'], lines[i])
331
333
  for r in mappings if r.get('hierarchy') is not None):
@@ -334,12 +336,10 @@ class RuleProcessor:
334
336
  if end_idx is None:
335
337
  end_idx = len(lines) - 1
336
338
 
337
- # Process content between start and end
338
339
  while current_idx < end_idx:
339
340
  line = lines[current_idx]
340
341
  matched = False
341
342
 
342
- # Check for nested patterns
343
343
  for nested_rule in mappings:
344
344
  if re.match(nested_rule['pattern'], line):
345
345
  nested_content, next_idx = self._process_block(
@@ -355,7 +355,6 @@ class RuleProcessor:
355
355
  content.append(line)
356
356
  current_idx += 1
357
357
 
358
- # Include end pattern line if keep_end is True
359
358
  if rule.get('keep_end', False) and end_idx < len(lines):
360
359
  content.append(lines[end_idx])
361
360
 
@@ -371,7 +370,6 @@ class RuleProcessor:
371
370
  result = {'content': []}
372
371
  hierarchy_stack = [result]
373
372
 
374
- # Sort mappings by hierarchy level
375
373
  mappings = sorted(
376
374
  self.rules['mappings'],
377
375
  key=lambda x: x.get('hierarchy', float('inf'))
@@ -385,28 +383,23 @@ class RuleProcessor:
385
383
  for rule in mappings:
386
384
  if re.match(rule['pattern'], line):
387
385
  if rule.get('hierarchy') is not None:
388
- # Create new section
389
386
  new_section = {
390
387
  'type': rule['name'],
391
388
  'text': line,
392
389
  'content': []
393
390
  }
394
391
 
395
- # Find correct parent based on hierarchy level
396
392
  while len(hierarchy_stack) > rule['hierarchy'] + 1:
397
393
  hierarchy_stack.pop()
398
394
 
399
- # Add to parent's content
400
395
  parent = hierarchy_stack[-1]
401
396
  if isinstance(parent.get('content'), list):
402
397
  parent['content'].append(new_section)
403
398
 
404
- # Update hierarchy stack
405
399
  hierarchy_stack.append(new_section)
406
400
  i += 1
407
401
 
408
402
  else:
409
- # Handle block with potential nesting
410
403
  block, end_idx = self._process_block(lines, i, rule, mappings)
411
404
  parent = hierarchy_stack[-1]
412
405
  if isinstance(parent.get('content'), list):
@@ -422,7 +415,6 @@ class RuleProcessor:
422
415
  parent['content'].append(line)
423
416
  i += 1
424
417
 
425
- # Join text content if specified
426
418
  if self.rules.get('join_text') is not None:
427
419
  result['content'] = self._join_consecutive_strings(result['content'])
428
420
 
@@ -1,11 +1,20 @@
1
1
  import xmltodict
2
2
  from ..mapping import JSONTransformer
3
3
 
4
+ def remove_namespace(path, key, value):
5
+ # Remove namespace from keys
6
+ if ':' in key:
7
+ # Keep only the part after the last colon
8
+ return key.split(':')[-1], value
9
+ return key, value
10
+
4
11
  def xml2dict(content, mapping_dict=None):
5
- data = xmltodict.parse(content)
6
-
12
+
13
+ data = xmltodict.parse(content,postprocessor=remove_namespace)
14
+
7
15
  if mapping_dict is None:
8
16
  return data
17
+
9
18
 
10
19
  transformer = JSONTransformer(mapping_dict)
11
20
  transformed_data = transformer.transform(data)
@@ -1,4 +1,4 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: doc2dict
3
- Version: 0.2.2
3
+ Version: 0.2.4
4
4
  Requires-Python: >=3.8
@@ -2,7 +2,7 @@ from setuptools import setup, find_packages
2
2
 
3
3
  setup(
4
4
  name="doc2dict",
5
- version="0.2.2",
5
+ version="0.2.4",
6
6
  packages=find_packages(),
7
7
  install_requires=['selectolax','xmltodict'
8
8
  ],
File without changes
File without changes
File without changes