python-table-processor 0.3.3__py3-none-any.whl → 0.3.5__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: python-table-processor
3
- Version: 0.3.3
3
+ Version: 0.3.5
4
4
  Summary: A table data processor
5
5
  Home-page: https://github.com/akivajp/python-table-processor
6
6
  License: MIT
@@ -1,23 +1,23 @@
1
- table_processor/__init__.py,sha256=IqYTIOR96kwEp7nfLi1h2e7pZiFKmLonFJ-wwSegFkw,52
1
+ table_processor/__init__.py,sha256=gtxdmTbOklsCYOkQEth0hoRHsSwKt1ixChUIQPBOStA,52
2
2
  table_processor/cli.py,sha256=uluf_VsBzq3Ldxh3ePGseSrv8--lI7huu-6RjDaHPPg,2170
3
- table_processor/commands/convert_tables.py,sha256=j25KL0zRl5ikrBk73jAkKn4rPmDq7SecChZxecDpILQ,2123
3
+ table_processor/commands/convert_tables.py,sha256=w4IZftSlrbt-BMPjldULT_616LSiNE3h7GmrFF7K5OM,2276
4
4
  table_processor/commands/merge_tables.py,sha256=Uu-hjfPh69Of0_OZrY8sY0oscpLSdK4jwJ_t1qgfPMg,1485
5
- table_processor/core/actions.py,sha256=WLlq-G8Pb9xjsLpXK1QVCwhK20ijz2oIbTFibihdFTE,21557
5
+ table_processor/core/actions.py,sha256=GKUKzoPxUEuWSeuY-HsqFoS5Bdm99bjkRhuBgG1s_uk,23658
6
6
  table_processor/core/config.py,sha256=Z1s2r3y4oQXdbDCwIY_hYzoq6VRv7Sb3oUlfBCu13Lc,11646
7
7
  table_processor/core/constants.py,sha256=w-1TxtpXywb6Qh8M4YWuzcCHUvCjy5zJYxaXC3RQS6A,158
8
- table_processor/core/convert.py,sha256=cibzqJ5hHsfX_7qUz4kE2Gi-WH4UMqecBl3kSb4IVsY,11140
8
+ table_processor/core/convert.py,sha256=dzGcpBwJ5omiE8Pw-VmHnGoOqDyh1Rip6BiLdvdf3V8,11561
9
9
  table_processor/core/functions/assign_id.py,sha256=dA-wgZA7E01gbx8qGvo1PeMd3aq9PhMdyITzr3BKXIE,4204
10
10
  table_processor/core/functions/flatten_row.py,sha256=2l-s4YXgBM58IthpfVzd7L4EGUCekSi7ss5CaM_3MUU,686
11
- table_processor/core/functions/get_nested_field_value.py,sha256=vF4pH6nxR3-2bDfAvrMfHcPFNiS4MnGkXL8FdcScdIw,685
11
+ table_processor/core/functions/get_nested_field_value.py,sha256=ZEJi8XtDONoYmhPoqYpVZRGtEWvsfAksV4FXtkMAaq8,946
12
12
  table_processor/core/functions/nest_row.py,sha256=b5AKfxE39g-NOLranQRy0dfTZq-27Or8ZlgCB8N5kFs,640
13
13
  table_processor/core/functions/search_column_value.py,sha256=GscwLaMTORHFDhdCdjzdoy46UGyFHrGwDQA2b_tZhsQ,740
14
14
  table_processor/core/functions/set_flat_field_value.py,sha256=7-BgD2yEUGyBaFaIJ18bYv82Qdy0CN8NiC5ZtsAdv44,531
15
15
  table_processor/core/functions/set_nested_field_value.py,sha256=csO0v_nW94q-M9bNL91xR-TLHPXDDxnLK8gN-0LLd6k,595
16
16
  table_processor/core/functions/set_row_value.py,sha256=wlF_nVGZY74XCzd6pBJJyti62m6y0s1kc5Q6uq9jn8o,641
17
- table_processor/core/merge.py,sha256=Yc5OV9n1xV9QtgY4srl6Susd5cOKbuYseEtYuBampDw,4672
18
- table_processor/core/types.py,sha256=qASw6zDJXFpaswKjhSQ39wcTH8BcCmp7xKW_hgP-RRY,2665
19
- python_table_processor-0.3.3.dist-info/LICENSE,sha256=1_Gn0I1neLPxDLfLiHEyxjDg-pbAra1p2iHEqutGS_c,1068
20
- python_table_processor-0.3.3.dist-info/METADATA,sha256=mvhOnT4vISkKBt5HCEecLCTIRrwrWivJyn63II1w5Jk,1098
21
- python_table_processor-0.3.3.dist-info/WHEEL,sha256=sP946D7jFCHeNz5Iq4fL4Lu-PrWrFsgfLXbbkciIZwg,88
22
- python_table_processor-0.3.3.dist-info/entry_points.txt,sha256=lq97F6m51yZ93qmVIqkUHJfbILZuYLxrMWXSqrP4JpM,118
23
- python_table_processor-0.3.3.dist-info/RECORD,,
17
+ table_processor/core/merge.py,sha256=NoxVfhpOr7oxfmMZt1Xqx94fkIhAJQcYuYzR3XcJsns,4809
18
+ table_processor/core/types.py,sha256=xNvy1vL1ei87kt-MgSfRJSllYrNKRo23pHtJA15DyLQ,2895
19
+ python_table_processor-0.3.5.dist-info/LICENSE,sha256=1_Gn0I1neLPxDLfLiHEyxjDg-pbAra1p2iHEqutGS_c,1068
20
+ python_table_processor-0.3.5.dist-info/METADATA,sha256=MA1YvRPDWpWhbua0uFd5AOyFJ7EtxUYJ9XV3bYyr2S8,1098
21
+ python_table_processor-0.3.5.dist-info/WHEEL,sha256=sP946D7jFCHeNz5Iq4fL4Lu-PrWrFsgfLXbbkciIZwg,88
22
+ python_table_processor-0.3.5.dist-info/entry_points.txt,sha256=lq97F6m51yZ93qmVIqkUHJfbILZuYLxrMWXSqrP4JpM,118
23
+ python_table_processor-0.3.5.dist-info/RECORD,,
@@ -1,2 +1,2 @@
1
- __version__ = "0.3.3"
2
- __version_tuple__ = (0, 3, 3)
1
+ __version__ = "0.3.5"
2
+ __version_tuple__ = (0, 3, 5)
@@ -20,6 +20,7 @@ def run(
20
20
  output_debug = args.output_debug,
21
21
  verbose = args.verbose,
22
22
  ignore_file_rows = args.ignore_file_rows,
23
+ skip_header = args.skip_header,
23
24
  )
24
25
 
25
26
  def setup_parser(
@@ -77,4 +78,9 @@ def setup_parser(
77
78
  action='store_true',
78
79
  help='Output debug information',
79
80
  )
81
+ parser.add_argument(
82
+ '--skip-header',
83
+ action='store_true',
84
+ help='Skip header',
85
+ )
80
86
  parser.set_defaults(handler=run)
@@ -29,6 +29,7 @@ from . types import (
29
29
  AssignFormatConfig,
30
30
  AssignIdConfig,
31
31
  AssignLengthConfig,
32
+ CastConfig,
32
33
  FilterConfig,
33
34
  GlobalStatus,
34
35
  JoinConfig,
@@ -147,6 +148,31 @@ def setup_actions_with_args(
147
148
  source = source,
148
149
  ))
149
150
  continue
151
+ if action_name == 'cast':
152
+ required = options.get('required', False)
153
+ as_type = options.get('as', 'literal')
154
+ if as_type in ['boolean']:
155
+ as_type = 'bool'
156
+ if as_type not in ['bool', 'int', 'float', 'str']:
157
+ raise ValueError(
158
+ f'Unsupported as type: {as_type}'
159
+ )
160
+ assign_default = False
161
+ default_value = None
162
+ if 'default' in options:
163
+ assign_default = True
164
+ default_value = options['default']
165
+ if default_value in ['None', 'none', 'Null', 'null']:
166
+ default_value = None
167
+ config.actions.append(CastConfig(
168
+ target = target,
169
+ source = source,
170
+ as_type = as_type,
171
+ required = required,
172
+ assign_default = assign_default,
173
+ default_value = default_value,
174
+ ))
175
+ continue
150
176
  if action_name == 'filter-empty':
151
177
  config.actions.append(FilterConfig(
152
178
  field = target,
@@ -323,6 +349,8 @@ def do_action(
323
349
  return assign_id(status.id_context_map, row, action)
324
350
  if isinstance(action, AssignLengthConfig):
325
351
  return assign_length(row, action)
352
+ if isinstance(action, CastConfig):
353
+ return cast(row, action)
326
354
  if isinstance(action, FilterConfig):
327
355
  if filter_row(row, action):
328
356
  return row
@@ -685,3 +713,37 @@ def assign_length(
685
713
  if found:
686
714
  set_row_staging_value(row, config.target, len(value))
687
715
  return row
716
+
717
+ def cast(
718
+ row: Row,
719
+ config: CastConfig,
720
+ ):
721
+ value, found = search_column_value(row.nested, config.source)
722
+ if config.required:
723
+ if not found:
724
+ raise ValueError(
725
+ f'Required field not found, field: {config.source}'
726
+ )
727
+ if config.as_type == 'bool':
728
+ cast_func = bool
729
+ elif config.as_type == 'int':
730
+ cast_func = int
731
+ elif config.as_type == 'float':
732
+ cast_func = float
733
+ elif config.as_type == 'str':
734
+ cast_func = str
735
+ else:
736
+ raise ValueError(
737
+ f'Unsupported as type: {config.as_type}'
738
+ )
739
+ try:
740
+ casted = cast_func(value)
741
+ except:
742
+ if config.assign_default:
743
+ casted = config.default_value
744
+ else:
745
+ raise ValueError(
746
+ f'Failed to cast: {value}'
747
+ )
748
+ set_row_staging_value(row, config.target, casted)
749
+ return row
@@ -64,12 +64,13 @@ def register_loader(
64
64
 
65
65
  def load(
66
66
  input_file: str,
67
+ **kwargs,
67
68
  ):
68
69
  ext = os.path.splitext(input_file)[1]
69
70
  if ext not in dict_loaders:
70
71
  raise ValueError(f'Unsupported file type: {ext}')
71
72
  loader = dict_loaders[ext]
72
- return loader(input_file)
73
+ return loader(input_file, **kwargs)
73
74
 
74
75
  dict_savers: dict[str, callable] = {}
75
76
  def register_saver(
@@ -93,16 +94,34 @@ def save(
93
94
  @register_loader('.csv')
94
95
  def load_csv(
95
96
  input_file: str,
97
+ **kwargs,
96
98
  ):
99
+ skip_header = kwargs.get('skip_header', False)
97
100
  # utf-8
98
101
  #df = pd.read_csv(input_file)
99
102
  # UTF-8 with BOM
100
- df = pd.read_csv(input_file, encoding='utf-8-sig')
103
+ if skip_header:
104
+ df = pd.read_csv(
105
+ input_file,
106
+ encoding='utf-8-sig',
107
+ header=None,
108
+ )
109
+ #new_column_names = [f'__values__.{i}' for i in df.columns]
110
+ new_column_names = [f'{i}' for i in df.columns]
111
+ df = df.rename(columns=dict(
112
+ zip(df.columns, new_column_names)
113
+ ))
114
+ else:
115
+ df = pd.read_csv(
116
+ input_file,
117
+ encoding='utf-8-sig',
118
+ )
101
119
  return df
102
120
 
103
121
  @register_loader('.xlsx')
104
122
  def load_excel(
105
123
  input_file: str,
124
+ **kwargs,
106
125
  ):
107
126
  #df = pd.read_excel(input_file)
108
127
  # NOTE: Excelで勝手に日時データなどに変換されてしまうことを防ぐため
@@ -123,6 +142,7 @@ def load_excel(
123
142
  @register_loader('.json')
124
143
  def load_json(
125
144
  input_file: str,
145
+ **kiwargs,
126
146
  ):
127
147
  with open(input_file, 'r') as f:
128
148
  data = json.load(f)
@@ -165,6 +185,7 @@ def save_json(
165
185
  @register_loader('.jsonl')
166
186
  def load_jsonl(
167
187
  input_file: str,
188
+ **kwargs,
168
189
  ):
169
190
  rows = []
170
191
  with open(input_file, 'r') as f:
@@ -274,6 +295,7 @@ def convert(
274
295
  action_delimiter: str = ':',
275
296
  verbose: bool = False,
276
297
  ignore_file_rows: list[str] | None = None,
298
+ skip_header: bool = False,
277
299
  ):
278
300
  ic.enable()
279
301
  ic()
@@ -306,11 +328,7 @@ def convert(
306
328
  if not os.path.exists(input_file):
307
329
  raise FileNotFoundError(f'File not found: {input_file}')
308
330
  base_name = os.path.basename(input_file)
309
- ext = os.path.splitext(input_file)[1]
310
- ic(ext)
311
- if ext not in dict_loaders:
312
- raise ValueError(f'Unsupported file type: {ext}')
313
- df = dict_loaders[ext](input_file)
331
+ df = load(input_file, skip_header=skip_header)
314
332
  # NOTE: NaN を None に変換しておかないと厄介
315
333
  df = df.replace([np.nan], [None])
316
334
  #ic(df)
@@ -15,11 +15,17 @@ def get_nested_field_value(
15
15
  if index < len(data):
16
16
  return data[index], True
17
17
  return None, False
18
+ if '.' in field:
19
+ field, rest = field.split('.', 1)
20
+ if field.isdigit():
21
+ index = int(field)
22
+ if index < len(data):
23
+ return get_nested_field_value(data[index], rest)
18
24
  if isinstance(data, dict):
19
25
  if field in data:
20
26
  return data[field], True
21
- if '.' in field:
22
- field, rest = field.split('.', 1)
23
- if field in data:
24
- return get_nested_field_value(data[field], rest)
27
+ if '.' in field:
28
+ field, rest = field.split('.', 1)
29
+ if field in data:
30
+ return get_nested_field_value(data[field], rest)
25
31
  return None, False
@@ -116,17 +116,20 @@ def merge(
116
116
  # dict_key_to_row[key].flat.update(row.flat)
117
117
  if primary_key not in dict_key_to_row:
118
118
  if ignore_not_found:
119
+ ic(primary_key)
120
+ ic(row.flat['__staging__.__file_row_index__'])
119
121
  list_ignored_keys.append(primary_key)
120
122
  continue
121
123
  ic(index)
122
- raise ValueError(f'Key not found: {key}')
124
+ raise ValueError(f'Key not found: {primary_key}')
123
125
  previous_row = dict_key_to_row[primary_key]
124
126
  #ic(previous_row)
125
127
  #ic(previous_row.flat)
126
128
  for key, value in row.flat.items():
127
129
  if key.startswith('__staging__.'):
128
130
  continue
129
- ic(key, value)
131
+ #ic(key)
132
+ #ic(key, value)
130
133
  set_row_value(previous_row, key, value)
131
134
  #set_row_value(previous_row, '指示追従性?', 'test')
132
135
  #set_row_value(previous_row, 'modified', True)
@@ -51,6 +51,17 @@ class AssignLengthConfig:
51
51
  target: str
52
52
  source: str
53
53
 
54
+ @dataclasses.dataclass
55
+ class CastConfig:
56
+ target: str
57
+ source: str
58
+ as_type: Literal[
59
+ 'bool', 'int', 'float', 'str'
60
+ ]
61
+ required: bool = False
62
+ assign_default: bool = False
63
+ default_value: Any = None
64
+
54
65
  @dataclasses.dataclass
55
66
  class FilterConfig:
56
67
  field: str