tabpro 0.4.2__tar.gz → 0.4.4__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {tabpro-0.4.2 → tabpro-0.4.4}/PKG-INFO +3 -1
- {tabpro-0.4.2 → tabpro-0.4.4}/pyproject.toml +3 -1
- tabpro-0.4.4/tabpro/__init__.py +14 -0
- {tabpro-0.4.2 → tabpro-0.4.4}/tabpro/commands/convert_tables.py +3 -3
- {tabpro-0.4.2 → tabpro-0.4.4}/tabpro/core/actions.py +3 -1
- tabpro-0.4.4/tabpro/core/classes/row.py +83 -0
- {tabpro-0.4.2 → tabpro-0.4.4}/tabpro/core/convert.py +84 -58
- tabpro-0.4.4/tabpro/core/io/__init__.py +18 -0
- tabpro-0.4.4/tabpro/core/io/extensions/io_csv.py +104 -0
- tabpro-0.4.4/tabpro/core/io/extensions/io_excel.py +66 -0
- tabpro-0.4.4/tabpro/core/io/extensions/io_json.py +57 -0
- tabpro-0.4.4/tabpro/core/io/extensions/io_jsonl.py +64 -0
- tabpro-0.4.2/tabpro/core/io/loader.py → tabpro-0.4.4/tabpro/core/io/extensions/manage_loaders.py +3 -3
- tabpro-0.4.4/tabpro/core/io/extensions/manage_writers.py +43 -0
- tabpro-0.4.4/tabpro/core/io/loader.py +74 -0
- tabpro-0.4.4/tabpro/core/io/writer.py +110 -0
- {tabpro-0.4.2 → tabpro-0.4.4}/tabpro/core/merge.py +20 -49
- tabpro-0.4.4/tabpro/core/progress.py +86 -0
- tabpro-0.4.2/tabpro/__init__.py +0 -7
- tabpro-0.4.2/tabpro/core/io/__init__.py +0 -12
- tabpro-0.4.2/tabpro/core/io/io_csv.py +0 -41
- tabpro-0.4.2/tabpro/core/io/io_excel.py +0 -45
- tabpro-0.4.2/tabpro/core/io/io_json.py +0 -86
- tabpro-0.4.2/tabpro/core/io/saver.py +0 -31
- {tabpro-0.4.2 → tabpro-0.4.4}/LICENSE +0 -0
- {tabpro-0.4.2 → tabpro-0.4.4}/README.md +0 -0
- {tabpro-0.4.2 → tabpro-0.4.4}/tabpro/cli.py +0 -0
- {tabpro-0.4.2 → tabpro-0.4.4}/tabpro/commands/merge_tables.py +0 -0
- {tabpro-0.4.2 → tabpro-0.4.4}/tabpro/core/config.py +0 -0
- {tabpro-0.4.2 → tabpro-0.4.4}/tabpro/core/constants.py +0 -0
- {tabpro-0.4.2 → tabpro-0.4.4}/tabpro/core/functions/assign_id.py +0 -0
- {tabpro-0.4.2 → tabpro-0.4.4}/tabpro/core/functions/flatten_row.py +0 -0
- {tabpro-0.4.2 → tabpro-0.4.4}/tabpro/core/functions/get_nested_field_value.py +0 -0
- {tabpro-0.4.2 → tabpro-0.4.4}/tabpro/core/functions/nest_row.py +0 -0
- {tabpro-0.4.2 → tabpro-0.4.4}/tabpro/core/functions/search_column_value.py +0 -0
- {tabpro-0.4.2 → tabpro-0.4.4}/tabpro/core/functions/set_flat_field_value.py +0 -0
- {tabpro-0.4.2 → tabpro-0.4.4}/tabpro/core/functions/set_nested_field_value.py +0 -0
- {tabpro-0.4.2 → tabpro-0.4.4}/tabpro/core/functions/set_row_value.py +0 -0
- {tabpro-0.4.2 → tabpro-0.4.4}/tabpro/core/types.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.1
|
|
2
2
|
Name: tabpro
|
|
3
|
-
Version: 0.4.
|
|
3
|
+
Version: 0.4.4
|
|
4
4
|
Summary: A table data processor
|
|
5
5
|
Home-page: https://github.com/akivajp/tabpro
|
|
6
6
|
License: MIT
|
|
@@ -16,7 +16,9 @@ Requires-Dist: icecream (>=2.1.3,<3.0.0)
|
|
|
16
16
|
Requires-Dist: logzero (>=1.7.0,<2.0.0)
|
|
17
17
|
Requires-Dist: openpyxl (>=3.1.5,<4.0.0)
|
|
18
18
|
Requires-Dist: pandas (>=2.2.3,<3.0.0)
|
|
19
|
+
Requires-Dist: pytest (>=8.3.5,<9.0.0)
|
|
19
20
|
Requires-Dist: pyyaml (>=6.0.2,<7.0.0)
|
|
21
|
+
Requires-Dist: rich (>=13.9.4,<14.0.0)
|
|
20
22
|
Requires-Dist: tqdm (>=4.67.0,<5.0.0)
|
|
21
23
|
Requires-Dist: xlsxwriter (>=3.2.0,<4.0.0)
|
|
22
24
|
Project-URL: Repository, https://github.com/akivajp/tabpro
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[tool.poetry]
|
|
2
2
|
name = "tabpro"
|
|
3
|
-
version = "0.4.
|
|
3
|
+
version = "0.4.4"
|
|
4
4
|
description = "A table data processor"
|
|
5
5
|
authors = ["Akiva Miura <akiva.miura@gmail.com>"]
|
|
6
6
|
license = "MIT"
|
|
@@ -30,6 +30,8 @@ pandas = "^2.2.3"
|
|
|
30
30
|
openpyxl = "^3.1.5"
|
|
31
31
|
pyyaml = "^6.0.2"
|
|
32
32
|
xlsxwriter = "^3.2.0"
|
|
33
|
+
pytest = "^8.3.5"
|
|
34
|
+
rich = "^13.9.4"
|
|
33
35
|
|
|
34
36
|
|
|
35
37
|
[build-system]
|
|
@@ -20,7 +20,7 @@ def run(
|
|
|
20
20
|
output_debug = args.output_debug,
|
|
21
21
|
verbose = args.verbose,
|
|
22
22
|
ignore_file_rows = args.ignore_file_rows,
|
|
23
|
-
|
|
23
|
+
no_header = args.no_header,
|
|
24
24
|
)
|
|
25
25
|
|
|
26
26
|
def setup_parser(
|
|
@@ -79,8 +79,8 @@ def setup_parser(
|
|
|
79
79
|
help='Output debug information',
|
|
80
80
|
)
|
|
81
81
|
parser.add_argument(
|
|
82
|
-
'--
|
|
82
|
+
'--no-header',
|
|
83
83
|
action='store_true',
|
|
84
|
-
help='
|
|
84
|
+
help='CSV/TSV like data without header row',
|
|
85
85
|
)
|
|
86
86
|
parser.set_defaults(handler=run)
|
|
@@ -38,9 +38,11 @@ from . types import (
|
|
|
38
38
|
PickConfig,
|
|
39
39
|
PushConfig,
|
|
40
40
|
SplitConfig,
|
|
41
|
-
Row,
|
|
41
|
+
#Row,
|
|
42
42
|
)
|
|
43
43
|
|
|
44
|
+
from . classes.row import Row
|
|
45
|
+
|
|
44
46
|
from . functions.assign_id import assign_id
|
|
45
47
|
from . functions.flatten_row import flatten_row
|
|
46
48
|
from . functions.nest_row import nest_row
|
|
@@ -0,0 +1,83 @@
|
|
|
1
|
+
'''
|
|
2
|
+
Row class
|
|
3
|
+
'''
|
|
4
|
+
|
|
5
|
+
from collections import (
|
|
6
|
+
OrderedDict
|
|
7
|
+
)
|
|
8
|
+
|
|
9
|
+
from typing import (
|
|
10
|
+
Mapping,
|
|
11
|
+
)
|
|
12
|
+
|
|
13
|
+
from .. functions.get_nested_field_value import get_nested_field_value
|
|
14
|
+
from .. functions.set_nested_field_value import set_nested_field_value
|
|
15
|
+
from .. functions.set_flat_field_value import set_flat_field_value
|
|
16
|
+
|
|
17
|
+
class Row(Mapping):
|
|
18
|
+
def __init__(
|
|
19
|
+
self,
|
|
20
|
+
):
|
|
21
|
+
self.flat = OrderedDict()
|
|
22
|
+
self.nested = OrderedDict()
|
|
23
|
+
self._prefix: str | None = None
|
|
24
|
+
self._staging: Row | None = None
|
|
25
|
+
|
|
26
|
+
@property
|
|
27
|
+
def staging(self):
|
|
28
|
+
if self._staging is None:
|
|
29
|
+
row = Row()
|
|
30
|
+
row.flat = self.flat
|
|
31
|
+
row.nested = self.nested
|
|
32
|
+
row._prefix = '__staging__'
|
|
33
|
+
self._staging = row
|
|
34
|
+
return self._staging
|
|
35
|
+
|
|
36
|
+
def clone(self):
|
|
37
|
+
cloned = self.__class__.from_dict(self.flat)
|
|
38
|
+
cloned._prefix = self._prefix
|
|
39
|
+
return cloned
|
|
40
|
+
|
|
41
|
+
def get(self, key, default=None):
|
|
42
|
+
if self._prefix:
|
|
43
|
+
key = f'{self._prefix}.{key}'
|
|
44
|
+
value, found = get_nested_field_value(self.nested, key)
|
|
45
|
+
if not found:
|
|
46
|
+
return default
|
|
47
|
+
return value
|
|
48
|
+
|
|
49
|
+
def __getitem__(self, key):
|
|
50
|
+
if self._prefix:
|
|
51
|
+
key = f'{self._prefix}.{key}'
|
|
52
|
+
value, found = get_nested_field_value(self.nested, key)
|
|
53
|
+
if not found:
|
|
54
|
+
raise KeyError(f'Key not found: {key}')
|
|
55
|
+
return value
|
|
56
|
+
|
|
57
|
+
def __setitem__(self, key, value):
|
|
58
|
+
if self._prefix:
|
|
59
|
+
key = f'{self._prefix}.{key}'
|
|
60
|
+
set_nested_field_value(self.nested, key, value)
|
|
61
|
+
set_flat_field_value(self.flat, key, value)
|
|
62
|
+
|
|
63
|
+
def __contains__(self, key):
|
|
64
|
+
if self._prefix:
|
|
65
|
+
key = f'{self._prefix}.{key}'
|
|
66
|
+
_, found = get_nested_field_value(self.nested, key)
|
|
67
|
+
return found
|
|
68
|
+
|
|
69
|
+
def __iter__(self):
|
|
70
|
+
return iter(self.flat)
|
|
71
|
+
|
|
72
|
+
def __len__(self):
|
|
73
|
+
return len(self.flat)
|
|
74
|
+
|
|
75
|
+
def __repr__(self):
|
|
76
|
+
return f'Row(flat={self.flat}, nested={self.nested})'
|
|
77
|
+
|
|
78
|
+
@staticmethod
|
|
79
|
+
def from_dict(data: dict):
|
|
80
|
+
row = Row()
|
|
81
|
+
for key, value in data.items():
|
|
82
|
+
row[key] = value
|
|
83
|
+
return row
|
|
@@ -1,11 +1,16 @@
|
|
|
1
1
|
# -*- coding: utf-8 -*-
|
|
2
2
|
|
|
3
3
|
import json
|
|
4
|
-
import math
|
|
5
4
|
import os
|
|
5
|
+
import sys
|
|
6
6
|
|
|
7
7
|
from collections import OrderedDict
|
|
8
8
|
|
|
9
|
+
import rich
|
|
10
|
+
from rich.console import Console
|
|
11
|
+
from rich.panel import Panel
|
|
12
|
+
from rich.text import Text
|
|
13
|
+
|
|
9
14
|
from typing import (
|
|
10
15
|
Mapping,
|
|
11
16
|
)
|
|
@@ -13,9 +18,12 @@ from typing import (
|
|
|
13
18
|
# 3-rd party modules
|
|
14
19
|
|
|
15
20
|
from icecream import ic
|
|
21
|
+
from logzero import logger
|
|
16
22
|
import numpy as np
|
|
17
23
|
import pandas as pd
|
|
18
24
|
|
|
25
|
+
from . progress import Progress
|
|
26
|
+
|
|
19
27
|
# local
|
|
20
28
|
|
|
21
29
|
from . config import (
|
|
@@ -33,11 +41,6 @@ from . constants import (
|
|
|
33
41
|
from . functions.get_nested_field_value import get_nested_field_value
|
|
34
42
|
from . functions.get_nested_field_value import get_nested_field_value
|
|
35
43
|
from . functions.search_column_value import search_column_value
|
|
36
|
-
from . functions.set_nested_field_value import set_nested_field_value
|
|
37
|
-
from . functions.set_row_value import (
|
|
38
|
-
set_row_value,
|
|
39
|
-
set_row_staging_value,
|
|
40
|
-
)
|
|
41
44
|
|
|
42
45
|
from . actions import (
|
|
43
46
|
do_actions,
|
|
@@ -53,11 +56,22 @@ from . types import (
|
|
|
53
56
|
|
|
54
57
|
from . io import (
|
|
55
58
|
get_loader,
|
|
56
|
-
get_saver,
|
|
57
|
-
|
|
58
|
-
|
|
59
|
+
#get_saver,
|
|
60
|
+
get_writer,
|
|
61
|
+
#load,
|
|
62
|
+
#save,
|
|
59
63
|
)
|
|
60
64
|
|
|
65
|
+
def capture_dict(
|
|
66
|
+
row: dict,
|
|
67
|
+
):
|
|
68
|
+
console = Console()
|
|
69
|
+
with console.capture() as capture:
|
|
70
|
+
#console.print(row)
|
|
71
|
+
console.print_json(data=row)
|
|
72
|
+
text = Text.from_ansi(capture.get())
|
|
73
|
+
return text
|
|
74
|
+
|
|
61
75
|
def assign_array(
|
|
62
76
|
row: OrderedDict,
|
|
63
77
|
dict_config: Mapping[str, list[AssignArrayConfig]],
|
|
@@ -106,17 +120,22 @@ def convert(
|
|
|
106
120
|
action_delimiter: str = ':',
|
|
107
121
|
verbose: bool = False,
|
|
108
122
|
ignore_file_rows: list[str] | None = None,
|
|
109
|
-
|
|
123
|
+
no_header: bool = False,
|
|
110
124
|
):
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
125
|
+
#console = Console()
|
|
126
|
+
progress = Progress(
|
|
127
|
+
#console = console,
|
|
128
|
+
redirect_stdout = False,
|
|
129
|
+
)
|
|
130
|
+
progress.start()
|
|
131
|
+
#ic.enable()
|
|
132
|
+
console = progress.console
|
|
133
|
+
console.log('input_files: ', input_files)
|
|
115
134
|
row_list_filtered_out = []
|
|
116
135
|
set_ignore_file_rows = set()
|
|
117
136
|
global_status = GlobalStatus()
|
|
118
137
|
config = setup_config(config_path)
|
|
119
|
-
|
|
138
|
+
#console.log('config: ', config)
|
|
120
139
|
if ignore_file_rows:
|
|
121
140
|
set_ignore_file_rows = set(ignore_file_rows)
|
|
122
141
|
if list_pick_columns:
|
|
@@ -127,40 +146,44 @@ def convert(
|
|
|
127
146
|
list_actions,
|
|
128
147
|
action_delimiter=action_delimiter
|
|
129
148
|
)
|
|
149
|
+
writer = None
|
|
130
150
|
if output_file:
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
|
|
151
|
+
writer = get_writer(
|
|
152
|
+
output_file,
|
|
153
|
+
progress=progress,
|
|
154
|
+
)
|
|
155
|
+
num_stacked_rows = 0
|
|
134
156
|
for input_file in input_files:
|
|
135
|
-
ic(input_file)
|
|
136
157
|
if not os.path.exists(input_file):
|
|
137
158
|
raise FileNotFoundError(f'File not found: {input_file}')
|
|
138
159
|
base_name = os.path.basename(input_file)
|
|
139
|
-
|
|
140
|
-
|
|
141
|
-
|
|
142
|
-
|
|
143
|
-
|
|
144
|
-
#
|
|
145
|
-
|
|
146
|
-
#new_rows = []
|
|
147
|
-
new_flat_rows = []
|
|
148
|
-
for index, flat_row in df.iterrows():
|
|
160
|
+
loader = get_loader(
|
|
161
|
+
input_file,
|
|
162
|
+
no_header=no_header,
|
|
163
|
+
progress=progress,
|
|
164
|
+
)
|
|
165
|
+
console.log('# rows: ', len(loader))
|
|
166
|
+
for index, row in enumerate(loader):
|
|
149
167
|
file_row_index = f'{input_file}:{index}'
|
|
150
168
|
if file_row_index in set_ignore_file_rows:
|
|
151
169
|
continue
|
|
152
170
|
short_file_row_index = f'{base_name}:{index}'
|
|
153
171
|
if short_file_row_index in set_ignore_file_rows:
|
|
154
172
|
continue
|
|
155
|
-
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
|
|
163
|
-
|
|
173
|
+
orig_row = row.clone()
|
|
174
|
+
if STAGING_FIELD not in row:
|
|
175
|
+
row.staging[FILE_FIELD] = input_file
|
|
176
|
+
row.staging[FILE_ROW_INDEX_FIELD] = file_row_index
|
|
177
|
+
row.staging[ROW_INDEX_FIELD] = index
|
|
178
|
+
row.staging[INPUT_FIELD] = orig_row.nested
|
|
179
|
+
if loader.extension in ['.csv', '.xlsx']:
|
|
180
|
+
#values = []
|
|
181
|
+
#for (key, value) in orig_row.flat.items():
|
|
182
|
+
# values.append(value)
|
|
183
|
+
#if values:
|
|
184
|
+
# row.staging[f'{INPUT_FIELD}.__values__'] = values
|
|
185
|
+
for key_index, (key, value) in enumerate(orig_row.flat.items()):
|
|
186
|
+
row.staging[f'{INPUT_FIELD}.__values__.{key_index}'] = value
|
|
164
187
|
if config.process.assign_array:
|
|
165
188
|
row.flat= assign_array(row.flat, config.process.assign_array)
|
|
166
189
|
if config.actions:
|
|
@@ -178,32 +201,35 @@ def convert(
|
|
|
178
201
|
except Exception as e:
|
|
179
202
|
if verbose:
|
|
180
203
|
ic(index)
|
|
181
|
-
ic(flat_row)
|
|
204
|
+
#ic(flat_row)
|
|
182
205
|
ic(row.flat)
|
|
183
206
|
raise e
|
|
184
207
|
if config.pick:
|
|
185
208
|
remap_columns(row, config.pick)
|
|
209
|
+
if writer is None:
|
|
210
|
+
if sys.stdout.isatty():
|
|
211
|
+
if num_stacked_rows == 0:
|
|
212
|
+
console.print(
|
|
213
|
+
Panel(
|
|
214
|
+
capture_dict(
|
|
215
|
+
row.nested
|
|
216
|
+
),
|
|
217
|
+
title='First Row',
|
|
218
|
+
)
|
|
219
|
+
)
|
|
186
220
|
if not output_debug:
|
|
187
221
|
pop_row_staging(row)
|
|
188
|
-
|
|
189
|
-
|
|
190
|
-
|
|
191
|
-
|
|
192
|
-
|
|
193
|
-
|
|
194
|
-
|
|
195
|
-
|
|
196
|
-
|
|
197
|
-
#ic(all_df)
|
|
198
|
-
ic(len(all_df))
|
|
199
|
-
#ic(all_df.columns)
|
|
200
|
-
#ic(all_df.iloc[0])
|
|
201
|
-
if output_file:
|
|
202
|
-
ic('Saving to: ', output_file)
|
|
203
|
-
saver(all_df, output_file)
|
|
204
|
-
else:
|
|
205
|
-
ic(all_df)
|
|
222
|
+
if writer:
|
|
223
|
+
writer.push_row(row)
|
|
224
|
+
else:
|
|
225
|
+
pass
|
|
226
|
+
num_stacked_rows += 1
|
|
227
|
+
console.log('Total input rows: ', num_stacked_rows)
|
|
228
|
+
if writer:
|
|
229
|
+
writer.close()
|
|
230
|
+
#else:
|
|
231
|
+
# ic(all_df)
|
|
206
232
|
if row_list_filtered_out:
|
|
207
233
|
df_filtered_out = pd.DataFrame(row_list_filtered_out)
|
|
208
234
|
ic('Saving filtered out to: ', output_file_filtered_out)
|
|
209
|
-
|
|
235
|
+
writer(df_filtered_out, output_file_filtered_out)
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
from . extensions import io_csv
|
|
2
|
+
from . extensions import io_excel
|
|
3
|
+
from . extensions import io_json
|
|
4
|
+
from . extensions import io_jsonl
|
|
5
|
+
|
|
6
|
+
from . loader import Loader
|
|
7
|
+
from . extensions.manage_writers import (
|
|
8
|
+
get_writer,
|
|
9
|
+
save,
|
|
10
|
+
)
|
|
11
|
+
|
|
12
|
+
get_loader = Loader
|
|
13
|
+
|
|
14
|
+
__all__ = [
|
|
15
|
+
'get_loader',
|
|
16
|
+
'get_writer',
|
|
17
|
+
'save',
|
|
18
|
+
]
|
|
@@ -0,0 +1,104 @@
|
|
|
1
|
+
#import pandas as pd
|
|
2
|
+
|
|
3
|
+
import csv
|
|
4
|
+
|
|
5
|
+
from collections import OrderedDict
|
|
6
|
+
|
|
7
|
+
from rich.console import Console
|
|
8
|
+
|
|
9
|
+
from tqdm.auto import tqdm
|
|
10
|
+
|
|
11
|
+
from . manage_loaders import (
|
|
12
|
+
Row,
|
|
13
|
+
register_loader,
|
|
14
|
+
)
|
|
15
|
+
from . manage_writers import (
|
|
16
|
+
BaseWriter,
|
|
17
|
+
register_writer,
|
|
18
|
+
)
|
|
19
|
+
|
|
20
|
+
from ... progress import (
|
|
21
|
+
Progress,
|
|
22
|
+
track,
|
|
23
|
+
)
|
|
24
|
+
|
|
25
|
+
@register_loader('.csv')
|
|
26
|
+
def load_csv(
|
|
27
|
+
input_file: str,
|
|
28
|
+
progress: Progress | None = None,
|
|
29
|
+
**kwargs,
|
|
30
|
+
):
|
|
31
|
+
no_header = kwargs.get('no_header', False)
|
|
32
|
+
# UTF-8 with BOM
|
|
33
|
+
quiet = kwargs.get('quiet', False)
|
|
34
|
+
encoding = kwargs.get('encoding', 'utf-8-sig')
|
|
35
|
+
if progress is None:
|
|
36
|
+
console = Console()
|
|
37
|
+
else:
|
|
38
|
+
console = progress.console
|
|
39
|
+
console.log('Loading CSV data from: ', input_file)
|
|
40
|
+
def get_iter(reader):
|
|
41
|
+
return track(
|
|
42
|
+
reader,
|
|
43
|
+
description='Loading rows...',
|
|
44
|
+
disable=quiet,
|
|
45
|
+
progress=progress,
|
|
46
|
+
)
|
|
47
|
+
reader = csv.reader(open(input_file, 'r', encoding=encoding))
|
|
48
|
+
if no_header:
|
|
49
|
+
for i, row in enumerate(get_iter(reader)):
|
|
50
|
+
d = {}
|
|
51
|
+
for j, field in enumerate(row):
|
|
52
|
+
d[f'{j}'] = field
|
|
53
|
+
yield Row.from_dict(d)
|
|
54
|
+
else:
|
|
55
|
+
for i, row in enumerate(get_iter(reader)):
|
|
56
|
+
if i == 0:
|
|
57
|
+
header = row
|
|
58
|
+
continue
|
|
59
|
+
d = OrderedDict()
|
|
60
|
+
for j, field in enumerate(row):
|
|
61
|
+
d[header[j]] = field
|
|
62
|
+
#for j in range(len(row), len(header)):
|
|
63
|
+
# d[f'__staging__.__values__.{j}'] = None
|
|
64
|
+
yield Row.from_dict(d)
|
|
65
|
+
|
|
66
|
+
@register_writer('.csv')
|
|
67
|
+
class CsvWriter(BaseWriter):
|
|
68
|
+
def __init__(
|
|
69
|
+
self,
|
|
70
|
+
output_file: str,
|
|
71
|
+
**kwargs,
|
|
72
|
+
):
|
|
73
|
+
self.writer: csv.DictWriter | None = None
|
|
74
|
+
super().__init__(
|
|
75
|
+
output_file,
|
|
76
|
+
**kwargs,
|
|
77
|
+
)
|
|
78
|
+
|
|
79
|
+
def support_streaming(self):
|
|
80
|
+
return True
|
|
81
|
+
|
|
82
|
+
def _write_row(self, row: Row):
|
|
83
|
+
if self.writer is None:
|
|
84
|
+
# NOTE: 最初の行を取得するまでヘッダーを決定できない
|
|
85
|
+
self.writer = csv.DictWriter(self.fobj, fieldnames=row.flat.keys())
|
|
86
|
+
self.writer.writeheader()
|
|
87
|
+
self.writer.writerow(row.flat)
|
|
88
|
+
|
|
89
|
+
def _write_all_rows(self):
|
|
90
|
+
if self.rows is None:
|
|
91
|
+
return
|
|
92
|
+
self._open()
|
|
93
|
+
self.writer = csv.DictWriter(self.fobj)
|
|
94
|
+
header = []
|
|
95
|
+
for row in self.rows:
|
|
96
|
+
for key in row.keys():
|
|
97
|
+
if key not in header:
|
|
98
|
+
header.append(key)
|
|
99
|
+
self.writer.writeheader(header)
|
|
100
|
+
for row in self.rows:
|
|
101
|
+
self._write_row(row)
|
|
102
|
+
self.writer = None
|
|
103
|
+
self.fobj.close()
|
|
104
|
+
self.fobj = None
|
|
@@ -0,0 +1,66 @@
|
|
|
1
|
+
from icecream import ic
|
|
2
|
+
from tqdm.auto import tqdm
|
|
3
|
+
import numpy as np
|
|
4
|
+
import pandas as pd
|
|
5
|
+
|
|
6
|
+
from rich.console import Console
|
|
7
|
+
|
|
8
|
+
from logzero import logger
|
|
9
|
+
|
|
10
|
+
from . manage_loaders import (
|
|
11
|
+
Row,
|
|
12
|
+
register_loader,
|
|
13
|
+
)
|
|
14
|
+
from . manage_writers import (
|
|
15
|
+
BaseWriter,
|
|
16
|
+
register_writer,
|
|
17
|
+
)
|
|
18
|
+
|
|
19
|
+
@register_loader('.xlsx')
|
|
20
|
+
def load_excel(
|
|
21
|
+
input_file: str,
|
|
22
|
+
console: Console | None = None,
|
|
23
|
+
**kwargs,
|
|
24
|
+
):
|
|
25
|
+
quiet = kwargs.get('quiet', False)
|
|
26
|
+
if not quiet:
|
|
27
|
+
if console is None:
|
|
28
|
+
console = Console()
|
|
29
|
+
console.log('Loading excel data from: ', input_file)
|
|
30
|
+
# NOTE: Excelで勝手に日時データなどに変換されてしまうことを防ぐため
|
|
31
|
+
df = pd.read_excel(input_file, dtype=str)
|
|
32
|
+
# NOTE: 列番号でもアクセスできるようフィールドを追加する
|
|
33
|
+
#df_with_column_number = pd.read_excel(
|
|
34
|
+
# input_file, dtype=str, header=None, skiprows=1
|
|
35
|
+
#)
|
|
36
|
+
#new_column_names = [f'__staging__.__input__.__values__.{i}' for i in df_with_column_number.columns]
|
|
37
|
+
#df2 = df_with_column_number.rename(columns=dict(
|
|
38
|
+
# zip(df_with_column_number.columns, new_column_names)
|
|
39
|
+
#))
|
|
40
|
+
#df = pd.concat([df, df2], axis=1)
|
|
41
|
+
#df = df.dropna(axis=0, how='all')
|
|
42
|
+
#df = df.dropna(axis=1, how='all')
|
|
43
|
+
# NOTE: NaN を None に変換しておかないと厄介
|
|
44
|
+
df = df.replace([np.nan], [None])
|
|
45
|
+
#return df
|
|
46
|
+
for i, row in df.iterrows():
|
|
47
|
+
yield Row.from_dict(row.to_dict())
|
|
48
|
+
|
|
49
|
+
@register_writer('.xlsx')
|
|
50
|
+
class ExcelWriter(BaseWriter):
|
|
51
|
+
def __init__(
|
|
52
|
+
self,
|
|
53
|
+
target: str,
|
|
54
|
+
**kwargs,
|
|
55
|
+
):
|
|
56
|
+
super().__init__(target, **kwargs)
|
|
57
|
+
|
|
58
|
+
def support_streaming(self):
|
|
59
|
+
return False
|
|
60
|
+
|
|
61
|
+
def _write_all_rows(
|
|
62
|
+
self,
|
|
63
|
+
):
|
|
64
|
+
df = pd.DataFrame([row.flat for row in self.rows])
|
|
65
|
+
df.to_excel(self.target, index=False)
|
|
66
|
+
self.finished = True
|
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
import json
|
|
2
|
+
|
|
3
|
+
from rich.console import Console
|
|
4
|
+
|
|
5
|
+
from . manage_loaders import (
|
|
6
|
+
Row,
|
|
7
|
+
register_loader,
|
|
8
|
+
)
|
|
9
|
+
from . manage_writers import (
|
|
10
|
+
BaseWriter,
|
|
11
|
+
register_writer,
|
|
12
|
+
)
|
|
13
|
+
|
|
14
|
+
from ... progress import Progress
|
|
15
|
+
|
|
16
|
+
@register_loader('.json')
|
|
17
|
+
def load_json(
|
|
18
|
+
input_file: str,
|
|
19
|
+
progress: Progress | None = None,
|
|
20
|
+
**kwargs,
|
|
21
|
+
):
|
|
22
|
+
quiet = kwargs.get('quiet', False)
|
|
23
|
+
if not quiet:
|
|
24
|
+
if progress is not None:
|
|
25
|
+
console = progress.console
|
|
26
|
+
else:
|
|
27
|
+
console = Console()
|
|
28
|
+
console.log('Loading JSON data from: ', input_file)
|
|
29
|
+
with open(input_file, 'r') as f:
|
|
30
|
+
data = json.load(f)
|
|
31
|
+
if not isinstance(data, list):
|
|
32
|
+
raise ValueError(f'Invalid JSON array data: {input_file}')
|
|
33
|
+
for row in data:
|
|
34
|
+
yield Row.from_dict(row)
|
|
35
|
+
|
|
36
|
+
@register_writer('.json')
|
|
37
|
+
class JsonWriter(BaseWriter):
|
|
38
|
+
def __init__(
|
|
39
|
+
self,
|
|
40
|
+
output_file: str,
|
|
41
|
+
**kwargs,
|
|
42
|
+
):
|
|
43
|
+
super().__init__(output_file, **kwargs)
|
|
44
|
+
|
|
45
|
+
def support_streaming(self):
|
|
46
|
+
return False
|
|
47
|
+
|
|
48
|
+
def _write_all_rows(self):
|
|
49
|
+
self._open()
|
|
50
|
+
if not self.quiet:
|
|
51
|
+
console = self._get_console()
|
|
52
|
+
console.log(f'Writing {len(self.rows)} JSON rows into: ', self.target)
|
|
53
|
+
rows = [row.nested for row in self.rows]
|
|
54
|
+
self.fobj.write(json.dumps(rows, indent=2, ensure_ascii=False))
|
|
55
|
+
self.fobj.close()
|
|
56
|
+
self.fobj = None
|
|
57
|
+
self.finished = True
|