fluidattacks_dynamo_etl 4.0.5__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- fluidattacks_dynamo_etl/__init__.py +11 -0
- fluidattacks_dynamo_etl/_core.py +74 -0
- fluidattacks_dynamo_etl/_logger.py +27 -0
- fluidattacks_dynamo_etl/_multi_file/__init__.py +0 -0
- fluidattacks_dynamo_etl/_multi_file/classifier.py +37 -0
- fluidattacks_dynamo_etl/_multi_file/splitter.py +220 -0
- fluidattacks_dynamo_etl/_parallel.py +124 -0
- fluidattacks_dynamo_etl/_utils/__init__.py +93 -0
- fluidattacks_dynamo_etl/_utils/dict.py +23 -0
- fluidattacks_dynamo_etl/_utils/iter.py +32 -0
- fluidattacks_dynamo_etl/_utils/merge.py +15 -0
- fluidattacks_dynamo_etl/_utils/non_empty.py +58 -0
- fluidattacks_dynamo_etl/_utils/pure_io.py +225 -0
- fluidattacks_dynamo_etl/cli/__init__.py +84 -0
- fluidattacks_dynamo_etl/cli/_schemas.py +37 -0
- fluidattacks_dynamo_etl/cue_schema/__init__.py +25 -0
- fluidattacks_dynamo_etl/cue_schema/_utils.py +28 -0
- fluidattacks_dynamo_etl/cue_schema/core/__init__.py +0 -0
- fluidattacks_dynamo_etl/cue_schema/core/primitive.py +24 -0
- fluidattacks_dynamo_etl/cue_schema/core/resolved.py +205 -0
- fluidattacks_dynamo_etl/cue_schema/core/unresolved.py +145 -0
- fluidattacks_dynamo_etl/cue_schema/decode/__init__.py +0 -0
- fluidattacks_dynamo_etl/cue_schema/decode/core.py +77 -0
- fluidattacks_dynamo_etl/cue_schema/decode/schemas.py +50 -0
- fluidattacks_dynamo_etl/cue_schema/decode/u_type.py +219 -0
- fluidattacks_dynamo_etl/cue_schema/encode/__init__.py +0 -0
- fluidattacks_dynamo_etl/cue_schema/encode/data_type.py +46 -0
- fluidattacks_dynamo_etl/cue_schema/flattener/__init__.py +24 -0
- fluidattacks_dynamo_etl/cue_schema/flattener/base.py +114 -0
- fluidattacks_dynamo_etl/cue_schema/flattener/core.py +11 -0
- fluidattacks_dynamo_etl/cue_schema/flattener/inner.py +68 -0
- fluidattacks_dynamo_etl/cue_schema/merge.py +82 -0
- fluidattacks_dynamo_etl/cue_schema/resolve.py +61 -0
- fluidattacks_dynamo_etl/cue_schema/simplify.py +124 -0
- fluidattacks_dynamo_etl/data_schema/__init__.py +0 -0
- fluidattacks_dynamo_etl/data_schema/cache.py +38 -0
- fluidattacks_dynamo_etl/data_schema/decode.py +27 -0
- fluidattacks_dynamo_etl/data_values/__init__.py +0 -0
- fluidattacks_dynamo_etl/data_values/flattener/__init__.py +29 -0
- fluidattacks_dynamo_etl/data_values/flattener/_flattener.py +77 -0
- fluidattacks_dynamo_etl/data_values/flattener/_nested_id.py +62 -0
- fluidattacks_dynamo_etl/data_values/flattener/_to_table.py +168 -0
- fluidattacks_dynamo_etl/data_values/flattener/core.py +114 -0
- fluidattacks_dynamo_etl/data_values/transform/__init__.py +0 -0
- fluidattacks_dynamo_etl/data_values/transform/adjust_obj.py +30 -0
- fluidattacks_dynamo_etl/data_values/transform/adjust_value.py +186 -0
- fluidattacks_dynamo_etl/data_values/transform/complete_record.py +39 -0
- fluidattacks_dynamo_etl/data_values/transform/csv_format.py +52 -0
- fluidattacks_dynamo_etl/data_values/transform/postfix.py +59 -0
- fluidattacks_dynamo_etl/data_values/transform/str_date.py +39 -0
- fluidattacks_dynamo_etl/dynamo/__init__.py +0 -0
- fluidattacks_dynamo_etl/dynamo/export/__init__.py +29 -0
- fluidattacks_dynamo_etl/dynamo/export/_client.py +136 -0
- fluidattacks_dynamo_etl/dynamo/export/_core.py +58 -0
- fluidattacks_dynamo_etl/dynamo/export/decode.py +38 -0
- fluidattacks_dynamo_etl/dynamo/table/__init__.py +23 -0
- fluidattacks_dynamo_etl/dynamo/table/_client.py +167 -0
- fluidattacks_dynamo_etl/dynamo/table/_core.py +53 -0
- fluidattacks_dynamo_etl/dynamo/table/_decode.py +89 -0
- fluidattacks_dynamo_etl/dynamo/table/_encode.py +87 -0
- fluidattacks_dynamo_etl/dynamo/value/__init__.py +31 -0
- fluidattacks_dynamo_etl/dynamo/value/_factory.py +150 -0
- fluidattacks_dynamo_etl/dynamo/value/_scalar.py +131 -0
- fluidattacks_dynamo_etl/dynamo/value/_scalar_set.py +100 -0
- fluidattacks_dynamo_etl/dynamo/value/_transform.py +138 -0
- fluidattacks_dynamo_etl/dynamo/value/_value.py +106 -0
- fluidattacks_dynamo_etl/dynamo/value/decode.py +133 -0
- fluidattacks_dynamo_etl/etls/__init__.py +20 -0
- fluidattacks_dynamo_etl/etls/core.py +26 -0
- fluidattacks_dynamo_etl/etls/generic/__init__.py +0 -0
- fluidattacks_dynamo_etl/etls/generic/core.py +54 -0
- fluidattacks_dynamo_etl/etls/generic/executor.py +54 -0
- fluidattacks_dynamo_etl/etls/generic/phases.py +144 -0
- fluidattacks_dynamo_etl/etls/generic/segment.py +26 -0
- fluidattacks_dynamo_etl/etls/integrates/__init__.py +19 -0
- fluidattacks_dynamo_etl/etls/integrates/_core.py +26 -0
- fluidattacks_dynamo_etl/etls/integrates/schema.py +183 -0
- fluidattacks_dynamo_etl/etls/sifts.py +43 -0
- fluidattacks_dynamo_etl/jobs_sdk/__init__.py +0 -0
- fluidattacks_dynamo_etl/jobs_sdk/batch/__init__.py +33 -0
- fluidattacks_dynamo_etl/jobs_sdk/batch/_core.py +23 -0
- fluidattacks_dynamo_etl/jobs_sdk/batch/_jobs.py +234 -0
- fluidattacks_dynamo_etl/jschema/__init__.py +0 -0
- fluidattacks_dynamo_etl/jschema/decode/__init__.py +49 -0
- fluidattacks_dynamo_etl/jschema/decode/_core.py +67 -0
- fluidattacks_dynamo_etl/jschema/decode/_number.py +98 -0
- fluidattacks_dynamo_etl/jschema/decode/_object.py +100 -0
- fluidattacks_dynamo_etl/jschema/decode/_string.py +66 -0
- fluidattacks_dynamo_etl/jschema/encode/__init__.py +46 -0
- fluidattacks_dynamo_etl/jschema/encode/_number.py +67 -0
- fluidattacks_dynamo_etl/jschema/encode/_object.py +60 -0
- fluidattacks_dynamo_etl/jschema/encode/_string.py +32 -0
- fluidattacks_dynamo_etl/jschema/number.py +142 -0
- fluidattacks_dynamo_etl/jschema/object.py +91 -0
- fluidattacks_dynamo_etl/jschema/string.py +65 -0
- fluidattacks_dynamo_etl/phases/__init__.py +0 -0
- fluidattacks_dynamo_etl/phases/core.py +76 -0
- fluidattacks_dynamo_etl/phases/download_from_export/__init__.py +94 -0
- fluidattacks_dynamo_etl/phases/download_from_export/_decode_segment.py +28 -0
- fluidattacks_dynamo_etl/phases/download_transform.py +54 -0
- fluidattacks_dynamo_etl/phases/plan_and_send.py +26 -0
- fluidattacks_dynamo_etl/phases/segmentation_plan/__init__.py +3 -0
- fluidattacks_dynamo_etl/phases/segmentation_plan/_assignments.py +89 -0
- fluidattacks_dynamo_etl/phases/segmentation_plan/_core.py +8 -0
- fluidattacks_dynamo_etl/phases/segmentation_plan/_decode.py +13 -0
- fluidattacks_dynamo_etl/phases/segmentation_plan/_encode.py +27 -0
- fluidattacks_dynamo_etl/phases/transform.py +88 -0
- fluidattacks_dynamo_etl/phases/upload_csv.py +114 -0
- fluidattacks_dynamo_etl/phases/upload_snowflake/__init__.py +150 -0
- fluidattacks_dynamo_etl/phases/upload_snowflake/_copy.py +114 -0
- fluidattacks_dynamo_etl/phases/upload_snowflake/_prepare.py +97 -0
- fluidattacks_dynamo_etl/s3/__init__.py +14 -0
- fluidattacks_dynamo_etl/s3/_core.py +26 -0
- fluidattacks_dynamo_etl/s3/_operations.py +144 -0
- fluidattacks_dynamo_etl-4.0.5.dist-info/METADATA +18 -0
- fluidattacks_dynamo_etl-4.0.5.dist-info/RECORD +118 -0
- fluidattacks_dynamo_etl-4.0.5.dist-info/WHEEL +4 -0
- fluidattacks_dynamo_etl-4.0.5.dist-info/entry_points.txt +3 -0
|
@@ -0,0 +1,74 @@
|
|
|
1
|
+
from __future__ import (
|
|
2
|
+
annotations,
|
|
3
|
+
)
|
|
4
|
+
|
|
5
|
+
from dataclasses import (
|
|
6
|
+
dataclass,
|
|
7
|
+
field,
|
|
8
|
+
)
|
|
9
|
+
from enum import (
|
|
10
|
+
Enum,
|
|
11
|
+
)
|
|
12
|
+
from typing import (
|
|
13
|
+
TYPE_CHECKING,
|
|
14
|
+
TypeVar,
|
|
15
|
+
)
|
|
16
|
+
|
|
17
|
+
from fa_purity import (
|
|
18
|
+
Maybe,
|
|
19
|
+
Result,
|
|
20
|
+
ResultE,
|
|
21
|
+
)
|
|
22
|
+
|
|
23
|
+
if TYPE_CHECKING:
|
|
24
|
+
from collections.abc import (
|
|
25
|
+
Callable,
|
|
26
|
+
)
|
|
27
|
+
|
|
28
|
+
from fluidattacks_batch_client.core import Natural
|
|
29
|
+
|
|
30
|
+
_T = TypeVar("_T")
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
@dataclass(frozen=True)
|
|
34
|
+
class _Private:
|
|
35
|
+
pass
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
class SupportedEtl(Enum):
|
|
39
|
+
integrates = "integrates"
|
|
40
|
+
sifts = "sifts"
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
class TargetTables(Enum):
|
|
44
|
+
INTEGRATES = "integrates_vms"
|
|
45
|
+
SIFTS = "sifts_state"
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
@dataclass(frozen=True)
|
|
49
|
+
class TotalSegments:
|
|
50
|
+
total: Natural
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
@dataclass(frozen=True)
|
|
54
|
+
class Segment:
|
|
55
|
+
_private: _Private = field(repr=False, hash=False, compare=False)
|
|
56
|
+
_raw: Maybe[int]
|
|
57
|
+
|
|
58
|
+
@staticmethod
|
|
59
|
+
def from_raw(raw: str) -> ResultE[Segment]:
|
|
60
|
+
try:
|
|
61
|
+
item = Segment(_Private(), Maybe.some(int(raw)))
|
|
62
|
+
return Result.success(item, Exception)
|
|
63
|
+
except ValueError as err:
|
|
64
|
+
if raw == "auto":
|
|
65
|
+
item = Segment(_Private(), Maybe.empty())
|
|
66
|
+
return Result.success(item, Exception)
|
|
67
|
+
return Result.failure(Exception(err))
|
|
68
|
+
|
|
69
|
+
def map(
|
|
70
|
+
self,
|
|
71
|
+
case_num: Callable[[int], _T],
|
|
72
|
+
case_auto: Callable[[], _T],
|
|
73
|
+
) -> _T:
|
|
74
|
+
return self._raw.map(case_num).or_else_call(case_auto)
|
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
from fa_purity import (
|
|
2
|
+
Cmd,
|
|
3
|
+
)
|
|
4
|
+
from fluidattacks_utils_logger import (
|
|
5
|
+
set_main_log,
|
|
6
|
+
)
|
|
7
|
+
from fluidattacks_utils_logger.env import (
|
|
8
|
+
current_app_env,
|
|
9
|
+
observes_debug,
|
|
10
|
+
)
|
|
11
|
+
from fluidattacks_utils_logger.handlers import (
|
|
12
|
+
LoggingConf,
|
|
13
|
+
)
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def set_logger(root_name: str, version: str) -> Cmd[None]:
|
|
17
|
+
app_env = current_app_env()
|
|
18
|
+
debug = observes_debug()
|
|
19
|
+
conf = app_env.map(
|
|
20
|
+
lambda env: LoggingConf(
|
|
21
|
+
app_name="observes",
|
|
22
|
+
app_type="etl",
|
|
23
|
+
app_version=version,
|
|
24
|
+
release_stage=env,
|
|
25
|
+
),
|
|
26
|
+
)
|
|
27
|
+
return debug.bind(lambda d: conf.bind(lambda c: set_main_log(root_name, c, d)))
|
|
File without changes
|
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from dataclasses import dataclass
|
|
4
|
+
from typing import TYPE_CHECKING, Generic, TypeVar
|
|
5
|
+
|
|
6
|
+
if TYPE_CHECKING:
|
|
7
|
+
from collections.abc import Callable
|
|
8
|
+
|
|
9
|
+
from fa_purity import Cmd, UnitType
|
|
10
|
+
from fluidattacks_etl_utils.mutable import MutableMap
|
|
11
|
+
|
|
12
|
+
_File = TypeVar("_File")
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
@dataclass(frozen=True)
|
|
16
|
+
class Record:
|
|
17
|
+
stream: str
|
|
18
|
+
line: str
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
@dataclass(frozen=True)
|
|
22
|
+
class StreamClassifier(Generic[_File]):
|
|
23
|
+
"""
|
|
24
|
+
Handle writes of tagged records into various writable objects.
|
|
25
|
+
|
|
26
|
+
- A new writable "_File" object is created per stream/tag.
|
|
27
|
+
- each record is redirected/written into its stream file.
|
|
28
|
+
"""
|
|
29
|
+
|
|
30
|
+
_new_file: Cmd[_File]
|
|
31
|
+
_write_line: Callable[[_File, str], Cmd[UnitType]]
|
|
32
|
+
streams: MutableMap[str, _File]
|
|
33
|
+
|
|
34
|
+
def write_record(self, record: Record) -> Cmd[UnitType]:
|
|
35
|
+
return self.streams.get_or_create(record.stream, self._new_file).bind(
|
|
36
|
+
lambda f: self._write_line(f, record.line),
|
|
37
|
+
)
|
|
@@ -0,0 +1,220 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import logging
|
|
4
|
+
from dataclasses import dataclass
|
|
5
|
+
from typing import TYPE_CHECKING
|
|
6
|
+
|
|
7
|
+
from fa_purity import (
|
|
8
|
+
Cmd,
|
|
9
|
+
CmdSmash,
|
|
10
|
+
NewFrozenList,
|
|
11
|
+
PureIterFactory,
|
|
12
|
+
PureIterTransform,
|
|
13
|
+
ResultTransform,
|
|
14
|
+
UnitType,
|
|
15
|
+
unit,
|
|
16
|
+
)
|
|
17
|
+
from fa_purity.lock import ThreadLock
|
|
18
|
+
from fluidattacks_etl_utils.typing import none_to_unit, unit_to_none
|
|
19
|
+
|
|
20
|
+
from fluidattacks_dynamo_etl._utils.non_empty import MutableVar, NonEmptyList
|
|
21
|
+
from fluidattacks_dynamo_etl._utils.pure_io import FileProperties, WritableIO
|
|
22
|
+
|
|
23
|
+
if TYPE_CHECKING:
|
|
24
|
+
from collections.abc import Callable
|
|
25
|
+
from pathlib import Path
|
|
26
|
+
|
|
27
|
+
LOG = logging.getLogger(__name__)
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
@dataclass(frozen=True)
|
|
31
|
+
class File:
|
|
32
|
+
props: FileProperties[str]
|
|
33
|
+
writer: WritableIO[str]
|
|
34
|
+
lock: ThreadLock
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
@dataclass(frozen=True)
|
|
38
|
+
class MultifileSplitter:
|
|
39
|
+
"""Interface for handling lines writes over multiple files."""
|
|
40
|
+
|
|
41
|
+
write_line: Callable[[str], Cmd[UnitType]]
|
|
42
|
+
write_lines_batch: Callable[[NewFrozenList[str]], Cmd[UnitType]]
|
|
43
|
+
close: Cmd[UnitType]
|
|
44
|
+
paths: Cmd[NonEmptyList[Path]]
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
@dataclass(frozen=True)
|
|
48
|
+
class _WriteBuffer:
|
|
49
|
+
_data: dict[UnitType, list[str]]
|
|
50
|
+
|
|
51
|
+
@staticmethod
|
|
52
|
+
def new() -> Cmd[_WriteBuffer]:
|
|
53
|
+
return Cmd.wrap_impure(lambda: _WriteBuffer({unit: []}))
|
|
54
|
+
|
|
55
|
+
def append(self, line: str) -> Cmd[UnitType]:
|
|
56
|
+
def _action() -> UnitType:
|
|
57
|
+
self._data[unit].append(line)
|
|
58
|
+
return unit
|
|
59
|
+
|
|
60
|
+
return Cmd.wrap_impure(_action)
|
|
61
|
+
|
|
62
|
+
def append_many(self, lines: NewFrozenList[str]) -> Cmd[UnitType]:
|
|
63
|
+
def _action() -> UnitType:
|
|
64
|
+
self._data[unit].extend(lines)
|
|
65
|
+
return unit
|
|
66
|
+
|
|
67
|
+
return Cmd.wrap_impure(_action)
|
|
68
|
+
|
|
69
|
+
@property
|
|
70
|
+
def flush(self) -> Cmd[tuple[str, ...]]:
|
|
71
|
+
def _action() -> tuple[str, ...]:
|
|
72
|
+
result = tuple(self._data[unit])
|
|
73
|
+
self._data[unit].clear()
|
|
74
|
+
return result
|
|
75
|
+
|
|
76
|
+
return Cmd.wrap_impure(_action)
|
|
77
|
+
|
|
78
|
+
@property
|
|
79
|
+
def size(self) -> Cmd[int]:
|
|
80
|
+
return Cmd.wrap_impure(lambda: len(self._data[unit]))
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
@dataclass(frozen=True)
|
|
84
|
+
class _DefaultSplitter:
|
|
85
|
+
_new_file: Cmd[File]
|
|
86
|
+
_lines_limit: int
|
|
87
|
+
_buffer_size: int
|
|
88
|
+
_files: MutableVar[NonEmptyList[File]]
|
|
89
|
+
_line_counter: MutableVar[int]
|
|
90
|
+
_write_buffer: _WriteBuffer
|
|
91
|
+
_lock: ThreadLock
|
|
92
|
+
|
|
93
|
+
def _flush_lines(
|
|
94
|
+
self,
|
|
95
|
+
lines: tuple[str, ...],
|
|
96
|
+
counter: int,
|
|
97
|
+
files: NonEmptyList[File],
|
|
98
|
+
) -> Cmd[UnitType]:
|
|
99
|
+
if not lines:
|
|
100
|
+
return Cmd.wrap_value(unit)
|
|
101
|
+
current = ResultTransform.get_index(files.next_items, -1).value_or(files.item)
|
|
102
|
+
remaining = self._lines_limit - counter
|
|
103
|
+
if len(lines) <= remaining:
|
|
104
|
+
return current.lock.execute_with_lock(
|
|
105
|
+
current.writer.write_lines(PureIterFactory.from_list(lines)),
|
|
106
|
+
) + self._line_counter.update(counter + len(lines))
|
|
107
|
+
fitting = lines[:remaining]
|
|
108
|
+
overflow = lines[remaining:]
|
|
109
|
+
write_fitting = current.lock.execute_with_lock(
|
|
110
|
+
current.writer.write_lines(PureIterFactory.from_list(fitting)),
|
|
111
|
+
)
|
|
112
|
+
|
|
113
|
+
def _add_next_file(current_files: NonEmptyList[File]) -> Cmd[UnitType]:
|
|
114
|
+
return self._new_file.bind(
|
|
115
|
+
lambda new_f: Cmd.wrap_impure(
|
|
116
|
+
lambda: LOG.info(
|
|
117
|
+
"New file created: %s (file #%d)",
|
|
118
|
+
new_f.props.path,
|
|
119
|
+
len(current_files.next_items.items) + 2,
|
|
120
|
+
),
|
|
121
|
+
)
|
|
122
|
+
+ self._files.update(
|
|
123
|
+
NonEmptyList(
|
|
124
|
+
current_files.item,
|
|
125
|
+
NewFrozenList.new(*current_files.next_items, new_f),
|
|
126
|
+
),
|
|
127
|
+
).bind(
|
|
128
|
+
lambda _: self._flush_lines(
|
|
129
|
+
overflow,
|
|
130
|
+
0,
|
|
131
|
+
NonEmptyList(new_f, NewFrozenList.new()),
|
|
132
|
+
),
|
|
133
|
+
),
|
|
134
|
+
)
|
|
135
|
+
|
|
136
|
+
return write_fitting + self._files.get.bind(_add_next_file)
|
|
137
|
+
|
|
138
|
+
@property
|
|
139
|
+
def _flush_buffer(self) -> Cmd[UnitType]:
|
|
140
|
+
return self._write_buffer.size.bind(
|
|
141
|
+
lambda size: self._write_buffer.flush.bind(
|
|
142
|
+
lambda lines: self._line_counter.get.bind(
|
|
143
|
+
lambda c: self._files.get.bind(lambda f: self._flush_lines(lines, c, f)),
|
|
144
|
+
),
|
|
145
|
+
)
|
|
146
|
+
if size > 0
|
|
147
|
+
else Cmd.wrap_value(unit),
|
|
148
|
+
)
|
|
149
|
+
|
|
150
|
+
def write_line(self, line: str) -> Cmd[UnitType]:
|
|
151
|
+
def _maybe_flush(_: UnitType) -> Cmd[UnitType]:
|
|
152
|
+
return self._write_buffer.size.bind(
|
|
153
|
+
lambda size: self._flush_buffer
|
|
154
|
+
if size >= self._buffer_size
|
|
155
|
+
else Cmd.wrap_value(unit),
|
|
156
|
+
)
|
|
157
|
+
|
|
158
|
+
return self._lock.execute_with_lock(
|
|
159
|
+
self._write_buffer.append(line + "\n").bind(_maybe_flush),
|
|
160
|
+
)
|
|
161
|
+
|
|
162
|
+
def write_lines_batch(self, lines: NewFrozenList[str]) -> Cmd[UnitType]:
|
|
163
|
+
if not lines.items:
|
|
164
|
+
return Cmd.wrap_value(unit)
|
|
165
|
+
tagged = lines.map(lambda i: i + "\n")
|
|
166
|
+
|
|
167
|
+
def _maybe_flush(_: UnitType) -> Cmd[UnitType]:
|
|
168
|
+
return self._write_buffer.size.bind(
|
|
169
|
+
lambda size: self._flush_buffer
|
|
170
|
+
if size >= self._buffer_size
|
|
171
|
+
else Cmd.wrap_value(unit),
|
|
172
|
+
)
|
|
173
|
+
|
|
174
|
+
return self._lock.execute_with_lock(
|
|
175
|
+
self._write_buffer.append_many(tagged).bind(_maybe_flush),
|
|
176
|
+
)
|
|
177
|
+
|
|
178
|
+
@property
|
|
179
|
+
def close(self) -> Cmd[UnitType]:
|
|
180
|
+
close_files = self._files.get.bind(
|
|
181
|
+
lambda n: n.item.writer.close
|
|
182
|
+
+ PureIterTransform.consume(
|
|
183
|
+
PureIterFactory.from_list(n.next_items.items).map(
|
|
184
|
+
lambda f: f.writer.close.map(unit_to_none),
|
|
185
|
+
),
|
|
186
|
+
).map(none_to_unit),
|
|
187
|
+
)
|
|
188
|
+
return self._lock.execute_with_lock(self._flush_buffer) + close_files
|
|
189
|
+
|
|
190
|
+
@property
|
|
191
|
+
def paths(self) -> Cmd[NonEmptyList[Path]]:
|
|
192
|
+
return self._files.get.map(lambda n: n.map(lambda f: f.props.path))
|
|
193
|
+
|
|
194
|
+
def to_interface(self) -> MultifileSplitter:
|
|
195
|
+
return MultifileSplitter(
|
|
196
|
+
write_line=self.write_line,
|
|
197
|
+
write_lines_batch=self.write_lines_batch,
|
|
198
|
+
close=self.close,
|
|
199
|
+
paths=self.paths,
|
|
200
|
+
)
|
|
201
|
+
|
|
202
|
+
|
|
203
|
+
def new_splitter(new_file: Cmd[File], lines_limit: int, buffer_size: int) -> Cmd[MultifileSplitter]:
|
|
204
|
+
return CmdSmash.smash_cmds_3(
|
|
205
|
+
new_file.bind(lambda f: MutableVar.new(NonEmptyList(f, NewFrozenList.new()))),
|
|
206
|
+
MutableVar.new(0),
|
|
207
|
+
ThreadLock.new(),
|
|
208
|
+
).bind(
|
|
209
|
+
lambda t: _WriteBuffer.new().map(
|
|
210
|
+
lambda buf: _DefaultSplitter(
|
|
211
|
+
new_file,
|
|
212
|
+
lines_limit,
|
|
213
|
+
buffer_size,
|
|
214
|
+
t[0],
|
|
215
|
+
t[1],
|
|
216
|
+
buf,
|
|
217
|
+
t[2],
|
|
218
|
+
).to_interface(),
|
|
219
|
+
),
|
|
220
|
+
)
|
|
@@ -0,0 +1,124 @@
|
|
|
1
|
+
from __future__ import (
|
|
2
|
+
annotations,
|
|
3
|
+
)
|
|
4
|
+
|
|
5
|
+
import logging
|
|
6
|
+
import queue
|
|
7
|
+
from dataclasses import (
|
|
8
|
+
dataclass,
|
|
9
|
+
field,
|
|
10
|
+
)
|
|
11
|
+
from typing import (
|
|
12
|
+
TYPE_CHECKING,
|
|
13
|
+
Generic,
|
|
14
|
+
TypeVar,
|
|
15
|
+
)
|
|
16
|
+
|
|
17
|
+
from fa_purity import (
|
|
18
|
+
Cmd,
|
|
19
|
+
Coproduct,
|
|
20
|
+
Maybe,
|
|
21
|
+
PureIter,
|
|
22
|
+
PureIterFactory,
|
|
23
|
+
Stream,
|
|
24
|
+
StreamFactory,
|
|
25
|
+
StreamTransform,
|
|
26
|
+
UnitType,
|
|
27
|
+
Unsafe,
|
|
28
|
+
)
|
|
29
|
+
from fluidattacks_etl_utils.typing import none_to_unit, unit_to_none
|
|
30
|
+
|
|
31
|
+
if TYPE_CHECKING:
|
|
32
|
+
from fluidattacks_etl_utils.parallel import ThreadPool
|
|
33
|
+
|
|
34
|
+
LOG = logging.getLogger(__name__)
|
|
35
|
+
_T = TypeVar("_T")
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
@dataclass(frozen=True)
|
|
39
|
+
class _Private:
|
|
40
|
+
pass
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
@dataclass(frozen=True)
|
|
44
|
+
class _FinishMark:
|
|
45
|
+
pass
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
@dataclass(frozen=True)
|
|
49
|
+
class Queue(Generic[_T]):
|
|
50
|
+
_private: _Private = field(repr=False, hash=False, compare=False)
|
|
51
|
+
_inner: queue.Queue[_T]
|
|
52
|
+
|
|
53
|
+
@staticmethod
|
|
54
|
+
def new(maxsize: int = 0) -> Cmd[Queue[_T]]:
|
|
55
|
+
def _action() -> Queue[_T]:
|
|
56
|
+
inner: queue.Queue[_T] = queue.Queue(maxsize)
|
|
57
|
+
return Queue(_Private(), inner)
|
|
58
|
+
|
|
59
|
+
return Cmd.wrap_impure(_action)
|
|
60
|
+
|
|
61
|
+
def put(self, item: _T) -> Cmd[UnitType]:
|
|
62
|
+
def _action() -> None:
|
|
63
|
+
self._inner.put(item)
|
|
64
|
+
LOG.info("Queue.put qsize=%s", self._inner.qsize())
|
|
65
|
+
|
|
66
|
+
return Cmd.wrap_impure(_action).map(none_to_unit)
|
|
67
|
+
|
|
68
|
+
def get(self) -> Cmd[_T]:
|
|
69
|
+
return Cmd.wrap_impure(lambda: self._inner.get())
|
|
70
|
+
|
|
71
|
+
def empty(self) -> Cmd[bool]:
|
|
72
|
+
return Cmd.wrap_impure(lambda: self._inner.empty())
|
|
73
|
+
|
|
74
|
+
def qsize(self) -> Cmd[int]:
|
|
75
|
+
return Cmd.wrap_impure(lambda: self._inner.qsize())
|
|
76
|
+
|
|
77
|
+
def task_done(self) -> Cmd[UnitType]:
|
|
78
|
+
return Cmd.wrap_impure(lambda: self._inner.task_done()).map(none_to_unit)
|
|
79
|
+
|
|
80
|
+
def join(self) -> Cmd[UnitType]:
|
|
81
|
+
return Cmd.wrap_impure(lambda: self._inner.join()).map(none_to_unit)
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def squash_cmd_stream(cmd: Cmd[Stream[_T]]) -> Stream[_T]:
|
|
85
|
+
new_iter = cmd.bind(lambda s: Unsafe.stream_to_iter(s))
|
|
86
|
+
return Unsafe.stream_from_cmd(new_iter)
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def _combine_streams(
|
|
90
|
+
pool: ThreadPool,
|
|
91
|
+
queue: Queue[Coproduct[_T, _FinishMark]],
|
|
92
|
+
streams: PureIter[Stream[_T]],
|
|
93
|
+
) -> Stream[_T]:
|
|
94
|
+
queue_emissions = streams.map(
|
|
95
|
+
lambda s: s.map(lambda i: queue.put(Coproduct.inl(i)))
|
|
96
|
+
.map(lambda c: c.map(unit_to_none))
|
|
97
|
+
.transform(StreamTransform.consume)
|
|
98
|
+
.map(none_to_unit),
|
|
99
|
+
)
|
|
100
|
+
signal_done = queue.put(Coproduct.inr(_FinishMark()))
|
|
101
|
+
emit_all = pool.in_background(pool.in_threads_unit(queue_emissions) + signal_done)
|
|
102
|
+
|
|
103
|
+
def _filter_finish(item: Coproduct[_T, _FinishMark]) -> Maybe[_T]:
|
|
104
|
+
return item.map(Maybe.some, lambda _: Maybe.empty())
|
|
105
|
+
|
|
106
|
+
stream_queue = (
|
|
107
|
+
PureIterFactory.infinite_range(0, 0)
|
|
108
|
+
.map(lambda _: queue.get().map(_filter_finish))
|
|
109
|
+
.transform(StreamFactory.from_commands)
|
|
110
|
+
.transform(StreamTransform.until_empty)
|
|
111
|
+
)
|
|
112
|
+
return squash_cmd_stream(emit_all.map(lambda _: stream_queue))
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def combine_streams(
|
|
116
|
+
pool: ThreadPool,
|
|
117
|
+
streams: PureIter[Stream[_T]],
|
|
118
|
+
max_queue_size: int,
|
|
119
|
+
) -> Stream[_T]:
|
|
120
|
+
return squash_cmd_stream(
|
|
121
|
+
Queue[Coproduct[_T, _FinishMark]]
|
|
122
|
+
.new(max_queue_size)
|
|
123
|
+
.map(lambda q: _combine_streams(pool, q, streams)),
|
|
124
|
+
)
|
|
@@ -0,0 +1,93 @@
|
|
|
1
|
+
from __future__ import (
|
|
2
|
+
annotations,
|
|
3
|
+
)
|
|
4
|
+
|
|
5
|
+
import logging
|
|
6
|
+
from typing import (
|
|
7
|
+
TYPE_CHECKING,
|
|
8
|
+
TypeVar,
|
|
9
|
+
)
|
|
10
|
+
|
|
11
|
+
from fa_purity import (
|
|
12
|
+
Cmd,
|
|
13
|
+
CmdUnwrapper,
|
|
14
|
+
Coproduct,
|
|
15
|
+
PureIter,
|
|
16
|
+
Result,
|
|
17
|
+
ResultE,
|
|
18
|
+
ResultFactory,
|
|
19
|
+
Stream,
|
|
20
|
+
Unsafe,
|
|
21
|
+
)
|
|
22
|
+
|
|
23
|
+
if TYPE_CHECKING:
|
|
24
|
+
from collections.abc import (
|
|
25
|
+
Callable,
|
|
26
|
+
Iterable,
|
|
27
|
+
)
|
|
28
|
+
|
|
29
|
+
LOG = logging.getLogger(__name__)
|
|
30
|
+
_T = TypeVar("_T")
|
|
31
|
+
_S = TypeVar("_S")
|
|
32
|
+
_A = TypeVar("_A")
|
|
33
|
+
_F = TypeVar("_F")
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def append_to_stream(stream: Stream[_T], item: _T) -> Stream[_T]:
|
|
37
|
+
def _iter(unwrapper: CmdUnwrapper) -> Iterable[_T]:
|
|
38
|
+
yield from unwrapper.act(Unsafe.stream_to_iter(stream))
|
|
39
|
+
yield item
|
|
40
|
+
|
|
41
|
+
new_iter = Cmd.new_cmd(_iter)
|
|
42
|
+
return Unsafe.stream_from_cmd(new_iter)
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def join_stream(stream_1: Stream[_T], stream_2: Stream[_T]) -> Stream[_T]:
|
|
46
|
+
def _iter(unwrapper: CmdUnwrapper) -> Iterable[_T]:
|
|
47
|
+
yield from unwrapper.act(Unsafe.stream_to_iter(stream_1))
|
|
48
|
+
yield from unwrapper.act(Unsafe.stream_to_iter(stream_2))
|
|
49
|
+
|
|
50
|
+
new_iter = Cmd.new_cmd(_iter)
|
|
51
|
+
return Unsafe.stream_from_cmd(new_iter)
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def chain_cmd_result(
|
|
55
|
+
cmd_1: Cmd[Result[_S, _F]],
|
|
56
|
+
cmd_2: Callable[[_S], Cmd[Result[_A, _F]]],
|
|
57
|
+
) -> Cmd[Result[_A, _F]]:
|
|
58
|
+
factory: ResultFactory[_A, _F] = ResultFactory()
|
|
59
|
+
return cmd_1.bind(
|
|
60
|
+
lambda r: r.map(cmd_2).alt(lambda e: Cmd.wrap_value(factory.failure(e))).to_union(),
|
|
61
|
+
)
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def merge_result_exceptions(
|
|
65
|
+
result: Result[_T, Coproduct[Exception, Exception]],
|
|
66
|
+
) -> ResultE[_T]:
|
|
67
|
+
return result.alt(lambda c: c.map(lambda x: x, lambda x: x))
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def consume_results(cmds: PureIter[Cmd[Result[None, _F]]]) -> Cmd[Result[None, _F]]:
|
|
71
|
+
def _action(unwrapper: CmdUnwrapper) -> Result[None, _F]:
|
|
72
|
+
for cmd in cmds:
|
|
73
|
+
result = unwrapper.act(cmd)
|
|
74
|
+
success = result.map(lambda _: True).value_or(False)
|
|
75
|
+
if not success:
|
|
76
|
+
return result
|
|
77
|
+
return Result.success(None)
|
|
78
|
+
|
|
79
|
+
return Cmd.new_cmd(_action)
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def consume_stream_until_error(
|
|
83
|
+
results: Stream[Result[_T, _F]],
|
|
84
|
+
success_result: _T,
|
|
85
|
+
) -> Cmd[Result[_T, _F]]:
|
|
86
|
+
def _action(unwrapper: CmdUnwrapper) -> Result[_T, _F]:
|
|
87
|
+
for result in unwrapper.act(Unsafe.stream_to_iter(results)):
|
|
88
|
+
success = result.map(lambda _: True).value_or(False)
|
|
89
|
+
if not success:
|
|
90
|
+
return result
|
|
91
|
+
return Result.success(success_result)
|
|
92
|
+
|
|
93
|
+
return Cmd.new_cmd(_action)
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
from typing import (
|
|
2
|
+
TypeVar,
|
|
3
|
+
)
|
|
4
|
+
|
|
5
|
+
from fa_purity import (
|
|
6
|
+
FrozenDict,
|
|
7
|
+
ResultE,
|
|
8
|
+
ResultFactory,
|
|
9
|
+
)
|
|
10
|
+
|
|
11
|
+
_K = TypeVar("_K")
|
|
12
|
+
_V = TypeVar("_V")
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def not_empty_dict(items: FrozenDict[_K, _V]) -> ResultE[FrozenDict[_K, _V]]:
|
|
16
|
+
_factory: ResultFactory[FrozenDict[_K, _V], Exception] = ResultFactory()
|
|
17
|
+
if items:
|
|
18
|
+
return _factory.success(items)
|
|
19
|
+
return _factory.failure(ValueError("Empty FrozenDict"))
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def merge(items_1: FrozenDict[_K, _V], items_2: FrozenDict[_K, _V]) -> FrozenDict[_K, _V]:
|
|
23
|
+
return FrozenDict(dict(items_1) | dict(items_2))
|
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
from collections.abc import Callable, Iterable
|
|
2
|
+
from typing import (
|
|
3
|
+
TypeVar,
|
|
4
|
+
)
|
|
5
|
+
|
|
6
|
+
from fa_purity import (
|
|
7
|
+
Cmd,
|
|
8
|
+
NewFrozenList,
|
|
9
|
+
PureIter,
|
|
10
|
+
Result,
|
|
11
|
+
ResultTransform,
|
|
12
|
+
Unsafe,
|
|
13
|
+
)
|
|
14
|
+
|
|
15
|
+
_T = TypeVar("_T")
|
|
16
|
+
_S = TypeVar("_S")
|
|
17
|
+
_F = TypeVar("_F")
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def append(items_1: PureIter[_T], items_2: PureIter[_T]) -> PureIter[_T]:
|
|
21
|
+
def _new_iter() -> Iterable[_T]:
|
|
22
|
+
yield from items_1
|
|
23
|
+
yield from items_2
|
|
24
|
+
|
|
25
|
+
return Unsafe.pure_iter_from_cmd(Cmd.wrap_impure(_new_iter))
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def apply_and_handle_error(
|
|
29
|
+
function: Callable[[_T], Result[_S, _F]],
|
|
30
|
+
items: PureIter[_T],
|
|
31
|
+
) -> Result[NewFrozenList[_S], _F]:
|
|
32
|
+
return ResultTransform.all_ok_2(NewFrozenList(tuple(items.map(function))))
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
from typing import TypeVar
|
|
2
|
+
|
|
3
|
+
from fa_purity import Coproduct, Result
|
|
4
|
+
|
|
5
|
+
_T = TypeVar("_T")
|
|
6
|
+
_A = TypeVar("_A")
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
def merge_failures(item: Result[_A, Coproduct[_T, _T]]) -> Result[_A, _T]:
|
|
10
|
+
return item.alt(
|
|
11
|
+
lambda c: c.map(
|
|
12
|
+
lambda x: x,
|
|
13
|
+
lambda x: x,
|
|
14
|
+
),
|
|
15
|
+
)
|