python-corekit 0.1.1__py3-none-any.whl → 0.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- corekit/api/__init__.py +18 -3
- corekit/api/application.py +275 -0
- corekit/api/lifespan.py +233 -0
- corekit/api/middleware.py +93 -0
- corekit/api/routers.py +109 -1
- corekit/concurrency/__init__.py +2 -2
- corekit/concurrency/decorators.py +32 -5
- corekit/concurrency/thread_local.py +2 -2
- corekit/concurrency/worker.py +74 -65
- corekit/config/loader.py +42 -5
- corekit/config/settings.py +11 -1
- corekit/connections/__init__.py +7 -1
- corekit/connections/connectable.py +45 -4
- corekit/connections/redis/connection.py +53 -10
- corekit/connections/sql/__init__.py +33 -4
- corekit/connections/sql/connection.py +56 -3
- corekit/connections/sql/fields/__init__.py +2 -2
- corekit/connections/sql/fields/jsonb.py +13 -6
- corekit/connections/sql/migration/__init__.py +9 -5
- corekit/connections/sql/migration/base.py +3 -3
- corekit/connections/sql/migration/operations.py +135 -44
- corekit/connections/sql/migration/registry.py +2 -2
- corekit/connections/sql/operations/__init__.py +24 -0
- corekit/connections/sql/operations/base.py +111 -0
- corekit/connections/sql/operations/statements.py +170 -0
- corekit/connections/sql/query.py +4 -62
- corekit/connections/sql/table.py +33 -29
- corekit/crypto/__init__.py +3 -1
- corekit/crypto/constants.py +2 -2
- corekit/crypto/hasher.py +9 -4
- corekit/data/__init__.py +8 -0
- corekit/data/dataset.py +8 -2
- corekit/data/expressions/__init__.py +10 -2
- corekit/data/expressions/comparison.py +142 -123
- corekit/data/expressions/expression.py +71 -98
- corekit/data/expressions/operator.py +39 -0
- corekit/data/expressions/target.py +21 -0
- corekit/data/record.py +147 -147
- corekit/data/stats.py +162 -157
- corekit/decorators/__init__.py +2 -2
- corekit/decorators/exception_handling.py +38 -9
- corekit/docker/watchdog.py +50 -31
- corekit/etl/__init__.py +2 -1
- corekit/etl/connection.py +46 -44
- corekit/etl/extract/extractor.py +6 -13
- corekit/etl/orchestrator.py +19 -2
- corekit/etl/schemas.py +2 -2
- corekit/etl/transform/transformer.py +4 -1
- corekit/events/publisher.py +1 -1
- corekit/events/reader.py +26 -21
- corekit/events/sse.py +4 -1
- corekit/events/websocket.py +27 -13
- corekit/exceptions/__init__.py +33 -0
- corekit/exceptions/base.py +139 -10
- corekit/exceptions/enum.py +17 -0
- corekit/exceptions/types.py +6 -6
- corekit/files/__init__.py +2 -4
- corekit/files/base.py +15 -2
- corekit/files/enum.py +0 -5
- corekit/files/json.py +16 -2
- corekit/http/__init__.py +51 -0
- corekit/http/api.py +24 -0
- corekit/http/client.py +100 -73
- corekit/http/exceptions.py +140 -0
- corekit/http/response.py +50 -1
- corekit/http/status.py +89 -0
- corekit/jobs/__init__.py +26 -0
- corekit/jobs/registry.py +87 -0
- corekit/jobs/runner.py +80 -0
- corekit/jobs/task.py +173 -0
- corekit/log_monitor/models.py +8 -2
- corekit/log_monitor/service.py +77 -38
- corekit/notifications/base.py +18 -10
- corekit/observability/__init__.py +12 -3
- corekit/observability/benchmarkable.py +23 -5
- corekit/observability/loggable.py +21 -0
- corekit/observability/request_context.py +188 -0
- corekit/observability/timing/timer.py +4 -2
- corekit/registry/__init__.py +12 -7
- corekit/registry/ordered.py +86 -0
- corekit/registry/registry.py +55 -14
- corekit/schemas/__init__.py +10 -0
- corekit/schemas/enum.py +70 -49
- corekit/schemas/models/arbitrary.py +11 -11
- corekit/schemas/pydantic/fields.py +35 -35
- corekit/schemas/types.py +45 -40
- corekit/serialization/__init__.py +24 -0
- corekit/serialization/pickle_file.py +61 -0
- corekit/serialization/serializable.py +22 -2
- corekit/serialization/serializer.py +10 -3
- corekit/utils/__init__.py +59 -5
- corekit/utils/coercion.py +118 -0
- corekit/utils/collections.py +124 -0
- corekit/utils/ids.py +61 -5
- corekit/utils/payload.py +112 -0
- corekit/utils/raise_exc.py +8 -8
- corekit/utils/text.py +56 -0
- corekit/utils/time.py +74 -21
- corekit/utils/validators.py +15 -15
- corekit/utils/void.py +8 -8
- {python_corekit-0.1.1.dist-info → python_corekit-0.3.0.dist-info}/METADATA +103 -97
- python_corekit-0.3.0.dist-info/RECORD +145 -0
- corekit/constants.py +0 -45
- corekit/exceptions/http/exceptions.py +0 -37
- corekit/files/pickle.py +0 -12
- python_corekit-0.1.1.dist-info/RECORD +0 -125
- {python_corekit-0.1.1.dist-info → python_corekit-0.3.0.dist-info}/WHEEL +0 -0
- {python_corekit-0.1.1.dist-info → python_corekit-0.3.0.dist-info}/licenses/LICENSE +0 -0
- {python_corekit-0.1.1.dist-info → python_corekit-0.3.0.dist-info}/top_level.txt +0 -0
corekit/data/record.py
CHANGED
|
@@ -1,147 +1,147 @@
|
|
|
1
|
-
import functools
|
|
2
|
-
import keyword
|
|
3
|
-
import logging
|
|
4
|
-
from typing import Any, Callable, Iterator
|
|
5
|
-
|
|
6
|
-
logger = logging.getLogger(__name__)
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
def is_valid_key(key: Any) -> bool:
|
|
10
|
-
return isinstance(key, str) and key.isidentifier() and not keyword.iskeyword(key)
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
def return_constant(value: Any) -> Any:
|
|
14
|
-
return value
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
def resolve_default(default: Any, default_factory: Callable[[], Any] | None) -> Callable[[], Any]:
|
|
18
|
-
"""
|
|
19
|
-
Normalize (default, default_factory) into a single zero-arg factory.
|
|
20
|
-
|
|
21
|
-
Raises on a raw mutable default (list/dict/set/bytearray), since that
|
|
22
|
-
object would otherwise be shared by reference across every record;
|
|
23
|
-
use default_factory instead. The non-factory path returns
|
|
24
|
-
functools.partial(_return_constant, default) rather than a lambda,
|
|
25
|
-
since a local lambda can't be pickled.
|
|
26
|
-
"""
|
|
27
|
-
if default_factory is not None:
|
|
28
|
-
if default is not None:
|
|
29
|
-
logger.warning("Both default and default_factory were provided. Prioritizing default_factory.")
|
|
30
|
-
return default_factory
|
|
31
|
-
|
|
32
|
-
if isinstance(default, (list, dict, set, bytearray)):
|
|
33
|
-
raise TypeError(
|
|
34
|
-
f"mutable default {default!r} would be shared by reference across every "
|
|
35
|
-
f"record; use default_factory instead (e.g. default_factory=list)"
|
|
36
|
-
)
|
|
37
|
-
return functools.partial(return_constant, default)
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
def reconstruct_record(fields: tuple, values: tuple) -> "BaseRecord":
|
|
41
|
-
"""
|
|
42
|
-
Rebuild a record on unpickling. Module-level so pickle can reference
|
|
43
|
-
it by import path -- the record's actual class is generated
|
|
44
|
-
dynamically and has no module path of its own.
|
|
45
|
-
"""
|
|
46
|
-
cls = BaseRecord.create_new(fields)
|
|
47
|
-
instance = cls()
|
|
48
|
-
for key, value in zip(fields, values):
|
|
49
|
-
instance[key] = value
|
|
50
|
-
return instance
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
class BaseRecord:
|
|
54
|
-
"""
|
|
55
|
-
Base class for dynamically generated, schema-specific record types.
|
|
56
|
-
|
|
57
|
-
Subclasses are built per-dataset by create_new() with __slots__ for
|
|
58
|
-
each field, so instances carry no per-record __dict__ overhead.
|
|
59
|
-
__slots__ = () here means this base itself adds nothing on top of
|
|
60
|
-
that.
|
|
61
|
-
"""
|
|
62
|
-
|
|
63
|
-
__slots__ = ()
|
|
64
|
-
__fieldmap__ = {}
|
|
65
|
-
|
|
66
|
-
@staticmethod
|
|
67
|
-
def create_new(fields: tuple) -> type:
|
|
68
|
-
"""
|
|
69
|
-
Build a record class for the given field names.
|
|
70
|
-
|
|
71
|
-
Fields that aren't valid Python identifiers get a synthetic slot
|
|
72
|
-
name (_slot_0, _slot_1, ...) instead, tracked via __fieldmap__
|
|
73
|
-
so dict-style access (record["weird field"]) still works.
|
|
74
|
-
Synthetic names are checked against real field names so a field
|
|
75
|
-
literally named "_slot_0" can't collide with one.
|
|
76
|
-
"""
|
|
77
|
-
fieldmap: dict[Any, str] = {}
|
|
78
|
-
slot_names: list[str] = []
|
|
79
|
-
reserved = {key for key in fields if is_valid_key(key)}
|
|
80
|
-
used: set[str] = set()
|
|
81
|
-
counter = 0
|
|
82
|
-
for key in fields:
|
|
83
|
-
if is_valid_key(key):
|
|
84
|
-
slot = key
|
|
85
|
-
else:
|
|
86
|
-
candidate = f"_slot_{counter}"
|
|
87
|
-
counter += 1
|
|
88
|
-
while candidate in reserved or candidate in used:
|
|
89
|
-
candidate = f"_slot_{counter}"
|
|
90
|
-
counter += 1
|
|
91
|
-
slot = candidate
|
|
92
|
-
used.add(slot)
|
|
93
|
-
fieldmap[key] = slot
|
|
94
|
-
slot_names.append(slot)
|
|
95
|
-
return type("Record", (BaseRecord,), {"__slots__": tuple(slot_names), "__fieldmap__": fieldmap})
|
|
96
|
-
|
|
97
|
-
def __getitem__(self, key: Any) -> Any:
|
|
98
|
-
slot = self.__fieldmap__.get(key, key)
|
|
99
|
-
try:
|
|
100
|
-
return getattr(self, slot)
|
|
101
|
-
except AttributeError:
|
|
102
|
-
raise KeyError(key) from None
|
|
103
|
-
|
|
104
|
-
def __setitem__(self, key: Any, value: Any) -> Any:
|
|
105
|
-
slot = self.__fieldmap__.get(key)
|
|
106
|
-
if slot is None:
|
|
107
|
-
raise KeyError(f"{key!r} does not exist for record: {self}")
|
|
108
|
-
setattr(self, slot, value)
|
|
109
|
-
|
|
110
|
-
def __contains__(self, key: Any) -> bool:
|
|
111
|
-
return key in self.__fieldmap__
|
|
112
|
-
|
|
113
|
-
def __iter__(self) -> Iterator[Any]:
|
|
114
|
-
return iter(self.__fieldmap__)
|
|
115
|
-
|
|
116
|
-
def __len__(self) -> int:
|
|
117
|
-
return len(self.__fieldmap__)
|
|
118
|
-
|
|
119
|
-
def __eq__(self, other: Any) -> bool:
|
|
120
|
-
"""
|
|
121
|
-
Value equality: same fields and values, not same object.
|
|
122
|
-
|
|
123
|
-
Records are mutable, so __hash__ is implicitly disabled once
|
|
124
|
-
__eq__ is defined (Python does this automatically) -- that's
|
|
125
|
-
intentional, not an oversight, since a hashable object whose
|
|
126
|
-
hash can change under mutation is unsafe to use as a dict/set key.
|
|
127
|
-
"""
|
|
128
|
-
if isinstance(other, BaseRecord):
|
|
129
|
-
return self.to_dict() == other.to_dict()
|
|
130
|
-
if isinstance(other, dict):
|
|
131
|
-
return self.to_dict() == other
|
|
132
|
-
return NotImplemented
|
|
133
|
-
|
|
134
|
-
def __repr__(self) -> str:
|
|
135
|
-
fields = ", ".join(f"{key}={self[key]!r}" for key in self.__fieldmap__)
|
|
136
|
-
return f"{type(self).__name__}({fields})"
|
|
137
|
-
|
|
138
|
-
def __reduce__(self) -> tuple:
|
|
139
|
-
"""
|
|
140
|
-
Pickle as (reconstruct_fn, (fields, values)); see _reconstruct_record
|
|
141
|
-
"""
|
|
142
|
-
fields = tuple(self.__fieldmap__.keys())
|
|
143
|
-
values = tuple(self[k] for k in fields)
|
|
144
|
-
return reconstruct_record, (fields, values)
|
|
145
|
-
|
|
146
|
-
def to_dict(self) -> dict[Any, Any]:
|
|
147
|
-
return {k: self[k] for k in self.__fieldmap__}
|
|
1
|
+
import functools
|
|
2
|
+
import keyword
|
|
3
|
+
import logging
|
|
4
|
+
from typing import Any, Callable, Iterator
|
|
5
|
+
|
|
6
|
+
logger = logging.getLogger(__name__)
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
def is_valid_key(key: Any) -> bool:
|
|
10
|
+
return isinstance(key, str) and key.isidentifier() and not keyword.iskeyword(key)
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def return_constant(value: Any) -> Any:
|
|
14
|
+
return value
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def resolve_default(default: Any, default_factory: Callable[[], Any] | None) -> Callable[[], Any]:
|
|
18
|
+
"""
|
|
19
|
+
Normalize (default, default_factory) into a single zero-arg factory.
|
|
20
|
+
|
|
21
|
+
Raises on a raw mutable default (list/dict/set/bytearray), since that
|
|
22
|
+
object would otherwise be shared by reference across every record;
|
|
23
|
+
use default_factory instead. The non-factory path returns
|
|
24
|
+
functools.partial(_return_constant, default) rather than a lambda,
|
|
25
|
+
since a local lambda can't be pickled.
|
|
26
|
+
"""
|
|
27
|
+
if default_factory is not None:
|
|
28
|
+
if default is not None:
|
|
29
|
+
logger.warning("Both default and default_factory were provided. Prioritizing default_factory.")
|
|
30
|
+
return default_factory
|
|
31
|
+
|
|
32
|
+
if isinstance(default, (list, dict, set, bytearray)):
|
|
33
|
+
raise TypeError(
|
|
34
|
+
f"mutable default {default!r} would be shared by reference across every "
|
|
35
|
+
f"record; use default_factory instead (e.g. default_factory=list)"
|
|
36
|
+
)
|
|
37
|
+
return functools.partial(return_constant, default)
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def reconstruct_record(fields: tuple, values: tuple) -> "BaseRecord":
|
|
41
|
+
"""
|
|
42
|
+
Rebuild a record on unpickling. Module-level so pickle can reference
|
|
43
|
+
it by import path -- the record's actual class is generated
|
|
44
|
+
dynamically and has no module path of its own.
|
|
45
|
+
"""
|
|
46
|
+
cls = BaseRecord.create_new(fields)
|
|
47
|
+
instance = cls()
|
|
48
|
+
for key, value in zip(fields, values):
|
|
49
|
+
instance[key] = value
|
|
50
|
+
return instance
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
class BaseRecord:
|
|
54
|
+
"""
|
|
55
|
+
Base class for dynamically generated, schema-specific record types.
|
|
56
|
+
|
|
57
|
+
Subclasses are built per-dataset by create_new() with __slots__ for
|
|
58
|
+
each field, so instances carry no per-record __dict__ overhead.
|
|
59
|
+
__slots__ = () here means this base itself adds nothing on top of
|
|
60
|
+
that.
|
|
61
|
+
"""
|
|
62
|
+
|
|
63
|
+
__slots__ = ()
|
|
64
|
+
__fieldmap__ = {}
|
|
65
|
+
|
|
66
|
+
@staticmethod
|
|
67
|
+
def create_new(fields: tuple) -> type:
|
|
68
|
+
"""
|
|
69
|
+
Build a record class for the given field names.
|
|
70
|
+
|
|
71
|
+
Fields that aren't valid Python identifiers get a synthetic slot
|
|
72
|
+
name (_slot_0, _slot_1, ...) instead, tracked via __fieldmap__
|
|
73
|
+
so dict-style access (record["weird field"]) still works.
|
|
74
|
+
Synthetic names are checked against real field names so a field
|
|
75
|
+
literally named "_slot_0" can't collide with one.
|
|
76
|
+
"""
|
|
77
|
+
fieldmap: dict[Any, str] = {}
|
|
78
|
+
slot_names: list[str] = []
|
|
79
|
+
reserved = {key for key in fields if is_valid_key(key)}
|
|
80
|
+
used: set[str] = set()
|
|
81
|
+
counter = 0
|
|
82
|
+
for key in fields:
|
|
83
|
+
if is_valid_key(key):
|
|
84
|
+
slot = key
|
|
85
|
+
else:
|
|
86
|
+
candidate = f"_slot_{counter}"
|
|
87
|
+
counter += 1
|
|
88
|
+
while candidate in reserved or candidate in used:
|
|
89
|
+
candidate = f"_slot_{counter}"
|
|
90
|
+
counter += 1
|
|
91
|
+
slot = candidate
|
|
92
|
+
used.add(slot)
|
|
93
|
+
fieldmap[key] = slot
|
|
94
|
+
slot_names.append(slot)
|
|
95
|
+
return type("Record", (BaseRecord,), {"__slots__": tuple(slot_names), "__fieldmap__": fieldmap})
|
|
96
|
+
|
|
97
|
+
def __getitem__(self, key: Any) -> Any:
|
|
98
|
+
slot = self.__fieldmap__.get(key, key)
|
|
99
|
+
try:
|
|
100
|
+
return getattr(self, slot)
|
|
101
|
+
except AttributeError:
|
|
102
|
+
raise KeyError(key) from None
|
|
103
|
+
|
|
104
|
+
def __setitem__(self, key: Any, value: Any) -> Any:
|
|
105
|
+
slot = self.__fieldmap__.get(key)
|
|
106
|
+
if slot is None:
|
|
107
|
+
raise KeyError(f"{key!r} does not exist for record: {self}")
|
|
108
|
+
setattr(self, slot, value)
|
|
109
|
+
|
|
110
|
+
def __contains__(self, key: Any) -> bool:
|
|
111
|
+
return key in self.__fieldmap__
|
|
112
|
+
|
|
113
|
+
def __iter__(self) -> Iterator[Any]:
|
|
114
|
+
return iter(self.__fieldmap__)
|
|
115
|
+
|
|
116
|
+
def __len__(self) -> int:
|
|
117
|
+
return len(self.__fieldmap__)
|
|
118
|
+
|
|
119
|
+
def __eq__(self, other: Any) -> bool:
|
|
120
|
+
"""
|
|
121
|
+
Value equality: same fields and values, not same object.
|
|
122
|
+
|
|
123
|
+
Records are mutable, so __hash__ is implicitly disabled once
|
|
124
|
+
__eq__ is defined (Python does this automatically) -- that's
|
|
125
|
+
intentional, not an oversight, since a hashable object whose
|
|
126
|
+
hash can change under mutation is unsafe to use as a dict/set key.
|
|
127
|
+
"""
|
|
128
|
+
if isinstance(other, BaseRecord):
|
|
129
|
+
return self.to_dict() == other.to_dict()
|
|
130
|
+
if isinstance(other, dict):
|
|
131
|
+
return self.to_dict() == other
|
|
132
|
+
return NotImplemented
|
|
133
|
+
|
|
134
|
+
def __repr__(self) -> str:
|
|
135
|
+
fields = ", ".join(f"{key}={self[key]!r}" for key in self.__fieldmap__)
|
|
136
|
+
return f"{type(self).__name__}({fields})"
|
|
137
|
+
|
|
138
|
+
def __reduce__(self) -> tuple:
|
|
139
|
+
"""
|
|
140
|
+
Pickle as (reconstruct_fn, (fields, values)); see _reconstruct_record
|
|
141
|
+
"""
|
|
142
|
+
fields = tuple(self.__fieldmap__.keys())
|
|
143
|
+
values = tuple(self[k] for k in fields)
|
|
144
|
+
return reconstruct_record, (fields, values)
|
|
145
|
+
|
|
146
|
+
def to_dict(self) -> dict[Any, Any]:
|
|
147
|
+
return {k: self[k] for k in self.__fieldmap__}
|
corekit/data/stats.py
CHANGED
|
@@ -1,157 +1,162 @@
|
|
|
1
|
-
import statistics
|
|
2
|
-
from collections import Counter
|
|
3
|
-
from dataclasses import dataclass
|
|
4
|
-
from dataclasses import field as dataclass_field
|
|
5
|
-
from typing import Any, Iterator
|
|
6
|
-
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
|
|
12
|
-
"""
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
"""
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
"""
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
"""
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
if self.
|
|
93
|
-
return
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
|
|
134
|
-
|
|
135
|
-
|
|
136
|
-
|
|
137
|
-
|
|
138
|
-
|
|
139
|
-
|
|
140
|
-
|
|
141
|
-
|
|
142
|
-
|
|
143
|
-
|
|
144
|
-
"""
|
|
145
|
-
|
|
146
|
-
|
|
147
|
-
|
|
148
|
-
|
|
149
|
-
|
|
150
|
-
|
|
151
|
-
|
|
152
|
-
fields
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
|
|
156
|
-
|
|
157
|
-
|
|
1
|
+
import statistics
|
|
2
|
+
from collections import Counter
|
|
3
|
+
from dataclasses import dataclass
|
|
4
|
+
from dataclasses import field as dataclass_field
|
|
5
|
+
from typing import Any, Iterator
|
|
6
|
+
|
|
7
|
+
from corekit.utils.coercion import safe_tuple
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
@dataclass(slots=True)
|
|
11
|
+
class FieldStats:
|
|
12
|
+
"""
|
|
13
|
+
Common shape shared by every field's statistics
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
field: Any
|
|
17
|
+
count: int
|
|
18
|
+
missing: int
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
@dataclass(slots=True)
|
|
22
|
+
class NumericFieldStats(FieldStats):
|
|
23
|
+
"""
|
|
24
|
+
Stats for a field where every non-missing value is int/float (not bool)
|
|
25
|
+
"""
|
|
26
|
+
|
|
27
|
+
mean: float
|
|
28
|
+
min: Any
|
|
29
|
+
max: Any
|
|
30
|
+
stdev: float
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
@dataclass(slots=True)
|
|
34
|
+
class CategoricalFieldStats(FieldStats):
|
|
35
|
+
"""
|
|
36
|
+
Stats for any field that isn't purely numeric
|
|
37
|
+
"""
|
|
38
|
+
|
|
39
|
+
unique: int
|
|
40
|
+
top: Any
|
|
41
|
+
freq: int
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
@dataclass(slots=True)
|
|
45
|
+
class ValueCounts:
|
|
46
|
+
"""
|
|
47
|
+
How often each distinct value of a field occurs, most frequent first
|
|
48
|
+
"""
|
|
49
|
+
|
|
50
|
+
field: Any
|
|
51
|
+
counts: dict[Any, int]
|
|
52
|
+
|
|
53
|
+
def most_common(self, n: int | None = None) -> list[tuple[Any, int]]:
|
|
54
|
+
items = list(self.counts.items())
|
|
55
|
+
return items[:n] if n is not None else items
|
|
56
|
+
|
|
57
|
+
def top(self) -> tuple[Any, int] | None:
|
|
58
|
+
return next(iter(self.counts.items()), None)
|
|
59
|
+
|
|
60
|
+
def __iter__(self) -> Iterator[tuple[Any, int]]:
|
|
61
|
+
return iter(self.counts.items())
|
|
62
|
+
|
|
63
|
+
def __len__(self) -> int:
|
|
64
|
+
return len(self.counts)
|
|
65
|
+
|
|
66
|
+
def __getitem__(self, value: Any) -> int:
|
|
67
|
+
return self.counts[value]
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
@dataclass(slots=True)
|
|
71
|
+
class FieldDescription:
|
|
72
|
+
field: Any
|
|
73
|
+
values: list[Any] = dataclass_field(default_factory=list)
|
|
74
|
+
non_missing: list[Any] = dataclass_field(default_factory=list)
|
|
75
|
+
num_values: int = dataclass_field(default=0)
|
|
76
|
+
num_non_missing: int = dataclass_field(default=0)
|
|
77
|
+
|
|
78
|
+
@classmethod
|
|
79
|
+
def from_data(cls, data: list[Any], field: Any) -> "FieldDescription":
|
|
80
|
+
instance = cls(field)
|
|
81
|
+
for record in data:
|
|
82
|
+
instance.add(record[field])
|
|
83
|
+
return instance
|
|
84
|
+
|
|
85
|
+
@property
|
|
86
|
+
def missing(self) -> int:
|
|
87
|
+
return self.num_values - self.num_non_missing
|
|
88
|
+
|
|
89
|
+
@property
|
|
90
|
+
def is_numeric(self) -> bool:
|
|
91
|
+
# all([]) is True, which would then call statistics.fmean([]) and crash.
|
|
92
|
+
if not self.non_missing:
|
|
93
|
+
return False
|
|
94
|
+
return all(isinstance(v, (int, float)) and not isinstance(v, bool) for v in self.non_missing)
|
|
95
|
+
|
|
96
|
+
def to_field_stats(self) -> FieldStats:
|
|
97
|
+
if self.is_numeric:
|
|
98
|
+
return NumericFieldStats(
|
|
99
|
+
field=self.field,
|
|
100
|
+
count=self.num_non_missing,
|
|
101
|
+
missing=self.missing,
|
|
102
|
+
mean=statistics.fmean(self.non_missing),
|
|
103
|
+
min=min(self.non_missing),
|
|
104
|
+
max=max(self.non_missing),
|
|
105
|
+
stdev=statistics.stdev(self.non_missing) if self.num_non_missing > 1 else 0.0,
|
|
106
|
+
)
|
|
107
|
+
|
|
108
|
+
counts = Counter(self.non_missing)
|
|
109
|
+
top_value, top_count = counts.most_common(1)[0] if counts else (None, 0)
|
|
110
|
+
return CategoricalFieldStats(
|
|
111
|
+
field=self.field,
|
|
112
|
+
count=self.num_non_missing,
|
|
113
|
+
missing=self.missing,
|
|
114
|
+
unique=len(counts),
|
|
115
|
+
top=top_value,
|
|
116
|
+
freq=top_count,
|
|
117
|
+
)
|
|
118
|
+
|
|
119
|
+
def add(self, value: Any) -> None:
|
|
120
|
+
self.values.append(value)
|
|
121
|
+
self.num_values += 1
|
|
122
|
+
if value is not None:
|
|
123
|
+
self.non_missing.append(value)
|
|
124
|
+
self.num_non_missing += 1
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
class DatasetStats:
|
|
128
|
+
"""
|
|
129
|
+
Read-only field statistics for a Dataset's current records.
|
|
130
|
+
|
|
131
|
+
Computed fresh on every call from the dataset's public interface (never
|
|
132
|
+
cached), so there's nothing here that can go stale after a mutation --
|
|
133
|
+
the same reasoning that ruled out persistent secondary indexes applies:
|
|
134
|
+
recomputing is cheap and always correct, caching is not free and can lie.
|
|
135
|
+
"""
|
|
136
|
+
|
|
137
|
+
def __init__(self, data: list[Any], schema: tuple[Any, ...] | None = None) -> None:
|
|
138
|
+
self._data = data
|
|
139
|
+
self._schema = safe_tuple(schema)
|
|
140
|
+
|
|
141
|
+
def value_counts(self, field: Any) -> ValueCounts:
|
|
142
|
+
"""
|
|
143
|
+
How often each distinct value of `field` occurs, most frequent first
|
|
144
|
+
"""
|
|
145
|
+
counts = dict(Counter(rec[field] for rec in self._data).most_common())
|
|
146
|
+
return ValueCounts(field=field, counts=counts)
|
|
147
|
+
|
|
148
|
+
def describe(self, field: Any | None = None) -> dict[Any, FieldStats]:
|
|
149
|
+
"""
|
|
150
|
+
Summary statistics per field, pandas-.describe()-style.
|
|
151
|
+
|
|
152
|
+
Numeric fields (int/float, excluding bool) get a NumericFieldStats
|
|
153
|
+
(mean/min/max/stdev). Everything else gets a CategoricalFieldStats
|
|
154
|
+
(unique/top/freq). Pass `field` to describe just one; omit to
|
|
155
|
+
describe every field in the schema.
|
|
156
|
+
"""
|
|
157
|
+
fields = (field,) if field is not None else self._schema
|
|
158
|
+
stats = {}
|
|
159
|
+
for f in fields:
|
|
160
|
+
field_description = FieldDescription.from_data(self._data, f)
|
|
161
|
+
stats[f] = field_description.to_field_stats()
|
|
162
|
+
return stats
|
corekit/decorators/__init__.py
CHANGED
|
@@ -1,2 +1,2 @@
|
|
|
1
|
-
from .exception_handling import exception_handler
|
|
2
|
-
from .warnings import deprecated
|
|
1
|
+
from .exception_handling import exception_handler
|
|
2
|
+
from .warnings import deprecated
|