python-corekit 0.1.1__py3-none-any.whl → 0.3.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (109) hide show
  1. corekit/api/__init__.py +18 -3
  2. corekit/api/application.py +275 -0
  3. corekit/api/lifespan.py +233 -0
  4. corekit/api/middleware.py +93 -0
  5. corekit/api/routers.py +109 -1
  6. corekit/concurrency/__init__.py +2 -2
  7. corekit/concurrency/decorators.py +32 -5
  8. corekit/concurrency/thread_local.py +2 -2
  9. corekit/concurrency/worker.py +74 -65
  10. corekit/config/loader.py +42 -5
  11. corekit/config/settings.py +11 -1
  12. corekit/connections/__init__.py +7 -1
  13. corekit/connections/connectable.py +45 -4
  14. corekit/connections/redis/connection.py +53 -10
  15. corekit/connections/sql/__init__.py +33 -4
  16. corekit/connections/sql/connection.py +56 -3
  17. corekit/connections/sql/fields/__init__.py +2 -2
  18. corekit/connections/sql/fields/jsonb.py +13 -6
  19. corekit/connections/sql/migration/__init__.py +9 -5
  20. corekit/connections/sql/migration/base.py +3 -3
  21. corekit/connections/sql/migration/operations.py +135 -44
  22. corekit/connections/sql/migration/registry.py +2 -2
  23. corekit/connections/sql/operations/__init__.py +24 -0
  24. corekit/connections/sql/operations/base.py +111 -0
  25. corekit/connections/sql/operations/statements.py +170 -0
  26. corekit/connections/sql/query.py +4 -62
  27. corekit/connections/sql/table.py +33 -29
  28. corekit/crypto/__init__.py +3 -1
  29. corekit/crypto/constants.py +2 -2
  30. corekit/crypto/hasher.py +9 -4
  31. corekit/data/__init__.py +8 -0
  32. corekit/data/dataset.py +8 -2
  33. corekit/data/expressions/__init__.py +10 -2
  34. corekit/data/expressions/comparison.py +142 -123
  35. corekit/data/expressions/expression.py +71 -98
  36. corekit/data/expressions/operator.py +39 -0
  37. corekit/data/expressions/target.py +21 -0
  38. corekit/data/record.py +147 -147
  39. corekit/data/stats.py +162 -157
  40. corekit/decorators/__init__.py +2 -2
  41. corekit/decorators/exception_handling.py +38 -9
  42. corekit/docker/watchdog.py +50 -31
  43. corekit/etl/__init__.py +2 -1
  44. corekit/etl/connection.py +46 -44
  45. corekit/etl/extract/extractor.py +6 -13
  46. corekit/etl/orchestrator.py +19 -2
  47. corekit/etl/schemas.py +2 -2
  48. corekit/etl/transform/transformer.py +4 -1
  49. corekit/events/publisher.py +1 -1
  50. corekit/events/reader.py +26 -21
  51. corekit/events/sse.py +4 -1
  52. corekit/events/websocket.py +27 -13
  53. corekit/exceptions/__init__.py +33 -0
  54. corekit/exceptions/base.py +139 -10
  55. corekit/exceptions/enum.py +17 -0
  56. corekit/exceptions/types.py +6 -6
  57. corekit/files/__init__.py +2 -4
  58. corekit/files/base.py +15 -2
  59. corekit/files/enum.py +0 -5
  60. corekit/files/json.py +16 -2
  61. corekit/http/__init__.py +51 -0
  62. corekit/http/api.py +24 -0
  63. corekit/http/client.py +100 -73
  64. corekit/http/exceptions.py +140 -0
  65. corekit/http/response.py +50 -1
  66. corekit/http/status.py +89 -0
  67. corekit/jobs/__init__.py +26 -0
  68. corekit/jobs/registry.py +87 -0
  69. corekit/jobs/runner.py +80 -0
  70. corekit/jobs/task.py +173 -0
  71. corekit/log_monitor/models.py +8 -2
  72. corekit/log_monitor/service.py +77 -38
  73. corekit/notifications/base.py +18 -10
  74. corekit/observability/__init__.py +12 -3
  75. corekit/observability/benchmarkable.py +23 -5
  76. corekit/observability/loggable.py +21 -0
  77. corekit/observability/request_context.py +188 -0
  78. corekit/observability/timing/timer.py +4 -2
  79. corekit/registry/__init__.py +12 -7
  80. corekit/registry/ordered.py +86 -0
  81. corekit/registry/registry.py +55 -14
  82. corekit/schemas/__init__.py +10 -0
  83. corekit/schemas/enum.py +70 -49
  84. corekit/schemas/models/arbitrary.py +11 -11
  85. corekit/schemas/pydantic/fields.py +35 -35
  86. corekit/schemas/types.py +45 -40
  87. corekit/serialization/__init__.py +24 -0
  88. corekit/serialization/pickle_file.py +61 -0
  89. corekit/serialization/serializable.py +22 -2
  90. corekit/serialization/serializer.py +10 -3
  91. corekit/utils/__init__.py +59 -5
  92. corekit/utils/coercion.py +118 -0
  93. corekit/utils/collections.py +124 -0
  94. corekit/utils/ids.py +61 -5
  95. corekit/utils/payload.py +112 -0
  96. corekit/utils/raise_exc.py +8 -8
  97. corekit/utils/text.py +56 -0
  98. corekit/utils/time.py +74 -21
  99. corekit/utils/validators.py +15 -15
  100. corekit/utils/void.py +8 -8
  101. {python_corekit-0.1.1.dist-info → python_corekit-0.3.0.dist-info}/METADATA +103 -97
  102. python_corekit-0.3.0.dist-info/RECORD +145 -0
  103. corekit/constants.py +0 -45
  104. corekit/exceptions/http/exceptions.py +0 -37
  105. corekit/files/pickle.py +0 -12
  106. python_corekit-0.1.1.dist-info/RECORD +0 -125
  107. {python_corekit-0.1.1.dist-info → python_corekit-0.3.0.dist-info}/WHEEL +0 -0
  108. {python_corekit-0.1.1.dist-info → python_corekit-0.3.0.dist-info}/licenses/LICENSE +0 -0
  109. {python_corekit-0.1.1.dist-info → python_corekit-0.3.0.dist-info}/top_level.txt +0 -0
corekit/data/record.py CHANGED
@@ -1,147 +1,147 @@
1
- import functools
2
- import keyword
3
- import logging
4
- from typing import Any, Callable, Iterator
5
-
6
- logger = logging.getLogger(__name__)
7
-
8
-
9
- def is_valid_key(key: Any) -> bool:
10
- return isinstance(key, str) and key.isidentifier() and not keyword.iskeyword(key)
11
-
12
-
13
- def return_constant(value: Any) -> Any:
14
- return value
15
-
16
-
17
- def resolve_default(default: Any, default_factory: Callable[[], Any] | None) -> Callable[[], Any]:
18
- """
19
- Normalize (default, default_factory) into a single zero-arg factory.
20
-
21
- Raises on a raw mutable default (list/dict/set/bytearray), since that
22
- object would otherwise be shared by reference across every record;
23
- use default_factory instead. The non-factory path returns
24
- functools.partial(_return_constant, default) rather than a lambda,
25
- since a local lambda can't be pickled.
26
- """
27
- if default_factory is not None:
28
- if default is not None:
29
- logger.warning("Both default and default_factory were provided. Prioritizing default_factory.")
30
- return default_factory
31
-
32
- if isinstance(default, (list, dict, set, bytearray)):
33
- raise TypeError(
34
- f"mutable default {default!r} would be shared by reference across every "
35
- f"record; use default_factory instead (e.g. default_factory=list)"
36
- )
37
- return functools.partial(return_constant, default)
38
-
39
-
40
- def reconstruct_record(fields: tuple, values: tuple) -> "BaseRecord":
41
- """
42
- Rebuild a record on unpickling. Module-level so pickle can reference
43
- it by import path -- the record's actual class is generated
44
- dynamically and has no module path of its own.
45
- """
46
- cls = BaseRecord.create_new(fields)
47
- instance = cls()
48
- for key, value in zip(fields, values):
49
- instance[key] = value
50
- return instance
51
-
52
-
53
- class BaseRecord:
54
- """
55
- Base class for dynamically generated, schema-specific record types.
56
-
57
- Subclasses are built per-dataset by create_new() with __slots__ for
58
- each field, so instances carry no per-record __dict__ overhead.
59
- __slots__ = () here means this base itself adds nothing on top of
60
- that.
61
- """
62
-
63
- __slots__ = ()
64
- __fieldmap__ = {}
65
-
66
- @staticmethod
67
- def create_new(fields: tuple) -> type:
68
- """
69
- Build a record class for the given field names.
70
-
71
- Fields that aren't valid Python identifiers get a synthetic slot
72
- name (_slot_0, _slot_1, ...) instead, tracked via __fieldmap__
73
- so dict-style access (record["weird field"]) still works.
74
- Synthetic names are checked against real field names so a field
75
- literally named "_slot_0" can't collide with one.
76
- """
77
- fieldmap: dict[Any, str] = {}
78
- slot_names: list[str] = []
79
- reserved = {key for key in fields if is_valid_key(key)}
80
- used: set[str] = set()
81
- counter = 0
82
- for key in fields:
83
- if is_valid_key(key):
84
- slot = key
85
- else:
86
- candidate = f"_slot_{counter}"
87
- counter += 1
88
- while candidate in reserved or candidate in used:
89
- candidate = f"_slot_{counter}"
90
- counter += 1
91
- slot = candidate
92
- used.add(slot)
93
- fieldmap[key] = slot
94
- slot_names.append(slot)
95
- return type("Record", (BaseRecord,), {"__slots__": tuple(slot_names), "__fieldmap__": fieldmap})
96
-
97
- def __getitem__(self, key: Any) -> Any:
98
- slot = self.__fieldmap__.get(key, key)
99
- try:
100
- return getattr(self, slot)
101
- except AttributeError:
102
- raise KeyError(key) from None
103
-
104
- def __setitem__(self, key: Any, value: Any) -> Any:
105
- slot = self.__fieldmap__.get(key)
106
- if slot is None:
107
- raise KeyError(f"{key!r} does not exist for record: {self}")
108
- setattr(self, slot, value)
109
-
110
- def __contains__(self, key: Any) -> bool:
111
- return key in self.__fieldmap__
112
-
113
- def __iter__(self) -> Iterator[Any]:
114
- return iter(self.__fieldmap__)
115
-
116
- def __len__(self) -> int:
117
- return len(self.__fieldmap__)
118
-
119
- def __eq__(self, other: Any) -> bool:
120
- """
121
- Value equality: same fields and values, not same object.
122
-
123
- Records are mutable, so __hash__ is implicitly disabled once
124
- __eq__ is defined (Python does this automatically) -- that's
125
- intentional, not an oversight, since a hashable object whose
126
- hash can change under mutation is unsafe to use as a dict/set key.
127
- """
128
- if isinstance(other, BaseRecord):
129
- return self.to_dict() == other.to_dict()
130
- if isinstance(other, dict):
131
- return self.to_dict() == other
132
- return NotImplemented
133
-
134
- def __repr__(self) -> str:
135
- fields = ", ".join(f"{key}={self[key]!r}" for key in self.__fieldmap__)
136
- return f"{type(self).__name__}({fields})"
137
-
138
- def __reduce__(self) -> tuple:
139
- """
140
- Pickle as (reconstruct_fn, (fields, values)); see _reconstruct_record
141
- """
142
- fields = tuple(self.__fieldmap__.keys())
143
- values = tuple(self[k] for k in fields)
144
- return reconstruct_record, (fields, values)
145
-
146
- def to_dict(self) -> dict[Any, Any]:
147
- return {k: self[k] for k in self.__fieldmap__}
1
+ import functools
2
+ import keyword
3
+ import logging
4
+ from typing import Any, Callable, Iterator
5
+
6
+ logger = logging.getLogger(__name__)
7
+
8
+
9
+ def is_valid_key(key: Any) -> bool:
10
+ return isinstance(key, str) and key.isidentifier() and not keyword.iskeyword(key)
11
+
12
+
13
+ def return_constant(value: Any) -> Any:
14
+ return value
15
+
16
+
17
+ def resolve_default(default: Any, default_factory: Callable[[], Any] | None) -> Callable[[], Any]:
18
+ """
19
+ Normalize (default, default_factory) into a single zero-arg factory.
20
+
21
+ Raises on a raw mutable default (list/dict/set/bytearray), since that
22
+ object would otherwise be shared by reference across every record;
23
+ use default_factory instead. The non-factory path returns
24
+ functools.partial(_return_constant, default) rather than a lambda,
25
+ since a local lambda can't be pickled.
26
+ """
27
+ if default_factory is not None:
28
+ if default is not None:
29
+ logger.warning("Both default and default_factory were provided. Prioritizing default_factory.")
30
+ return default_factory
31
+
32
+ if isinstance(default, (list, dict, set, bytearray)):
33
+ raise TypeError(
34
+ f"mutable default {default!r} would be shared by reference across every "
35
+ f"record; use default_factory instead (e.g. default_factory=list)"
36
+ )
37
+ return functools.partial(return_constant, default)
38
+
39
+
40
+ def reconstruct_record(fields: tuple, values: tuple) -> "BaseRecord":
41
+ """
42
+ Rebuild a record on unpickling. Module-level so pickle can reference
43
+ it by import path -- the record's actual class is generated
44
+ dynamically and has no module path of its own.
45
+ """
46
+ cls = BaseRecord.create_new(fields)
47
+ instance = cls()
48
+ for key, value in zip(fields, values):
49
+ instance[key] = value
50
+ return instance
51
+
52
+
53
+ class BaseRecord:
54
+ """
55
+ Base class for dynamically generated, schema-specific record types.
56
+
57
+ Subclasses are built per-dataset by create_new() with __slots__ for
58
+ each field, so instances carry no per-record __dict__ overhead.
59
+ __slots__ = () here means this base itself adds nothing on top of
60
+ that.
61
+ """
62
+
63
+ __slots__ = ()
64
+ __fieldmap__ = {}
65
+
66
+ @staticmethod
67
+ def create_new(fields: tuple) -> type:
68
+ """
69
+ Build a record class for the given field names.
70
+
71
+ Fields that aren't valid Python identifiers get a synthetic slot
72
+ name (_slot_0, _slot_1, ...) instead, tracked via __fieldmap__
73
+ so dict-style access (record["weird field"]) still works.
74
+ Synthetic names are checked against real field names so a field
75
+ literally named "_slot_0" can't collide with one.
76
+ """
77
+ fieldmap: dict[Any, str] = {}
78
+ slot_names: list[str] = []
79
+ reserved = {key for key in fields if is_valid_key(key)}
80
+ used: set[str] = set()
81
+ counter = 0
82
+ for key in fields:
83
+ if is_valid_key(key):
84
+ slot = key
85
+ else:
86
+ candidate = f"_slot_{counter}"
87
+ counter += 1
88
+ while candidate in reserved or candidate in used:
89
+ candidate = f"_slot_{counter}"
90
+ counter += 1
91
+ slot = candidate
92
+ used.add(slot)
93
+ fieldmap[key] = slot
94
+ slot_names.append(slot)
95
+ return type("Record", (BaseRecord,), {"__slots__": tuple(slot_names), "__fieldmap__": fieldmap})
96
+
97
+ def __getitem__(self, key: Any) -> Any:
98
+ slot = self.__fieldmap__.get(key, key)
99
+ try:
100
+ return getattr(self, slot)
101
+ except AttributeError:
102
+ raise KeyError(key) from None
103
+
104
+ def __setitem__(self, key: Any, value: Any) -> Any:
105
+ slot = self.__fieldmap__.get(key)
106
+ if slot is None:
107
+ raise KeyError(f"{key!r} does not exist for record: {self}")
108
+ setattr(self, slot, value)
109
+
110
+ def __contains__(self, key: Any) -> bool:
111
+ return key in self.__fieldmap__
112
+
113
+ def __iter__(self) -> Iterator[Any]:
114
+ return iter(self.__fieldmap__)
115
+
116
+ def __len__(self) -> int:
117
+ return len(self.__fieldmap__)
118
+
119
+ def __eq__(self, other: Any) -> bool:
120
+ """
121
+ Value equality: same fields and values, not same object.
122
+
123
+ Records are mutable, so __hash__ is implicitly disabled once
124
+ __eq__ is defined (Python does this automatically) -- that's
125
+ intentional, not an oversight, since a hashable object whose
126
+ hash can change under mutation is unsafe to use as a dict/set key.
127
+ """
128
+ if isinstance(other, BaseRecord):
129
+ return self.to_dict() == other.to_dict()
130
+ if isinstance(other, dict):
131
+ return self.to_dict() == other
132
+ return NotImplemented
133
+
134
+ def __repr__(self) -> str:
135
+ fields = ", ".join(f"{key}={self[key]!r}" for key in self.__fieldmap__)
136
+ return f"{type(self).__name__}({fields})"
137
+
138
+ def __reduce__(self) -> tuple:
139
+ """
140
+ Pickle as (reconstruct_fn, (fields, values)); see _reconstruct_record
141
+ """
142
+ fields = tuple(self.__fieldmap__.keys())
143
+ values = tuple(self[k] for k in fields)
144
+ return reconstruct_record, (fields, values)
145
+
146
+ def to_dict(self) -> dict[Any, Any]:
147
+ return {k: self[k] for k in self.__fieldmap__}
corekit/data/stats.py CHANGED
@@ -1,157 +1,162 @@
1
- import statistics
2
- from collections import Counter
3
- from dataclasses import dataclass
4
- from dataclasses import field as dataclass_field
5
- from typing import Any, Iterator
6
-
7
-
8
- @dataclass(slots=True)
9
- class FieldStats:
10
- """
11
- Common shape shared by every field's statistics
12
- """
13
-
14
- field: Any
15
- count: int
16
- missing: int
17
-
18
-
19
- @dataclass(slots=True)
20
- class NumericFieldStats(FieldStats):
21
- """
22
- Stats for a field where every non-missing value is int/float (not bool)
23
- """
24
-
25
- mean: float
26
- min: Any
27
- max: Any
28
- stdev: float
29
-
30
-
31
- @dataclass(slots=True)
32
- class CategoricalFieldStats(FieldStats):
33
- """
34
- Stats for any field that isn't purely numeric
35
- """
36
-
37
- unique: int
38
- top: Any
39
- freq: int
40
-
41
-
42
- @dataclass(slots=True)
43
- class ValueCounts:
44
- """
45
- How often each distinct value of a field occurs, most frequent first
46
- """
47
-
48
- field: Any
49
- counts: dict[Any, int]
50
-
51
- def most_common(self, n: int | None = None) -> list[tuple[Any, int]]:
52
- items = list(self.counts.items())
53
- return items[:n] if n is not None else items
54
-
55
- def top(self) -> tuple[Any, int] | None:
56
- return next(iter(self.counts.items()), None)
57
-
58
- def __iter__(self) -> Iterator[tuple[Any, int]]:
59
- return iter(self.counts.items())
60
-
61
- def __len__(self) -> int:
62
- return len(self.counts)
63
-
64
- def __getitem__(self, value: Any) -> int:
65
- return self.counts[value]
66
-
67
-
68
- @dataclass(slots=True)
69
- class FieldDescription:
70
- field: Any
71
- values: list[Any] = dataclass_field(default_factory=list)
72
- non_missing: list[Any] = dataclass_field(default_factory=list)
73
- num_values: int = dataclass_field(default=0)
74
- num_non_missing: int = dataclass_field(default=0)
75
-
76
- @classmethod
77
- def from_data(cls, data: list[Any], field: Any) -> "FieldDescription":
78
- instance = cls(field)
79
- for record in data:
80
- instance.add(record[field])
81
- return instance
82
-
83
- @property
84
- def missing(self) -> int:
85
- return self.num_values - self.num_non_missing
86
-
87
- @property
88
- def is_numeric(self) -> bool:
89
- return all(isinstance(v, (int, float)) and not isinstance(v, bool) for v in self.non_missing)
90
-
91
- def to_field_stats(self) -> FieldStats:
92
- if self.is_numeric:
93
- return NumericFieldStats(
94
- field=self.field,
95
- count=self.num_non_missing,
96
- missing=self.missing,
97
- mean=statistics.fmean(self.non_missing),
98
- min=min(self.non_missing),
99
- max=max(self.non_missing),
100
- stdev=statistics.stdev(self.non_missing) if self.num_non_missing > 1 else 0.0,
101
- )
102
-
103
- counts = Counter(self.non_missing)
104
- top_value, top_count = counts.most_common(1)[0] if counts else (None, 0)
105
- return CategoricalFieldStats(
106
- field=self.field,
107
- count=self.num_non_missing,
108
- missing=self.missing,
109
- unique=len(counts),
110
- top=top_value,
111
- freq=top_count,
112
- )
113
-
114
- def add(self, value: Any) -> None:
115
- self.values.append(value)
116
- self.num_values += 1
117
- if value is not None:
118
- self.non_missing.append(value)
119
- self.num_non_missing += 1
120
-
121
-
122
- class DatasetStats:
123
- """
124
- Read-only field statistics for a Dataset's current records.
125
-
126
- Computed fresh on every call from the dataset's public interface (never
127
- cached), so there's nothing here that can go stale after a mutation --
128
- the same reasoning that ruled out persistent secondary indexes applies:
129
- recomputing is cheap and always correct, caching is not free and can lie.
130
- """
131
-
132
- def __init__(self, data: list[Any], schema: tuple[Any, ...] | None = None) -> None:
133
- self._data = data
134
- self._schema = schema or ()
135
-
136
- def value_counts(self, field: Any) -> ValueCounts:
137
- """
138
- How often each distinct value of `field` occurs, most frequent first
139
- """
140
- counts = dict(Counter(rec[field] for rec in self._data).most_common())
141
- return ValueCounts(field=field, counts=counts)
142
-
143
- def describe(self, field: Any | None = None) -> dict[Any, FieldStats]:
144
- """
145
- Summary statistics per field, pandas-.describe()-style.
146
-
147
- Numeric fields (int/float, excluding bool) get a NumericFieldStats
148
- (mean/min/max/stdev). Everything else gets a CategoricalFieldStats
149
- (unique/top/freq). Pass `field` to describe just one; omit to
150
- describe every field in the schema.
151
- """
152
- fields = (field,) if field is not None else self._schema
153
- stats = {}
154
- for f in fields:
155
- field_description = FieldDescription.from_data(self._data, f)
156
- stats[f] = field_description.to_field_stats()
157
- return stats
1
+ import statistics
2
+ from collections import Counter
3
+ from dataclasses import dataclass
4
+ from dataclasses import field as dataclass_field
5
+ from typing import Any, Iterator
6
+
7
+ from corekit.utils.coercion import safe_tuple
8
+
9
+
10
+ @dataclass(slots=True)
11
+ class FieldStats:
12
+ """
13
+ Common shape shared by every field's statistics
14
+ """
15
+
16
+ field: Any
17
+ count: int
18
+ missing: int
19
+
20
+
21
+ @dataclass(slots=True)
22
+ class NumericFieldStats(FieldStats):
23
+ """
24
+ Stats for a field where every non-missing value is int/float (not bool)
25
+ """
26
+
27
+ mean: float
28
+ min: Any
29
+ max: Any
30
+ stdev: float
31
+
32
+
33
+ @dataclass(slots=True)
34
+ class CategoricalFieldStats(FieldStats):
35
+ """
36
+ Stats for any field that isn't purely numeric
37
+ """
38
+
39
+ unique: int
40
+ top: Any
41
+ freq: int
42
+
43
+
44
+ @dataclass(slots=True)
45
+ class ValueCounts:
46
+ """
47
+ How often each distinct value of a field occurs, most frequent first
48
+ """
49
+
50
+ field: Any
51
+ counts: dict[Any, int]
52
+
53
+ def most_common(self, n: int | None = None) -> list[tuple[Any, int]]:
54
+ items = list(self.counts.items())
55
+ return items[:n] if n is not None else items
56
+
57
+ def top(self) -> tuple[Any, int] | None:
58
+ return next(iter(self.counts.items()), None)
59
+
60
+ def __iter__(self) -> Iterator[tuple[Any, int]]:
61
+ return iter(self.counts.items())
62
+
63
+ def __len__(self) -> int:
64
+ return len(self.counts)
65
+
66
+ def __getitem__(self, value: Any) -> int:
67
+ return self.counts[value]
68
+
69
+
70
+ @dataclass(slots=True)
71
+ class FieldDescription:
72
+ field: Any
73
+ values: list[Any] = dataclass_field(default_factory=list)
74
+ non_missing: list[Any] = dataclass_field(default_factory=list)
75
+ num_values: int = dataclass_field(default=0)
76
+ num_non_missing: int = dataclass_field(default=0)
77
+
78
+ @classmethod
79
+ def from_data(cls, data: list[Any], field: Any) -> "FieldDescription":
80
+ instance = cls(field)
81
+ for record in data:
82
+ instance.add(record[field])
83
+ return instance
84
+
85
+ @property
86
+ def missing(self) -> int:
87
+ return self.num_values - self.num_non_missing
88
+
89
+ @property
90
+ def is_numeric(self) -> bool:
91
+ # all([]) is True, which would then call statistics.fmean([]) and crash.
92
+ if not self.non_missing:
93
+ return False
94
+ return all(isinstance(v, (int, float)) and not isinstance(v, bool) for v in self.non_missing)
95
+
96
+ def to_field_stats(self) -> FieldStats:
97
+ if self.is_numeric:
98
+ return NumericFieldStats(
99
+ field=self.field,
100
+ count=self.num_non_missing,
101
+ missing=self.missing,
102
+ mean=statistics.fmean(self.non_missing),
103
+ min=min(self.non_missing),
104
+ max=max(self.non_missing),
105
+ stdev=statistics.stdev(self.non_missing) if self.num_non_missing > 1 else 0.0,
106
+ )
107
+
108
+ counts = Counter(self.non_missing)
109
+ top_value, top_count = counts.most_common(1)[0] if counts else (None, 0)
110
+ return CategoricalFieldStats(
111
+ field=self.field,
112
+ count=self.num_non_missing,
113
+ missing=self.missing,
114
+ unique=len(counts),
115
+ top=top_value,
116
+ freq=top_count,
117
+ )
118
+
119
+ def add(self, value: Any) -> None:
120
+ self.values.append(value)
121
+ self.num_values += 1
122
+ if value is not None:
123
+ self.non_missing.append(value)
124
+ self.num_non_missing += 1
125
+
126
+
127
+ class DatasetStats:
128
+ """
129
+ Read-only field statistics for a Dataset's current records.
130
+
131
+ Computed fresh on every call from the dataset's public interface (never
132
+ cached), so there's nothing here that can go stale after a mutation --
133
+ the same reasoning that ruled out persistent secondary indexes applies:
134
+ recomputing is cheap and always correct, caching is not free and can lie.
135
+ """
136
+
137
+ def __init__(self, data: list[Any], schema: tuple[Any, ...] | None = None) -> None:
138
+ self._data = data
139
+ self._schema = safe_tuple(schema)
140
+
141
+ def value_counts(self, field: Any) -> ValueCounts:
142
+ """
143
+ How often each distinct value of `field` occurs, most frequent first
144
+ """
145
+ counts = dict(Counter(rec[field] for rec in self._data).most_common())
146
+ return ValueCounts(field=field, counts=counts)
147
+
148
+ def describe(self, field: Any | None = None) -> dict[Any, FieldStats]:
149
+ """
150
+ Summary statistics per field, pandas-.describe()-style.
151
+
152
+ Numeric fields (int/float, excluding bool) get a NumericFieldStats
153
+ (mean/min/max/stdev). Everything else gets a CategoricalFieldStats
154
+ (unique/top/freq). Pass `field` to describe just one; omit to
155
+ describe every field in the schema.
156
+ """
157
+ fields = (field,) if field is not None else self._schema
158
+ stats = {}
159
+ for f in fields:
160
+ field_description = FieldDescription.from_data(self._data, f)
161
+ stats[f] = field_description.to_field_stats()
162
+ return stats
@@ -1,2 +1,2 @@
1
- from .exception_handling import exception_handler
2
- from .warnings import deprecated
1
+ from .exception_handling import exception_handler
2
+ from .warnings import deprecated