sketch-profiler 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- sketch_profiler-0.1.0/PKG-INFO +108 -0
- sketch_profiler-0.1.0/README.md +100 -0
- sketch_profiler-0.1.0/dataprofiler/__init__.py +1 -0
- sketch_profiler-0.1.0/dataprofiler/countminsketch.py +26 -0
- sketch_profiler-0.1.0/dataprofiler/hyperloglog.py +41 -0
- sketch_profiler-0.1.0/dataprofiler/profiler.py +24 -0
- sketch_profiler-0.1.0/pyproject.toml +14 -0
- sketch_profiler-0.1.0/setup.cfg +4 -0
- sketch_profiler-0.1.0/sketch_profiler.egg-info/PKG-INFO +108 -0
- sketch_profiler-0.1.0/sketch_profiler.egg-info/SOURCES.txt +10 -0
- sketch_profiler-0.1.0/sketch_profiler.egg-info/dependency_links.txt +1 -0
- sketch_profiler-0.1.0/sketch_profiler.egg-info/top_level.txt +1 -0
|
@@ -0,0 +1,108 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: sketch-profiler
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Lightweight streaming data profiler using HyperLogLog and Count-Min Sketch
|
|
5
|
+
Author: Jainesh Bharti
|
|
6
|
+
Requires-Python: >=3.8
|
|
7
|
+
Description-Content-Type: text/markdown
|
|
8
|
+
|
|
9
|
+
\# DataProfiler
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
Lightweight, framework-agnostic streaming data profiler that estimates dataset statistics without loading the whole dataset into memory.
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
\## Why
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
Most profiling tools either need a full in-memory pass (pandas `.nunique()`, `.value\_counts()`) or a heavyweight cluster (Spark, DataSketches). DataProfiler works in a single pass, with constant memory, using nothing but plain Python.
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
\## How it works
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
\- \*\*HyperLogLog\*\* — estimates the number of unique values using probabilistic counting (\~2-5% error, constant memory, regardless of dataset size)
|
|
30
|
+
|
|
31
|
+
\- \*\*Count-Min Sketch\*\* — estimates how often any given value appears, without storing every value seen
|
|
32
|
+
|
|
33
|
+
\- Both are combined behind a single `DataProfiler` class, along with null/empty tracking
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
\## Usage
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
```python
|
|
42
|
+
|
|
43
|
+
from dataprofiler import DataProfiler
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
profiler = DataProfiler()
|
|
48
|
+
|
|
49
|
+
data = \["a", "b", "a", None, "c", "a", "", "b", "d", "a"]
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
for value in data:
|
|
54
|
+
|
|
55
|
+
  profiler.update(value)
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
profiler.report()
|
|
60
|
+
|
|
61
|
+
```
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
Output:
|
|
66
|
+
|
|
67
|
+
```
|
|
68
|
+
|
|
69
|
+
Total rows: 10
|
|
70
|
+
|
|
71
|
+
Null percent: 20.0
|
|
72
|
+
|
|
73
|
+
Unique estimate: 4
|
|
74
|
+
|
|
75
|
+
```
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
\## Project structure
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
```
|
|
84
|
+
|
|
85
|
+
dataprofiler/
|
|
86
|
+
|
|
87
|
+
├── hyperloglog.py # cardinality estimation
|
|
88
|
+
|
|
89
|
+
├── countminsketch.py # frequency estimation
|
|
90
|
+
|
|
91
|
+
├── profiler.py # combines both into DataProfiler
|
|
92
|
+
|
|
93
|
+
└── \_\_init\_\_.py
|
|
94
|
+
|
|
95
|
+
```
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
\## Running tests
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
```bash
|
|
104
|
+
|
|
105
|
+
pytest test\_profiler.py -v
|
|
106
|
+
|
|
107
|
+
```
|
|
108
|
+
|
|
@@ -0,0 +1,100 @@
|
|
|
1
|
+
\# DataProfiler
|
|
2
|
+
|
|
3
|
+
|
|
4
|
+
|
|
5
|
+
Lightweight, framework-agnostic streaming data profiler that estimates dataset statistics without loading the whole dataset into memory.
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
\## Why
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
Most profiling tools either need a full in-memory pass (pandas `.nunique()`, `.value\_counts()`) or a heavyweight cluster (Spark, DataSketches). DataProfiler works in a single pass, with constant memory, using nothing but plain Python.
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
\## How it works
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
\- \*\*HyperLogLog\*\* — estimates the number of unique values using probabilistic counting (\~2-5% error, constant memory, regardless of dataset size)
|
|
22
|
+
|
|
23
|
+
\- \*\*Count-Min Sketch\*\* — estimates how often any given value appears, without storing every value seen
|
|
24
|
+
|
|
25
|
+
\- Both are combined behind a single `DataProfiler` class, along with null/empty tracking
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
\## Usage
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
```python
|
|
34
|
+
|
|
35
|
+
from dataprofiler import DataProfiler
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
profiler = DataProfiler()
|
|
40
|
+
|
|
41
|
+
data = \["a", "b", "a", None, "c", "a", "", "b", "d", "a"]
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
for value in data:
|
|
46
|
+
|
|
47
|
+
  profiler.update(value)
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
profiler.report()
|
|
52
|
+
|
|
53
|
+
```
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
Output:
|
|
58
|
+
|
|
59
|
+
```
|
|
60
|
+
|
|
61
|
+
Total rows: 10
|
|
62
|
+
|
|
63
|
+
Null percent: 20.0
|
|
64
|
+
|
|
65
|
+
Unique estimate: 4
|
|
66
|
+
|
|
67
|
+
```
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
\## Project structure
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
```
|
|
76
|
+
|
|
77
|
+
dataprofiler/
|
|
78
|
+
|
|
79
|
+
├── hyperloglog.py # cardinality estimation
|
|
80
|
+
|
|
81
|
+
├── countminsketch.py # frequency estimation
|
|
82
|
+
|
|
83
|
+
├── profiler.py # combines both into DataProfiler
|
|
84
|
+
|
|
85
|
+
└── \_\_init\_\_.py
|
|
86
|
+
|
|
87
|
+
```
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
\## Running tests
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
```bash
|
|
96
|
+
|
|
97
|
+
pytest test\_profiler.py -v
|
|
98
|
+
|
|
99
|
+
```
|
|
100
|
+
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
from .profiler import DataProfiler
|
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
import hashlib
|
|
2
|
+
import random
|
|
3
|
+
|
|
4
|
+
|
|
5
|
+
class CountMinSketch:
|
|
6
|
+
def __init__(self, width=2000, depth=5):
|
|
7
|
+
self.width = width
|
|
8
|
+
self.depth = depth
|
|
9
|
+
self.table = [[0] * width for _ in range(depth)]
|
|
10
|
+
self.seeds = [random.randint(0, 10**6) for _ in range(depth)]
|
|
11
|
+
|
|
12
|
+
def _hash(self, value, seed):
|
|
13
|
+
h = hashlib.md5(f"{seed}-{value}".encode()).hexdigest()
|
|
14
|
+
return int(h, 16) % self.width
|
|
15
|
+
|
|
16
|
+
def add(self, value):
|
|
17
|
+
for i in range(self.depth):
|
|
18
|
+
idx = self._hash(value, self.seeds[i])
|
|
19
|
+
self.table[i][idx] += 1
|
|
20
|
+
|
|
21
|
+
def estimate(self, value):
|
|
22
|
+
counts = []
|
|
23
|
+
for i in range(self.depth):
|
|
24
|
+
idx = self._hash(value, self.seeds[i])
|
|
25
|
+
counts.append(self.table[i][idx])
|
|
26
|
+
return min(counts)
|
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
import hashlib
|
|
2
|
+
import math
|
|
3
|
+
|
|
4
|
+
|
|
5
|
+
class HyperLogLog:
|
|
6
|
+
def __init__(self, b=10):
|
|
7
|
+
self.b = b
|
|
8
|
+
self.m = 2 ** b
|
|
9
|
+
self.buckets = [0] * self.m
|
|
10
|
+
|
|
11
|
+
def _hash(self, value):
|
|
12
|
+
h = hashlib.md5(str(value).encode()).hexdigest()
|
|
13
|
+
return int(h, 16)
|
|
14
|
+
|
|
15
|
+
def _leading_zeros(self, x, bits=118):
|
|
16
|
+
binary = format(x, f'0{bits}b')
|
|
17
|
+
count = 0
|
|
18
|
+
for char in binary:
|
|
19
|
+
if char == '0':
|
|
20
|
+
count += 1
|
|
21
|
+
else:
|
|
22
|
+
break
|
|
23
|
+
return count
|
|
24
|
+
|
|
25
|
+
def add(self, value):
|
|
26
|
+
x = self._hash(value)
|
|
27
|
+
bucket_index = x & (self.m - 1)
|
|
28
|
+
remaining_bits = x >> self.b
|
|
29
|
+
rank = self._leading_zeros(remaining_bits, bits=128 - self.b) + 1
|
|
30
|
+
self.buckets[bucket_index] = max(self.buckets[bucket_index], rank)
|
|
31
|
+
|
|
32
|
+
def count(self):
|
|
33
|
+
alpha = 0.7213 / (1 + 1.079 / self.m)
|
|
34
|
+
raw_estimate = alpha * (self.m ** 2) / sum(2 ** -b for b in self.buckets)
|
|
35
|
+
|
|
36
|
+
if raw_estimate <= 2.5 * self.m:
|
|
37
|
+
zeros = self.buckets.count(0)
|
|
38
|
+
if zeros != 0:
|
|
39
|
+
return int(self.m * math.log(self.m / zeros))
|
|
40
|
+
|
|
41
|
+
return int(raw_estimate)
|
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
from .hyperloglog import HyperLogLog
|
|
2
|
+
from .countminsketch import CountMinSketch
|
|
3
|
+
|
|
4
|
+
|
|
5
|
+
class DataProfiler:
|
|
6
|
+
def __init__(self):
|
|
7
|
+
self.hll = HyperLogLog(b=10)
|
|
8
|
+
self.cms = CountMinSketch()
|
|
9
|
+
self.total_count = 0
|
|
10
|
+
self.null_count = 0
|
|
11
|
+
|
|
12
|
+
def update(self, value):
|
|
13
|
+
self.total_count += 1
|
|
14
|
+
if value is None or value == "":
|
|
15
|
+
self.null_count += 1
|
|
16
|
+
return
|
|
17
|
+
self.hll.add(value)
|
|
18
|
+
self.cms.add(value)
|
|
19
|
+
|
|
20
|
+
def report(self):
|
|
21
|
+
null_pct = (self.null_count / self.total_count) * 100
|
|
22
|
+
print("Total rows:", self.total_count)
|
|
23
|
+
print("Null percent:", round(null_pct, 2))
|
|
24
|
+
print("Unique estimate:", self.hll.count())
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=61.0"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "sketch-profiler"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "Lightweight streaming data profiler using HyperLogLog and Count-Min Sketch"
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.8"
|
|
11
|
+
authors = [{name = "Jainesh Bharti"}]
|
|
12
|
+
|
|
13
|
+
[tool.setuptools]
|
|
14
|
+
packages = ["dataprofiler"]
|
|
@@ -0,0 +1,108 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: sketch-profiler
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Lightweight streaming data profiler using HyperLogLog and Count-Min Sketch
|
|
5
|
+
Author: Jainesh Bharti
|
|
6
|
+
Requires-Python: >=3.8
|
|
7
|
+
Description-Content-Type: text/markdown
|
|
8
|
+
|
|
9
|
+
\# DataProfiler
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
Lightweight, framework-agnostic streaming data profiler that estimates dataset statistics without loading the whole dataset into memory.
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
\## Why
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
Most profiling tools either need a full in-memory pass (pandas `.nunique()`, `.value\_counts()`) or a heavyweight cluster (Spark, DataSketches). DataProfiler works in a single pass, with constant memory, using nothing but plain Python.
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
\## How it works
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
\- \*\*HyperLogLog\*\* — estimates the number of unique values using probabilistic counting (\~2-5% error, constant memory, regardless of dataset size)
|
|
30
|
+
|
|
31
|
+
\- \*\*Count-Min Sketch\*\* — estimates how often any given value appears, without storing every value seen
|
|
32
|
+
|
|
33
|
+
\- Both are combined behind a single `DataProfiler` class, along with null/empty tracking
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
\## Usage
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
```python
|
|
42
|
+
|
|
43
|
+
from dataprofiler import DataProfiler
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
profiler = DataProfiler()
|
|
48
|
+
|
|
49
|
+
data = \["a", "b", "a", None, "c", "a", "", "b", "d", "a"]
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
for value in data:
|
|
54
|
+
|
|
55
|
+
  profiler.update(value)
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
profiler.report()
|
|
60
|
+
|
|
61
|
+
```
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
Output:
|
|
66
|
+
|
|
67
|
+
```
|
|
68
|
+
|
|
69
|
+
Total rows: 10
|
|
70
|
+
|
|
71
|
+
Null percent: 20.0
|
|
72
|
+
|
|
73
|
+
Unique estimate: 4
|
|
74
|
+
|
|
75
|
+
```
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
\## Project structure
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
```
|
|
84
|
+
|
|
85
|
+
dataprofiler/
|
|
86
|
+
|
|
87
|
+
├── hyperloglog.py # cardinality estimation
|
|
88
|
+
|
|
89
|
+
├── countminsketch.py # frequency estimation
|
|
90
|
+
|
|
91
|
+
├── profiler.py # combines both into DataProfiler
|
|
92
|
+
|
|
93
|
+
└── \_\_init\_\_.py
|
|
94
|
+
|
|
95
|
+
```
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
\## Running tests
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
```bash
|
|
104
|
+
|
|
105
|
+
pytest test\_profiler.py -v
|
|
106
|
+
|
|
107
|
+
```
|
|
108
|
+
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
README.md
|
|
2
|
+
pyproject.toml
|
|
3
|
+
dataprofiler/__init__.py
|
|
4
|
+
dataprofiler/countminsketch.py
|
|
5
|
+
dataprofiler/hyperloglog.py
|
|
6
|
+
dataprofiler/profiler.py
|
|
7
|
+
sketch_profiler.egg-info/PKG-INFO
|
|
8
|
+
sketch_profiler.egg-info/SOURCES.txt
|
|
9
|
+
sketch_profiler.egg-info/dependency_links.txt
|
|
10
|
+
sketch_profiler.egg-info/top_level.txt
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
dataprofiler
|