sketch-profiler 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,108 @@
1
+ Metadata-Version: 2.4
2
+ Name: sketch-profiler
3
+ Version: 0.1.0
4
+ Summary: Lightweight streaming data profiler using HyperLogLog and Count-Min Sketch
5
+ Author: Jainesh Bharti
6
+ Requires-Python: >=3.8
7
+ Description-Content-Type: text/markdown
8
+
9
+ \# DataProfiler
10
+
11
+
12
+
13
+ Lightweight, framework-agnostic streaming data profiler that estimates dataset statistics without loading the whole dataset into memory.
14
+
15
+
16
+
17
+ \## Why
18
+
19
+
20
+
21
+ Most profiling tools either need a full in-memory pass (pandas `.nunique()`, `.value\_counts()`) or a heavyweight cluster (Spark, DataSketches). DataProfiler works in a single pass, with constant memory, using nothing but plain Python.
22
+
23
+
24
+
25
+ \## How it works
26
+
27
+
28
+
29
+ \- \*\*HyperLogLog\*\* — estimates the number of unique values using probabilistic counting (\~2-5% error, constant memory, regardless of dataset size)
30
+
31
+ \- \*\*Count-Min Sketch\*\* — estimates how often any given value appears, without storing every value seen
32
+
33
+ \- Both are combined behind a single `DataProfiler` class, along with null/empty tracking
34
+
35
+
36
+
37
+ \## Usage
38
+
39
+
40
+
41
+ ```python
42
+
43
+ from dataprofiler import DataProfiler
44
+
45
+
46
+
47
+ profiler = DataProfiler()
48
+
49
+ data = \["a", "b", "a", None, "c", "a", "", "b", "d", "a"]
50
+
51
+
52
+
53
+ for value in data:
54
+
55
+   profiler.update(value)
56
+
57
+
58
+
59
+ profiler.report()
60
+
61
+ ```
62
+
63
+
64
+
65
+ Output:
66
+
67
+ ```
68
+
69
+ Total rows: 10
70
+
71
+ Null percent: 20.0
72
+
73
+ Unique estimate: 4
74
+
75
+ ```
76
+
77
+
78
+
79
+ \## Project structure
80
+
81
+
82
+
83
+ ```
84
+
85
+ dataprofiler/
86
+
87
+ ├── hyperloglog.py # cardinality estimation
88
+
89
+ ├── countminsketch.py # frequency estimation
90
+
91
+ ├── profiler.py # combines both into DataProfiler
92
+
93
+ └── \_\_init\_\_.py
94
+
95
+ ```
96
+
97
+
98
+
99
+ \## Running tests
100
+
101
+
102
+
103
+ ```bash
104
+
105
+ pytest test\_profiler.py -v
106
+
107
+ ```
108
+
@@ -0,0 +1,100 @@
1
+ \# DataProfiler
2
+
3
+
4
+
5
+ Lightweight, framework-agnostic streaming data profiler that estimates dataset statistics without loading the whole dataset into memory.
6
+
7
+
8
+
9
+ \## Why
10
+
11
+
12
+
13
+ Most profiling tools either need a full in-memory pass (pandas `.nunique()`, `.value\_counts()`) or a heavyweight cluster (Spark, DataSketches). DataProfiler works in a single pass, with constant memory, using nothing but plain Python.
14
+
15
+
16
+
17
+ \## How it works
18
+
19
+
20
+
21
+ \- \*\*HyperLogLog\*\* — estimates the number of unique values using probabilistic counting (\~2-5% error, constant memory, regardless of dataset size)
22
+
23
+ \- \*\*Count-Min Sketch\*\* — estimates how often any given value appears, without storing every value seen
24
+
25
+ \- Both are combined behind a single `DataProfiler` class, along with null/empty tracking
26
+
27
+
28
+
29
+ \## Usage
30
+
31
+
32
+
33
+ ```python
34
+
35
+ from dataprofiler import DataProfiler
36
+
37
+
38
+
39
+ profiler = DataProfiler()
40
+
41
+ data = \["a", "b", "a", None, "c", "a", "", "b", "d", "a"]
42
+
43
+
44
+
45
+ for value in data:
46
+
47
+   profiler.update(value)
48
+
49
+
50
+
51
+ profiler.report()
52
+
53
+ ```
54
+
55
+
56
+
57
+ Output:
58
+
59
+ ```
60
+
61
+ Total rows: 10
62
+
63
+ Null percent: 20.0
64
+
65
+ Unique estimate: 4
66
+
67
+ ```
68
+
69
+
70
+
71
+ \## Project structure
72
+
73
+
74
+
75
+ ```
76
+
77
+ dataprofiler/
78
+
79
+ ├── hyperloglog.py # cardinality estimation
80
+
81
+ ├── countminsketch.py # frequency estimation
82
+
83
+ ├── profiler.py # combines both into DataProfiler
84
+
85
+ └── \_\_init\_\_.py
86
+
87
+ ```
88
+
89
+
90
+
91
+ \## Running tests
92
+
93
+
94
+
95
+ ```bash
96
+
97
+ pytest test\_profiler.py -v
98
+
99
+ ```
100
+
@@ -0,0 +1 @@
1
+ from .profiler import DataProfiler
@@ -0,0 +1,26 @@
1
+ import hashlib
2
+ import random
3
+
4
+
5
+ class CountMinSketch:
6
+ def __init__(self, width=2000, depth=5):
7
+ self.width = width
8
+ self.depth = depth
9
+ self.table = [[0] * width for _ in range(depth)]
10
+ self.seeds = [random.randint(0, 10**6) for _ in range(depth)]
11
+
12
+ def _hash(self, value, seed):
13
+ h = hashlib.md5(f"{seed}-{value}".encode()).hexdigest()
14
+ return int(h, 16) % self.width
15
+
16
+ def add(self, value):
17
+ for i in range(self.depth):
18
+ idx = self._hash(value, self.seeds[i])
19
+ self.table[i][idx] += 1
20
+
21
+ def estimate(self, value):
22
+ counts = []
23
+ for i in range(self.depth):
24
+ idx = self._hash(value, self.seeds[i])
25
+ counts.append(self.table[i][idx])
26
+ return min(counts)
@@ -0,0 +1,41 @@
1
+ import hashlib
2
+ import math
3
+
4
+
5
+ class HyperLogLog:
6
+ def __init__(self, b=10):
7
+ self.b = b
8
+ self.m = 2 ** b
9
+ self.buckets = [0] * self.m
10
+
11
+ def _hash(self, value):
12
+ h = hashlib.md5(str(value).encode()).hexdigest()
13
+ return int(h, 16)
14
+
15
+ def _leading_zeros(self, x, bits=118):
16
+ binary = format(x, f'0{bits}b')
17
+ count = 0
18
+ for char in binary:
19
+ if char == '0':
20
+ count += 1
21
+ else:
22
+ break
23
+ return count
24
+
25
+ def add(self, value):
26
+ x = self._hash(value)
27
+ bucket_index = x & (self.m - 1)
28
+ remaining_bits = x >> self.b
29
+ rank = self._leading_zeros(remaining_bits, bits=128 - self.b) + 1
30
+ self.buckets[bucket_index] = max(self.buckets[bucket_index], rank)
31
+
32
+ def count(self):
33
+ alpha = 0.7213 / (1 + 1.079 / self.m)
34
+ raw_estimate = alpha * (self.m ** 2) / sum(2 ** -b for b in self.buckets)
35
+
36
+ if raw_estimate <= 2.5 * self.m:
37
+ zeros = self.buckets.count(0)
38
+ if zeros != 0:
39
+ return int(self.m * math.log(self.m / zeros))
40
+
41
+ return int(raw_estimate)
@@ -0,0 +1,24 @@
1
+ from .hyperloglog import HyperLogLog
2
+ from .countminsketch import CountMinSketch
3
+
4
+
5
+ class DataProfiler:
6
+ def __init__(self):
7
+ self.hll = HyperLogLog(b=10)
8
+ self.cms = CountMinSketch()
9
+ self.total_count = 0
10
+ self.null_count = 0
11
+
12
+ def update(self, value):
13
+ self.total_count += 1
14
+ if value is None or value == "":
15
+ self.null_count += 1
16
+ return
17
+ self.hll.add(value)
18
+ self.cms.add(value)
19
+
20
+ def report(self):
21
+ null_pct = (self.null_count / self.total_count) * 100
22
+ print("Total rows:", self.total_count)
23
+ print("Null percent:", round(null_pct, 2))
24
+ print("Unique estimate:", self.hll.count())
@@ -0,0 +1,14 @@
1
+ [build-system]
2
+ requires = ["setuptools>=61.0"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "sketch-profiler"
7
+ version = "0.1.0"
8
+ description = "Lightweight streaming data profiler using HyperLogLog and Count-Min Sketch"
9
+ readme = "README.md"
10
+ requires-python = ">=3.8"
11
+ authors = [{name = "Jainesh Bharti"}]
12
+
13
+ [tool.setuptools]
14
+ packages = ["dataprofiler"]
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+
@@ -0,0 +1,108 @@
1
+ Metadata-Version: 2.4
2
+ Name: sketch-profiler
3
+ Version: 0.1.0
4
+ Summary: Lightweight streaming data profiler using HyperLogLog and Count-Min Sketch
5
+ Author: Jainesh Bharti
6
+ Requires-Python: >=3.8
7
+ Description-Content-Type: text/markdown
8
+
9
+ \# DataProfiler
10
+
11
+
12
+
13
+ Lightweight, framework-agnostic streaming data profiler that estimates dataset statistics without loading the whole dataset into memory.
14
+
15
+
16
+
17
+ \## Why
18
+
19
+
20
+
21
+ Most profiling tools either need a full in-memory pass (pandas `.nunique()`, `.value\_counts()`) or a heavyweight cluster (Spark, DataSketches). DataProfiler works in a single pass, with constant memory, using nothing but plain Python.
22
+
23
+
24
+
25
+ \## How it works
26
+
27
+
28
+
29
+ \- \*\*HyperLogLog\*\* — estimates the number of unique values using probabilistic counting (\~2-5% error, constant memory, regardless of dataset size)
30
+
31
+ \- \*\*Count-Min Sketch\*\* — estimates how often any given value appears, without storing every value seen
32
+
33
+ \- Both are combined behind a single `DataProfiler` class, along with null/empty tracking
34
+
35
+
36
+
37
+ \## Usage
38
+
39
+
40
+
41
+ ```python
42
+
43
+ from dataprofiler import DataProfiler
44
+
45
+
46
+
47
+ profiler = DataProfiler()
48
+
49
+ data = \["a", "b", "a", None, "c", "a", "", "b", "d", "a"]
50
+
51
+
52
+
53
+ for value in data:
54
+
55
+ &#x20; profiler.update(value)
56
+
57
+
58
+
59
+ profiler.report()
60
+
61
+ ```
62
+
63
+
64
+
65
+ Output:
66
+
67
+ ```
68
+
69
+ Total rows: 10
70
+
71
+ Null percent: 20.0
72
+
73
+ Unique estimate: 4
74
+
75
+ ```
76
+
77
+
78
+
79
+ \## Project structure
80
+
81
+
82
+
83
+ ```
84
+
85
+ dataprofiler/
86
+
87
+ ├── hyperloglog.py # cardinality estimation
88
+
89
+ ├── countminsketch.py # frequency estimation
90
+
91
+ ├── profiler.py # combines both into DataProfiler
92
+
93
+ └── \_\_init\_\_.py
94
+
95
+ ```
96
+
97
+
98
+
99
+ \## Running tests
100
+
101
+
102
+
103
+ ```bash
104
+
105
+ pytest test\_profiler.py -v
106
+
107
+ ```
108
+
@@ -0,0 +1,10 @@
1
+ README.md
2
+ pyproject.toml
3
+ dataprofiler/__init__.py
4
+ dataprofiler/countminsketch.py
5
+ dataprofiler/hyperloglog.py
6
+ dataprofiler/profiler.py
7
+ sketch_profiler.egg-info/PKG-INFO
8
+ sketch_profiler.egg-info/SOURCES.txt
9
+ sketch_profiler.egg-info/dependency_links.txt
10
+ sketch_profiler.egg-info/top_level.txt
@@ -0,0 +1 @@
1
+ dataprofiler