data-hacks3 0.0.4__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- data_hacks/bar_chart.py +116 -0
- data_hacks/histogram.py +300 -0
- data_hacks/ninety_five_percent.py +59 -0
- data_hacks/run_for.py +62 -0
- data_hacks/sample.py +65 -0
- data_hacks3-0.0.4.data/scripts/bar_chart.py +116 -0
- data_hacks3-0.0.4.data/scripts/histogram.py +300 -0
- data_hacks3-0.0.4.data/scripts/ninety_five_percent.py +59 -0
- data_hacks3-0.0.4.data/scripts/run_for.py +62 -0
- data_hacks3-0.0.4.data/scripts/sample.py +65 -0
- data_hacks3-0.0.4.dist-info/METADATA +18 -0
- data_hacks3-0.0.4.dist-info/RECORD +14 -0
- data_hacks3-0.0.4.dist-info/WHEEL +5 -0
- data_hacks3-0.0.4.dist-info/top_level.txt +1 -0
data_hacks/bar_chart.py
ADDED
|
@@ -0,0 +1,116 @@
|
|
|
1
|
+
#!/usr/bin/env python
|
|
2
|
+
# -*- coding: utf-8 -*-
|
|
3
|
+
#
|
|
4
|
+
# Copyright 2010 Bitly
|
|
5
|
+
#
|
|
6
|
+
# Licensed under the Apache License, Version 2.0 (the "License"); you may
|
|
7
|
+
# not use this file except in compliance with the License. You may obtain
|
|
8
|
+
# a copy of the License at
|
|
9
|
+
#
|
|
10
|
+
# http://www.apache.org/licenses/LICENSE-2.0
|
|
11
|
+
#
|
|
12
|
+
# Unless required by applicable law or agreed to in writing, software
|
|
13
|
+
# distributed under the License is distributed on an "AS IS" BASIS, WITHOUT
|
|
14
|
+
# WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. See the
|
|
15
|
+
# License for the specific language governing permissions and limitations
|
|
16
|
+
# under the License.
|
|
17
|
+
|
|
18
|
+
"""
|
|
19
|
+
Generate an ascii bar chart for input data
|
|
20
|
+
|
|
21
|
+
https://github.com/bitly/data_hacks
|
|
22
|
+
"""
|
|
23
|
+
import sys
|
|
24
|
+
import math
|
|
25
|
+
from collections import defaultdict
|
|
26
|
+
from optparse import OptionParser
|
|
27
|
+
from decimal import Decimal
|
|
28
|
+
|
|
29
|
+
def load_stream(input_stream):
|
|
30
|
+
for line in input_stream:
|
|
31
|
+
clean_line = line.strip()
|
|
32
|
+
if not clean_line:
|
|
33
|
+
# skip empty lines (ie: newlines)
|
|
34
|
+
continue
|
|
35
|
+
if clean_line[0] in ['"', "'"]:
|
|
36
|
+
clean_line = clean_line.strip('"').strip("'")
|
|
37
|
+
if clean_line:
|
|
38
|
+
yield clean_line
|
|
39
|
+
|
|
40
|
+
def run(input_stream, options):
|
|
41
|
+
data = defaultdict(int)
|
|
42
|
+
total = 0
|
|
43
|
+
for row in input_stream:
|
|
44
|
+
if options.agg_key_value:
|
|
45
|
+
kv = row.rstrip().rsplit(None, 1)
|
|
46
|
+
value = int(kv[1])
|
|
47
|
+
data[kv[0]] += value
|
|
48
|
+
total += value
|
|
49
|
+
elif options.agg_value_key:
|
|
50
|
+
kv = row.lstrip().split(None, 1)
|
|
51
|
+
value = int(kv[0])
|
|
52
|
+
data[kv[1]] += value
|
|
53
|
+
total += value
|
|
54
|
+
else:
|
|
55
|
+
data[row] += 1
|
|
56
|
+
total += 1
|
|
57
|
+
|
|
58
|
+
if not data:
|
|
59
|
+
print("Error: no data")
|
|
60
|
+
sys.exit(1)
|
|
61
|
+
|
|
62
|
+
max_length = max([len(key) for key in list(data.keys())])
|
|
63
|
+
max_length = min(max_length, 50)
|
|
64
|
+
value_characters = 80 - max_length
|
|
65
|
+
max_value = max(data.values())
|
|
66
|
+
scale = int(math.ceil(float(max_value) / value_characters))
|
|
67
|
+
scale = max(1, scale)
|
|
68
|
+
|
|
69
|
+
print(("# each " + options.dot + " represents a count of %d. total %d" % (scale, total)))
|
|
70
|
+
|
|
71
|
+
if options.sort_values:
|
|
72
|
+
data = [[value, key] for key, value in list(data.items())]
|
|
73
|
+
data.sort(key=lambda x: x[0], reverse=options.reverse_sort)
|
|
74
|
+
else:
|
|
75
|
+
# sort by keys
|
|
76
|
+
data = [[value, key] for key, value in list(data.items())]
|
|
77
|
+
if options.numeric_sort:
|
|
78
|
+
# keys could be numeric too
|
|
79
|
+
data.sort(key=lambda x: (Decimal(x[1])), reverse=options.reverse_sort)
|
|
80
|
+
else:
|
|
81
|
+
data.sort(key=lambda x: x[1], reverse=options.reverse_sort)
|
|
82
|
+
|
|
83
|
+
str_format = "%" + str(max_length) + "s [%6d] %s%s"
|
|
84
|
+
percentage = ""
|
|
85
|
+
for value, key in data:
|
|
86
|
+
if options.percentage:
|
|
87
|
+
percentage = " (%0.2f%%)" % (100 * Decimal(value) / Decimal(total))
|
|
88
|
+
print((str_format % (key[:max_length], value, int(value / scale) * options.dot, percentage)))
|
|
89
|
+
|
|
90
|
+
if __name__ == "__main__":
|
|
91
|
+
parser = OptionParser()
|
|
92
|
+
parser.usage = "cat data | %prog [options]"
|
|
93
|
+
parser.add_option("-a", "--agg", dest="agg_value_key", default=False, action="store_true",
|
|
94
|
+
help="Two column input format, space seperated with value<space>key")
|
|
95
|
+
parser.add_option("-A", "--agg-key-value", dest="agg_key_value", default=False, action="store_true",
|
|
96
|
+
help="Two column input format, space seperated with key<space>value")
|
|
97
|
+
parser.add_option("-k", "--sort-keys", dest="sort_keys", default=True, action="store_true",
|
|
98
|
+
help="sort by the key [default]")
|
|
99
|
+
parser.add_option("-v", "--sort-values", dest="sort_values", default=False, action="store_true",
|
|
100
|
+
help="sort by the frequence")
|
|
101
|
+
parser.add_option("-r", "--reverse-sort", dest="reverse_sort", default=False, action="store_true",
|
|
102
|
+
help="reverse the sort")
|
|
103
|
+
parser.add_option("-n", "--numeric-sort", dest="numeric_sort", default=False, action="store_true",
|
|
104
|
+
help="sort keys by numeric sequencing")
|
|
105
|
+
parser.add_option("-p", "--percentage", dest="percentage", default=False, action="store_true",
|
|
106
|
+
help="List percentage for each bar")
|
|
107
|
+
parser.add_option("--dot", dest="dot", default='∎', help="Dot representation")
|
|
108
|
+
|
|
109
|
+
(options, args) = parser.parse_args()
|
|
110
|
+
|
|
111
|
+
if sys.stdin.isatty():
|
|
112
|
+
parser.print_usage()
|
|
113
|
+
print("for more help use --help")
|
|
114
|
+
sys.exit(1)
|
|
115
|
+
run(load_stream(sys.stdin), options)
|
|
116
|
+
|
data_hacks/histogram.py
ADDED
|
@@ -0,0 +1,300 @@
|
|
|
1
|
+
#!/usr/bin/env python
|
|
2
|
+
# -*- coding: utf-8 -*-
|
|
3
|
+
#
|
|
4
|
+
# Copyright 2010 Bitly
|
|
5
|
+
#
|
|
6
|
+
# Licensed under the Apache License, Version 2.0 (the "License"); you may
|
|
7
|
+
# not use this file except in compliance with the License. You may obtain
|
|
8
|
+
# a copy of the License at
|
|
9
|
+
#
|
|
10
|
+
# http://www.apache.org/licenses/LICENSE-2.0
|
|
11
|
+
#
|
|
12
|
+
# Unless required by applicable law or agreed to in writing, software
|
|
13
|
+
# distributed under the License is distributed on an "AS IS" BASIS, WITHOUT
|
|
14
|
+
# WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. See the
|
|
15
|
+
# License for the specific language governing permissions and limitations
|
|
16
|
+
# under the License.
|
|
17
|
+
|
|
18
|
+
"""
|
|
19
|
+
Generate a text format histogram
|
|
20
|
+
|
|
21
|
+
This is a loose port to python of the Perl version at
|
|
22
|
+
http://www.pandamatak.com/people/anand/xfer/histo
|
|
23
|
+
|
|
24
|
+
https://github.com/bitly/data_hacks
|
|
25
|
+
"""
|
|
26
|
+
|
|
27
|
+
import sys
|
|
28
|
+
from decimal import Decimal
|
|
29
|
+
import logging
|
|
30
|
+
import math
|
|
31
|
+
from optparse import OptionParser
|
|
32
|
+
from collections import namedtuple
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
class MVSD(object):
|
|
36
|
+
"A class that calculates a running Mean / Variance / Standard Deviation"
|
|
37
|
+
def __init__(self):
|
|
38
|
+
self.is_started = False
|
|
39
|
+
self.ss = Decimal(0) # (running) sum of square deviations from mean
|
|
40
|
+
self.m = Decimal(0) # (running) mean
|
|
41
|
+
self.total_w = Decimal(0) # weight of items seen
|
|
42
|
+
|
|
43
|
+
def add(self, x, w=1):
|
|
44
|
+
"add another datapoint to the Mean / Variance / Standard Deviation"
|
|
45
|
+
if not isinstance(x, Decimal):
|
|
46
|
+
x = Decimal(x)
|
|
47
|
+
if not self.is_started:
|
|
48
|
+
self.m = x
|
|
49
|
+
self.ss = Decimal(0)
|
|
50
|
+
self.total_w = w
|
|
51
|
+
self.is_started = True
|
|
52
|
+
else:
|
|
53
|
+
temp_w = self.total_w + w
|
|
54
|
+
self.ss += (self.total_w * w * (x - self.m) *
|
|
55
|
+
(x - self.m)) / temp_w
|
|
56
|
+
self.m += (x - self.m) / temp_w
|
|
57
|
+
self.total_w = temp_w
|
|
58
|
+
|
|
59
|
+
def var(self):
|
|
60
|
+
return self.ss / self.total_w
|
|
61
|
+
|
|
62
|
+
def sd(self):
|
|
63
|
+
return math.sqrt(self.var())
|
|
64
|
+
|
|
65
|
+
def mean(self):
|
|
66
|
+
return self.m
|
|
67
|
+
|
|
68
|
+
DataPoint = namedtuple('DataPoint', ['value', 'count'])
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def test_mvsd():
|
|
72
|
+
mvsd = MVSD()
|
|
73
|
+
for x in range(10):
|
|
74
|
+
mvsd.add(x)
|
|
75
|
+
|
|
76
|
+
assert '%.2f' % mvsd.mean() == "4.50"
|
|
77
|
+
assert '%.2f' % mvsd.var() == "8.25"
|
|
78
|
+
assert '%.14f' % mvsd.sd() == "2.87228132326901"
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def load_stream(input_stream, agg_value_key, agg_key_value):
|
|
82
|
+
for line in input_stream:
|
|
83
|
+
clean_line = line.strip()
|
|
84
|
+
if not clean_line:
|
|
85
|
+
# skip empty lines (ie: newlines)
|
|
86
|
+
continue
|
|
87
|
+
if clean_line[0] in ['"', "'"]:
|
|
88
|
+
clean_line = clean_line.strip("\"'")
|
|
89
|
+
try:
|
|
90
|
+
if agg_key_value:
|
|
91
|
+
key, value = clean_line.rstrip().rsplit(None, 1)
|
|
92
|
+
yield DataPoint(Decimal(key), int(value))
|
|
93
|
+
elif agg_value_key:
|
|
94
|
+
value, key = clean_line.lstrip().split(None, 1)
|
|
95
|
+
yield DataPoint(Decimal(key), int(value))
|
|
96
|
+
else:
|
|
97
|
+
yield DataPoint(Decimal(clean_line), 1)
|
|
98
|
+
except:
|
|
99
|
+
logging.exception('failed %r', line)
|
|
100
|
+
print("invalid line %r" % line, file=sys.stderr)
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
def median(values, key=None):
|
|
104
|
+
if not key:
|
|
105
|
+
key = lambda x: x # identity; py3 map() does not accept None
|
|
106
|
+
length = len(values)
|
|
107
|
+
if length % 2:
|
|
108
|
+
median_indeces = [int(length/2)]
|
|
109
|
+
else:
|
|
110
|
+
median_indeces = [int(length/2)-1, int(length/2)]
|
|
111
|
+
|
|
112
|
+
values = sorted(values, key=key)
|
|
113
|
+
return sum(map(key,
|
|
114
|
+
[values[i] for i in median_indeces])) / len(median_indeces)
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
def test_median():
|
|
118
|
+
assert 6 == median([8, 7, 9, 1, 2, 6, 3]) # odd-sized list
|
|
119
|
+
assert 4.5 == median([4, 5, 2, 1, 9, 10]) # even-sized int list. (4+5)/2 = 4.5
|
|
120
|
+
# even-sized float list. (4.0+5)/2 = 4.5
|
|
121
|
+
assert "4.50" == "%.2f" % median([4.0, 5, 2, 1, 9, 10])
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def histogram(stream, options):
|
|
125
|
+
"""
|
|
126
|
+
Loop over the stream and add each entry to the dataset, printing out at the
|
|
127
|
+
end.
|
|
128
|
+
|
|
129
|
+
stream yields Decimal()
|
|
130
|
+
"""
|
|
131
|
+
if not options.min or not options.max:
|
|
132
|
+
# glob the iterator here so we can do min/max on it
|
|
133
|
+
data = list(stream)
|
|
134
|
+
else:
|
|
135
|
+
data = stream
|
|
136
|
+
bucket_scale = 1
|
|
137
|
+
|
|
138
|
+
if options.min:
|
|
139
|
+
min_v = Decimal(options.min)
|
|
140
|
+
else:
|
|
141
|
+
min_v = min(data, key=lambda x: x.value)
|
|
142
|
+
min_v = min_v.value
|
|
143
|
+
if options.max:
|
|
144
|
+
max_v = Decimal(options.max)
|
|
145
|
+
else:
|
|
146
|
+
max_v = max(data, key=lambda x: x.value)
|
|
147
|
+
max_v = max_v.value
|
|
148
|
+
|
|
149
|
+
if not max_v > min_v:
|
|
150
|
+
raise ValueError('max must be > min. max:%s min:%s' % (max_v, min_v))
|
|
151
|
+
diff = max_v - min_v
|
|
152
|
+
|
|
153
|
+
boundaries = []
|
|
154
|
+
bucket_counts = []
|
|
155
|
+
buckets = 0
|
|
156
|
+
|
|
157
|
+
if options.custbuckets:
|
|
158
|
+
bound = options.custbuckets.split(',')
|
|
159
|
+
bound_sort = sorted(map(Decimal, bound))
|
|
160
|
+
|
|
161
|
+
# if the last value is smaller than the maximum, replace it
|
|
162
|
+
if bound_sort[-1] < max_v:
|
|
163
|
+
bound_sort[-1] = max_v
|
|
164
|
+
|
|
165
|
+
# iterate through the sorted list and append to boundaries
|
|
166
|
+
for x in bound_sort:
|
|
167
|
+
if x >= min_v and x <= max_v:
|
|
168
|
+
boundaries.append(x)
|
|
169
|
+
elif x >= max_v:
|
|
170
|
+
boundaries.append(max_v)
|
|
171
|
+
break
|
|
172
|
+
|
|
173
|
+
# beware: the min_v is not included in the boundaries,
|
|
174
|
+
# so no need to do a -1!
|
|
175
|
+
bucket_counts = [0 for x in range(len(boundaries))]
|
|
176
|
+
buckets = len(boundaries)
|
|
177
|
+
elif options.logscale:
|
|
178
|
+
buckets = options.buckets and int(options.buckets) or 10
|
|
179
|
+
if buckets <= 0:
|
|
180
|
+
raise ValueError('# of buckets must be > 0')
|
|
181
|
+
|
|
182
|
+
def first_bucket_size(k, n):
|
|
183
|
+
r"""Logarithmic buckets means, the size of bucket i+1 is twice
|
|
184
|
+
the size of bucket i.
|
|
185
|
+
For k+1 buckets whose sum is n, we have
|
|
186
|
+
(note, k+1 buckets, since 0 is counted as well):
|
|
187
|
+
\sum_{i=0}^{k} x*2^i = n
|
|
188
|
+
x * \sum_{i=0}^{k} 2^i = n
|
|
189
|
+
x * (2^{k+1} - 1) = n
|
|
190
|
+
x = n/(2^{k+1} - 1)
|
|
191
|
+
"""
|
|
192
|
+
return n/(2**(k+1)-1)
|
|
193
|
+
|
|
194
|
+
def log_steps(k, n):
|
|
195
|
+
"k logarithmic steps whose sum is n"
|
|
196
|
+
x = first_bucket_size(k-1, n)
|
|
197
|
+
sum = 0
|
|
198
|
+
for i in range(k):
|
|
199
|
+
sum += 2**i * x
|
|
200
|
+
yield sum
|
|
201
|
+
bucket_counts = [0 for x in range(buckets)]
|
|
202
|
+
for step in log_steps(buckets, diff):
|
|
203
|
+
boundaries.append(min_v + step)
|
|
204
|
+
else:
|
|
205
|
+
buckets = options.buckets and int(options.buckets) or 10
|
|
206
|
+
if buckets <= 0:
|
|
207
|
+
raise ValueError('# of buckets must be > 0')
|
|
208
|
+
step = diff / buckets
|
|
209
|
+
bucket_counts = [0 for x in range(buckets)]
|
|
210
|
+
for x in range(buckets):
|
|
211
|
+
boundaries.append(min_v + (step * (x + 1)))
|
|
212
|
+
|
|
213
|
+
skipped = 0
|
|
214
|
+
samples = 0
|
|
215
|
+
mvsd = MVSD()
|
|
216
|
+
accepted_data = []
|
|
217
|
+
for record in data:
|
|
218
|
+
samples += record.count
|
|
219
|
+
if options.mvsd:
|
|
220
|
+
mvsd.add(record.value, record.count)
|
|
221
|
+
accepted_data.append(record)
|
|
222
|
+
# find the bucket this goes in
|
|
223
|
+
if record.value < min_v or record.value > max_v:
|
|
224
|
+
skipped += record.count
|
|
225
|
+
continue
|
|
226
|
+
for bucket_postion, boundary in enumerate(boundaries):
|
|
227
|
+
if record.value <= boundary:
|
|
228
|
+
bucket_counts[bucket_postion] += record.count
|
|
229
|
+
break
|
|
230
|
+
|
|
231
|
+
# auto-pick the hash scale
|
|
232
|
+
if max(bucket_counts) > 75:
|
|
233
|
+
bucket_scale = int(max(bucket_counts) / 75)
|
|
234
|
+
|
|
235
|
+
print(("# NumSamples = %d; Min = %0.2f; Max = %0.2f" %
|
|
236
|
+
(samples, min_v, max_v)))
|
|
237
|
+
if skipped:
|
|
238
|
+
print(("# %d value%s outside of min/max" %
|
|
239
|
+
(skipped, skipped > 1 and 's' or '')))
|
|
240
|
+
if options.mvsd:
|
|
241
|
+
print(("# Mean = %f; Variance = %f; SD = %f; Median %f" %
|
|
242
|
+
(mvsd.mean(), mvsd.var(), mvsd.sd(),
|
|
243
|
+
median(accepted_data, key=lambda x: x.value))))
|
|
244
|
+
print(("# each " + options.dot + " represents a count of %d" % bucket_scale))
|
|
245
|
+
|
|
246
|
+
bucket_min = min_v
|
|
247
|
+
bucket_max = min_v
|
|
248
|
+
percentage = ""
|
|
249
|
+
format_string = options.format + ' - ' + options.format + ' [%6d]: %s%s'
|
|
250
|
+
for bucket in range(buckets):
|
|
251
|
+
bucket_min = bucket_max
|
|
252
|
+
bucket_max = boundaries[bucket]
|
|
253
|
+
bucket_count = bucket_counts[bucket]
|
|
254
|
+
star_count = 0
|
|
255
|
+
if bucket_count:
|
|
256
|
+
star_count = int(bucket_count / bucket_scale)
|
|
257
|
+
if options.percentage:
|
|
258
|
+
percentage = " (%0.2f%%)" % (100 * Decimal(bucket_count) /
|
|
259
|
+
Decimal(samples))
|
|
260
|
+
print((format_string % (bucket_min, bucket_max, bucket_count, options.dot * star_count, percentage)))
|
|
261
|
+
|
|
262
|
+
|
|
263
|
+
if __name__ == "__main__":
|
|
264
|
+
parser = OptionParser()
|
|
265
|
+
parser.usage = "cat data | %prog [options]"
|
|
266
|
+
parser.add_option("-a", "--agg", dest="agg_value_key", default=False,
|
|
267
|
+
action="store_true", help="Two column input format, " +
|
|
268
|
+
"space seperated with value<space>key")
|
|
269
|
+
parser.add_option("-A", "--agg-key-value", dest="agg_key_value",
|
|
270
|
+
default=False, action="store_true", help="Two column " +
|
|
271
|
+
"input format, space seperated with key<space>value")
|
|
272
|
+
parser.add_option("-m", "--min", dest="min",
|
|
273
|
+
help="minimum value for graph")
|
|
274
|
+
parser.add_option("-x", "--max", dest="max",
|
|
275
|
+
help="maximum value for graph")
|
|
276
|
+
parser.add_option("-b", "--buckets", dest="buckets",
|
|
277
|
+
help="Number of buckets to use for the histogram")
|
|
278
|
+
parser.add_option("-l", "--logscale", dest="logscale", default=False,
|
|
279
|
+
action="store_true",
|
|
280
|
+
help="Buckets grow in logarithmic scale")
|
|
281
|
+
parser.add_option("-B", "--custom-buckets", dest="custbuckets",
|
|
282
|
+
help="Comma seperated list of bucket " +
|
|
283
|
+
"edges for the histogram")
|
|
284
|
+
parser.add_option("--no-mvsd", dest="mvsd", action="store_false",
|
|
285
|
+
default=True, help="Disable the calculation of Mean, " +
|
|
286
|
+
"Variance and SD (improves performance)")
|
|
287
|
+
parser.add_option("-f", "--bucket-format", dest="format", default="%10.4f",
|
|
288
|
+
help="format for bucket numbers")
|
|
289
|
+
parser.add_option("-p", "--percentage", dest="percentage", default=False,
|
|
290
|
+
action="store_true", help="List percentage for each bar")
|
|
291
|
+
parser.add_option("--dot", dest="dot", default='∎', help="Dot representation")
|
|
292
|
+
|
|
293
|
+
(options, args) = parser.parse_args()
|
|
294
|
+
if sys.stdin.isatty():
|
|
295
|
+
# if isatty() that means it's run without anything piped into it
|
|
296
|
+
parser.print_usage()
|
|
297
|
+
print("for more help use --help")
|
|
298
|
+
sys.exit(1)
|
|
299
|
+
histogram(load_stream(sys.stdin, options.agg_value_key,
|
|
300
|
+
options.agg_key_value), options)
|
|
@@ -0,0 +1,59 @@
|
|
|
1
|
+
#!/usr/bin/env python
|
|
2
|
+
#
|
|
3
|
+
# Copyright 2010 Bitly
|
|
4
|
+
#
|
|
5
|
+
# Licensed under the Apache License, Version 2.0 (the "License"); you may
|
|
6
|
+
# not use this file except in compliance with the License. You may obtain
|
|
7
|
+
# a copy of the License at
|
|
8
|
+
#
|
|
9
|
+
# http://www.apache.org/licenses/LICENSE-2.0
|
|
10
|
+
#
|
|
11
|
+
# Unless required by applicable law or agreed to in writing, software
|
|
12
|
+
# distributed under the License is distributed on an "AS IS" BASIS, WITHOUT
|
|
13
|
+
# WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. See the
|
|
14
|
+
# License for the specific language governing permissions and limitations
|
|
15
|
+
# under the License.
|
|
16
|
+
|
|
17
|
+
"""
|
|
18
|
+
Calculate the 95% time from a list of times given on stdin
|
|
19
|
+
|
|
20
|
+
https://github.com/bitly/data_hacks
|
|
21
|
+
"""
|
|
22
|
+
|
|
23
|
+
import sys
|
|
24
|
+
import os
|
|
25
|
+
from decimal import Decimal
|
|
26
|
+
|
|
27
|
+
def run():
|
|
28
|
+
count = 0
|
|
29
|
+
data = {}
|
|
30
|
+
for line in sys.stdin:
|
|
31
|
+
line = line.strip()
|
|
32
|
+
if not line:
|
|
33
|
+
# skip empty lines (ie: newlines)
|
|
34
|
+
continue
|
|
35
|
+
try:
|
|
36
|
+
t = Decimal(line)
|
|
37
|
+
count +=1
|
|
38
|
+
data[t] = data.get(t, 0) + 1
|
|
39
|
+
except:
|
|
40
|
+
print("invalid line %r" % line, file=sys.stderr)
|
|
41
|
+
print(calc_95(data, count))
|
|
42
|
+
|
|
43
|
+
def calc_95(data, count):
|
|
44
|
+
# find the time it took for x entry, where x is the threshold
|
|
45
|
+
threshold = Decimal(count) * Decimal('.95')
|
|
46
|
+
start = Decimal(0)
|
|
47
|
+
times = list(data.keys())
|
|
48
|
+
times.sort()
|
|
49
|
+
for t in times:
|
|
50
|
+
# increment our count by the # of items in this time bucket
|
|
51
|
+
start += data[t]
|
|
52
|
+
if start > threshold:
|
|
53
|
+
return t
|
|
54
|
+
|
|
55
|
+
if __name__ == "__main__":
|
|
56
|
+
if sys.stdin.isatty() or '--help' in sys.argv or '-h' in sys.argv:
|
|
57
|
+
print("Usage: cat data | %s" % os.path.basename(sys.argv[0]))
|
|
58
|
+
sys.exit(1)
|
|
59
|
+
run()
|
data_hacks/run_for.py
ADDED
|
@@ -0,0 +1,62 @@
|
|
|
1
|
+
#!/usr/bin/env python
|
|
2
|
+
#
|
|
3
|
+
# Copyright 2010 Bitly
|
|
4
|
+
#
|
|
5
|
+
# Licensed under the Apache License, Version 2.0 (the "License"); you may
|
|
6
|
+
# not use this file except in compliance with the License. You may obtain
|
|
7
|
+
# a copy of the License at
|
|
8
|
+
#
|
|
9
|
+
# http://www.apache.org/licenses/LICENSE-2.0
|
|
10
|
+
#
|
|
11
|
+
# Unless required by applicable law or agreed to in writing, software
|
|
12
|
+
# distributed under the License is distributed on an "AS IS" BASIS, WITHOUT
|
|
13
|
+
# WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. See the
|
|
14
|
+
# License for the specific language governing permissions and limitations
|
|
15
|
+
# under the License.
|
|
16
|
+
|
|
17
|
+
"""
|
|
18
|
+
Pass through data for a specified amount of time
|
|
19
|
+
|
|
20
|
+
https://github.com/bitly/data_hacks
|
|
21
|
+
"""
|
|
22
|
+
|
|
23
|
+
import time
|
|
24
|
+
import sys
|
|
25
|
+
import os
|
|
26
|
+
|
|
27
|
+
def getruntime(arg):
|
|
28
|
+
if not arg:
|
|
29
|
+
return
|
|
30
|
+
suffix = arg[-1]
|
|
31
|
+
base = int(arg[:-1])
|
|
32
|
+
if suffix == "s":
|
|
33
|
+
return base
|
|
34
|
+
elif suffix == "m":
|
|
35
|
+
return base * 60
|
|
36
|
+
elif suffix == "h":
|
|
37
|
+
return base * 60 * 60
|
|
38
|
+
elif suffix == "d":
|
|
39
|
+
return base * 60 * 60 * 24
|
|
40
|
+
else:
|
|
41
|
+
print("invalid time suffix %r. must be one of s,m,h,d" % arg, file=sys.stderr)
|
|
42
|
+
|
|
43
|
+
def run(runtime):
|
|
44
|
+
end = time.time() + runtime
|
|
45
|
+
for line in sys.stdin:
|
|
46
|
+
sys.stdout.write(line)
|
|
47
|
+
if time.time() > end:
|
|
48
|
+
return
|
|
49
|
+
|
|
50
|
+
if __name__ == "__main__":
|
|
51
|
+
usage = "Usage: tail -f access.log | %s [time] | ..." % os.path.basename(sys.argv[0])
|
|
52
|
+
help = "time can be in the format 10s, 10m, 10h, etc"
|
|
53
|
+
if sys.stdin.isatty():
|
|
54
|
+
print(usage)
|
|
55
|
+
print(help)
|
|
56
|
+
sys.exit(1)
|
|
57
|
+
|
|
58
|
+
runtime = getruntime(sys.argv[-1])
|
|
59
|
+
if not runtime:
|
|
60
|
+
print(usage)
|
|
61
|
+
sys.exit(1)
|
|
62
|
+
run(runtime)
|
data_hacks/sample.py
ADDED
|
@@ -0,0 +1,65 @@
|
|
|
1
|
+
#!/usr/bin/env python
|
|
2
|
+
#
|
|
3
|
+
# Copyright 2010 Bitly
|
|
4
|
+
#
|
|
5
|
+
# Licensed under the Apache License, Version 2.0 (the "License"); you may
|
|
6
|
+
# not use this file except in compliance with the License. You may obtain
|
|
7
|
+
# a copy of the License at
|
|
8
|
+
#
|
|
9
|
+
# http://www.apache.org/licenses/LICENSE-2.0
|
|
10
|
+
#
|
|
11
|
+
# Unless required by applicable law or agreed to in writing, software
|
|
12
|
+
# distributed under the License is distributed on an "AS IS" BASIS, WITHOUT
|
|
13
|
+
# WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. See the
|
|
14
|
+
# License for the specific language governing permissions and limitations
|
|
15
|
+
# under the License.
|
|
16
|
+
|
|
17
|
+
"""
|
|
18
|
+
Pass through a sampled percentage of data
|
|
19
|
+
|
|
20
|
+
https://github.com/bitly/data_hacks
|
|
21
|
+
"""
|
|
22
|
+
|
|
23
|
+
import sys
|
|
24
|
+
import random
|
|
25
|
+
from optparse import OptionParser
|
|
26
|
+
from decimal import Decimal
|
|
27
|
+
|
|
28
|
+
def run(sample_rate):
|
|
29
|
+
input_stream = sys.stdin
|
|
30
|
+
for line in input_stream:
|
|
31
|
+
if random.randint(1,100) <= sample_rate:
|
|
32
|
+
sys.stdout.write(line)
|
|
33
|
+
|
|
34
|
+
def get_sample_rate(rate_string):
|
|
35
|
+
""" return a rate as a percentage"""
|
|
36
|
+
if rate_string.endswith("%"):
|
|
37
|
+
rate = int(rate_string[:-1])
|
|
38
|
+
elif '/' in rate_string:
|
|
39
|
+
x, y = rate_string.split('/')
|
|
40
|
+
rate = Decimal(x) / (Decimal(y) * Decimal('1.0'))
|
|
41
|
+
rate = int(rate * 100)
|
|
42
|
+
else:
|
|
43
|
+
raise ValueError("rate %r is invalid rate format must be '10%%' or '1/10'" % rate_string)
|
|
44
|
+
if rate < 1 or rate > 100:
|
|
45
|
+
raise ValueError('rate %r must be 1%% <= rate <= 100%% ' % rate_string)
|
|
46
|
+
return rate
|
|
47
|
+
|
|
48
|
+
if __name__ == "__main__":
|
|
49
|
+
parser = OptionParser(usage="cat data | %prog [options] [sample_rate]")
|
|
50
|
+
parser.add_option("--verbose", dest="verbose", default=False, action="store_true")
|
|
51
|
+
(options, args) = parser.parse_args()
|
|
52
|
+
|
|
53
|
+
if not args or sys.stdin.isatty():
|
|
54
|
+
parser.print_usage()
|
|
55
|
+
sys.exit(1)
|
|
56
|
+
|
|
57
|
+
try:
|
|
58
|
+
sample_rate = get_sample_rate(sys.argv[-1])
|
|
59
|
+
except ValueError as e:
|
|
60
|
+
print(e, file=sys.stderr)
|
|
61
|
+
parser.print_usage()
|
|
62
|
+
sys.exit(1)
|
|
63
|
+
if options.verbose:
|
|
64
|
+
print("Sample rate is %d%%" % sample_rate, file=sys.stderr)
|
|
65
|
+
run(sample_rate)
|
|
@@ -0,0 +1,116 @@
|
|
|
1
|
+
#!python
|
|
2
|
+
# -*- coding: utf-8 -*-
|
|
3
|
+
#
|
|
4
|
+
# Copyright 2010 Bitly
|
|
5
|
+
#
|
|
6
|
+
# Licensed under the Apache License, Version 2.0 (the "License"); you may
|
|
7
|
+
# not use this file except in compliance with the License. You may obtain
|
|
8
|
+
# a copy of the License at
|
|
9
|
+
#
|
|
10
|
+
# http://www.apache.org/licenses/LICENSE-2.0
|
|
11
|
+
#
|
|
12
|
+
# Unless required by applicable law or agreed to in writing, software
|
|
13
|
+
# distributed under the License is distributed on an "AS IS" BASIS, WITHOUT
|
|
14
|
+
# WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. See the
|
|
15
|
+
# License for the specific language governing permissions and limitations
|
|
16
|
+
# under the License.
|
|
17
|
+
|
|
18
|
+
"""
|
|
19
|
+
Generate an ascii bar chart for input data
|
|
20
|
+
|
|
21
|
+
https://github.com/bitly/data_hacks
|
|
22
|
+
"""
|
|
23
|
+
import sys
|
|
24
|
+
import math
|
|
25
|
+
from collections import defaultdict
|
|
26
|
+
from optparse import OptionParser
|
|
27
|
+
from decimal import Decimal
|
|
28
|
+
|
|
29
|
+
def load_stream(input_stream):
|
|
30
|
+
for line in input_stream:
|
|
31
|
+
clean_line = line.strip()
|
|
32
|
+
if not clean_line:
|
|
33
|
+
# skip empty lines (ie: newlines)
|
|
34
|
+
continue
|
|
35
|
+
if clean_line[0] in ['"', "'"]:
|
|
36
|
+
clean_line = clean_line.strip('"').strip("'")
|
|
37
|
+
if clean_line:
|
|
38
|
+
yield clean_line
|
|
39
|
+
|
|
40
|
+
def run(input_stream, options):
|
|
41
|
+
data = defaultdict(int)
|
|
42
|
+
total = 0
|
|
43
|
+
for row in input_stream:
|
|
44
|
+
if options.agg_key_value:
|
|
45
|
+
kv = row.rstrip().rsplit(None, 1)
|
|
46
|
+
value = int(kv[1])
|
|
47
|
+
data[kv[0]] += value
|
|
48
|
+
total += value
|
|
49
|
+
elif options.agg_value_key:
|
|
50
|
+
kv = row.lstrip().split(None, 1)
|
|
51
|
+
value = int(kv[0])
|
|
52
|
+
data[kv[1]] += value
|
|
53
|
+
total += value
|
|
54
|
+
else:
|
|
55
|
+
data[row] += 1
|
|
56
|
+
total += 1
|
|
57
|
+
|
|
58
|
+
if not data:
|
|
59
|
+
print("Error: no data")
|
|
60
|
+
sys.exit(1)
|
|
61
|
+
|
|
62
|
+
max_length = max([len(key) for key in list(data.keys())])
|
|
63
|
+
max_length = min(max_length, 50)
|
|
64
|
+
value_characters = 80 - max_length
|
|
65
|
+
max_value = max(data.values())
|
|
66
|
+
scale = int(math.ceil(float(max_value) / value_characters))
|
|
67
|
+
scale = max(1, scale)
|
|
68
|
+
|
|
69
|
+
print(("# each " + options.dot + " represents a count of %d. total %d" % (scale, total)))
|
|
70
|
+
|
|
71
|
+
if options.sort_values:
|
|
72
|
+
data = [[value, key] for key, value in list(data.items())]
|
|
73
|
+
data.sort(key=lambda x: x[0], reverse=options.reverse_sort)
|
|
74
|
+
else:
|
|
75
|
+
# sort by keys
|
|
76
|
+
data = [[value, key] for key, value in list(data.items())]
|
|
77
|
+
if options.numeric_sort:
|
|
78
|
+
# keys could be numeric too
|
|
79
|
+
data.sort(key=lambda x: (Decimal(x[1])), reverse=options.reverse_sort)
|
|
80
|
+
else:
|
|
81
|
+
data.sort(key=lambda x: x[1], reverse=options.reverse_sort)
|
|
82
|
+
|
|
83
|
+
str_format = "%" + str(max_length) + "s [%6d] %s%s"
|
|
84
|
+
percentage = ""
|
|
85
|
+
for value, key in data:
|
|
86
|
+
if options.percentage:
|
|
87
|
+
percentage = " (%0.2f%%)" % (100 * Decimal(value) / Decimal(total))
|
|
88
|
+
print((str_format % (key[:max_length], value, int(value / scale) * options.dot, percentage)))
|
|
89
|
+
|
|
90
|
+
if __name__ == "__main__":
|
|
91
|
+
parser = OptionParser()
|
|
92
|
+
parser.usage = "cat data | %prog [options]"
|
|
93
|
+
parser.add_option("-a", "--agg", dest="agg_value_key", default=False, action="store_true",
|
|
94
|
+
help="Two column input format, space seperated with value<space>key")
|
|
95
|
+
parser.add_option("-A", "--agg-key-value", dest="agg_key_value", default=False, action="store_true",
|
|
96
|
+
help="Two column input format, space seperated with key<space>value")
|
|
97
|
+
parser.add_option("-k", "--sort-keys", dest="sort_keys", default=True, action="store_true",
|
|
98
|
+
help="sort by the key [default]")
|
|
99
|
+
parser.add_option("-v", "--sort-values", dest="sort_values", default=False, action="store_true",
|
|
100
|
+
help="sort by the frequence")
|
|
101
|
+
parser.add_option("-r", "--reverse-sort", dest="reverse_sort", default=False, action="store_true",
|
|
102
|
+
help="reverse the sort")
|
|
103
|
+
parser.add_option("-n", "--numeric-sort", dest="numeric_sort", default=False, action="store_true",
|
|
104
|
+
help="sort keys by numeric sequencing")
|
|
105
|
+
parser.add_option("-p", "--percentage", dest="percentage", default=False, action="store_true",
|
|
106
|
+
help="List percentage for each bar")
|
|
107
|
+
parser.add_option("--dot", dest="dot", default='∎', help="Dot representation")
|
|
108
|
+
|
|
109
|
+
(options, args) = parser.parse_args()
|
|
110
|
+
|
|
111
|
+
if sys.stdin.isatty():
|
|
112
|
+
parser.print_usage()
|
|
113
|
+
print("for more help use --help")
|
|
114
|
+
sys.exit(1)
|
|
115
|
+
run(load_stream(sys.stdin), options)
|
|
116
|
+
|
|
@@ -0,0 +1,300 @@
|
|
|
1
|
+
#!python
|
|
2
|
+
# -*- coding: utf-8 -*-
|
|
3
|
+
#
|
|
4
|
+
# Copyright 2010 Bitly
|
|
5
|
+
#
|
|
6
|
+
# Licensed under the Apache License, Version 2.0 (the "License"); you may
|
|
7
|
+
# not use this file except in compliance with the License. You may obtain
|
|
8
|
+
# a copy of the License at
|
|
9
|
+
#
|
|
10
|
+
# http://www.apache.org/licenses/LICENSE-2.0
|
|
11
|
+
#
|
|
12
|
+
# Unless required by applicable law or agreed to in writing, software
|
|
13
|
+
# distributed under the License is distributed on an "AS IS" BASIS, WITHOUT
|
|
14
|
+
# WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. See the
|
|
15
|
+
# License for the specific language governing permissions and limitations
|
|
16
|
+
# under the License.
|
|
17
|
+
|
|
18
|
+
"""
|
|
19
|
+
Generate a text format histogram
|
|
20
|
+
|
|
21
|
+
This is a loose port to python of the Perl version at
|
|
22
|
+
http://www.pandamatak.com/people/anand/xfer/histo
|
|
23
|
+
|
|
24
|
+
https://github.com/bitly/data_hacks
|
|
25
|
+
"""
|
|
26
|
+
|
|
27
|
+
import sys
|
|
28
|
+
from decimal import Decimal
|
|
29
|
+
import logging
|
|
30
|
+
import math
|
|
31
|
+
from optparse import OptionParser
|
|
32
|
+
from collections import namedtuple
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
class MVSD(object):
|
|
36
|
+
"A class that calculates a running Mean / Variance / Standard Deviation"
|
|
37
|
+
def __init__(self):
|
|
38
|
+
self.is_started = False
|
|
39
|
+
self.ss = Decimal(0) # (running) sum of square deviations from mean
|
|
40
|
+
self.m = Decimal(0) # (running) mean
|
|
41
|
+
self.total_w = Decimal(0) # weight of items seen
|
|
42
|
+
|
|
43
|
+
def add(self, x, w=1):
|
|
44
|
+
"add another datapoint to the Mean / Variance / Standard Deviation"
|
|
45
|
+
if not isinstance(x, Decimal):
|
|
46
|
+
x = Decimal(x)
|
|
47
|
+
if not self.is_started:
|
|
48
|
+
self.m = x
|
|
49
|
+
self.ss = Decimal(0)
|
|
50
|
+
self.total_w = w
|
|
51
|
+
self.is_started = True
|
|
52
|
+
else:
|
|
53
|
+
temp_w = self.total_w + w
|
|
54
|
+
self.ss += (self.total_w * w * (x - self.m) *
|
|
55
|
+
(x - self.m)) / temp_w
|
|
56
|
+
self.m += (x - self.m) / temp_w
|
|
57
|
+
self.total_w = temp_w
|
|
58
|
+
|
|
59
|
+
def var(self):
|
|
60
|
+
return self.ss / self.total_w
|
|
61
|
+
|
|
62
|
+
def sd(self):
|
|
63
|
+
return math.sqrt(self.var())
|
|
64
|
+
|
|
65
|
+
def mean(self):
|
|
66
|
+
return self.m
|
|
67
|
+
|
|
68
|
+
DataPoint = namedtuple('DataPoint', ['value', 'count'])
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def test_mvsd():
|
|
72
|
+
mvsd = MVSD()
|
|
73
|
+
for x in range(10):
|
|
74
|
+
mvsd.add(x)
|
|
75
|
+
|
|
76
|
+
assert '%.2f' % mvsd.mean() == "4.50"
|
|
77
|
+
assert '%.2f' % mvsd.var() == "8.25"
|
|
78
|
+
assert '%.14f' % mvsd.sd() == "2.87228132326901"
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def load_stream(input_stream, agg_value_key, agg_key_value):
|
|
82
|
+
for line in input_stream:
|
|
83
|
+
clean_line = line.strip()
|
|
84
|
+
if not clean_line:
|
|
85
|
+
# skip empty lines (ie: newlines)
|
|
86
|
+
continue
|
|
87
|
+
if clean_line[0] in ['"', "'"]:
|
|
88
|
+
clean_line = clean_line.strip("\"'")
|
|
89
|
+
try:
|
|
90
|
+
if agg_key_value:
|
|
91
|
+
key, value = clean_line.rstrip().rsplit(None, 1)
|
|
92
|
+
yield DataPoint(Decimal(key), int(value))
|
|
93
|
+
elif agg_value_key:
|
|
94
|
+
value, key = clean_line.lstrip().split(None, 1)
|
|
95
|
+
yield DataPoint(Decimal(key), int(value))
|
|
96
|
+
else:
|
|
97
|
+
yield DataPoint(Decimal(clean_line), 1)
|
|
98
|
+
except:
|
|
99
|
+
logging.exception('failed %r', line)
|
|
100
|
+
print("invalid line %r" % line, file=sys.stderr)
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
def median(values, key=None):
|
|
104
|
+
if not key:
|
|
105
|
+
key = lambda x: x # identity; py3 map() does not accept None
|
|
106
|
+
length = len(values)
|
|
107
|
+
if length % 2:
|
|
108
|
+
median_indeces = [int(length/2)]
|
|
109
|
+
else:
|
|
110
|
+
median_indeces = [int(length/2)-1, int(length/2)]
|
|
111
|
+
|
|
112
|
+
values = sorted(values, key=key)
|
|
113
|
+
return sum(map(key,
|
|
114
|
+
[values[i] for i in median_indeces])) / len(median_indeces)
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
def test_median():
|
|
118
|
+
assert 6 == median([8, 7, 9, 1, 2, 6, 3]) # odd-sized list
|
|
119
|
+
assert 4.5 == median([4, 5, 2, 1, 9, 10]) # even-sized int list. (4+5)/2 = 4.5
|
|
120
|
+
# even-sized float list. (4.0+5)/2 = 4.5
|
|
121
|
+
assert "4.50" == "%.2f" % median([4.0, 5, 2, 1, 9, 10])
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def histogram(stream, options):
|
|
125
|
+
"""
|
|
126
|
+
Loop over the stream and add each entry to the dataset, printing out at the
|
|
127
|
+
end.
|
|
128
|
+
|
|
129
|
+
stream yields Decimal()
|
|
130
|
+
"""
|
|
131
|
+
if not options.min or not options.max:
|
|
132
|
+
# glob the iterator here so we can do min/max on it
|
|
133
|
+
data = list(stream)
|
|
134
|
+
else:
|
|
135
|
+
data = stream
|
|
136
|
+
bucket_scale = 1
|
|
137
|
+
|
|
138
|
+
if options.min:
|
|
139
|
+
min_v = Decimal(options.min)
|
|
140
|
+
else:
|
|
141
|
+
min_v = min(data, key=lambda x: x.value)
|
|
142
|
+
min_v = min_v.value
|
|
143
|
+
if options.max:
|
|
144
|
+
max_v = Decimal(options.max)
|
|
145
|
+
else:
|
|
146
|
+
max_v = max(data, key=lambda x: x.value)
|
|
147
|
+
max_v = max_v.value
|
|
148
|
+
|
|
149
|
+
if not max_v > min_v:
|
|
150
|
+
raise ValueError('max must be > min. max:%s min:%s' % (max_v, min_v))
|
|
151
|
+
diff = max_v - min_v
|
|
152
|
+
|
|
153
|
+
boundaries = []
|
|
154
|
+
bucket_counts = []
|
|
155
|
+
buckets = 0
|
|
156
|
+
|
|
157
|
+
if options.custbuckets:
|
|
158
|
+
bound = options.custbuckets.split(',')
|
|
159
|
+
bound_sort = sorted(map(Decimal, bound))
|
|
160
|
+
|
|
161
|
+
# if the last value is smaller than the maximum, replace it
|
|
162
|
+
if bound_sort[-1] < max_v:
|
|
163
|
+
bound_sort[-1] = max_v
|
|
164
|
+
|
|
165
|
+
# iterate through the sorted list and append to boundaries
|
|
166
|
+
for x in bound_sort:
|
|
167
|
+
if x >= min_v and x <= max_v:
|
|
168
|
+
boundaries.append(x)
|
|
169
|
+
elif x >= max_v:
|
|
170
|
+
boundaries.append(max_v)
|
|
171
|
+
break
|
|
172
|
+
|
|
173
|
+
# beware: the min_v is not included in the boundaries,
|
|
174
|
+
# so no need to do a -1!
|
|
175
|
+
bucket_counts = [0 for x in range(len(boundaries))]
|
|
176
|
+
buckets = len(boundaries)
|
|
177
|
+
elif options.logscale:
|
|
178
|
+
buckets = options.buckets and int(options.buckets) or 10
|
|
179
|
+
if buckets <= 0:
|
|
180
|
+
raise ValueError('# of buckets must be > 0')
|
|
181
|
+
|
|
182
|
+
def first_bucket_size(k, n):
|
|
183
|
+
r"""Logarithmic buckets means, the size of bucket i+1 is twice
|
|
184
|
+
the size of bucket i.
|
|
185
|
+
For k+1 buckets whose sum is n, we have
|
|
186
|
+
(note, k+1 buckets, since 0 is counted as well):
|
|
187
|
+
\sum_{i=0}^{k} x*2^i = n
|
|
188
|
+
x * \sum_{i=0}^{k} 2^i = n
|
|
189
|
+
x * (2^{k+1} - 1) = n
|
|
190
|
+
x = n/(2^{k+1} - 1)
|
|
191
|
+
"""
|
|
192
|
+
return n/(2**(k+1)-1)
|
|
193
|
+
|
|
194
|
+
def log_steps(k, n):
|
|
195
|
+
"k logarithmic steps whose sum is n"
|
|
196
|
+
x = first_bucket_size(k-1, n)
|
|
197
|
+
sum = 0
|
|
198
|
+
for i in range(k):
|
|
199
|
+
sum += 2**i * x
|
|
200
|
+
yield sum
|
|
201
|
+
bucket_counts = [0 for x in range(buckets)]
|
|
202
|
+
for step in log_steps(buckets, diff):
|
|
203
|
+
boundaries.append(min_v + step)
|
|
204
|
+
else:
|
|
205
|
+
buckets = options.buckets and int(options.buckets) or 10
|
|
206
|
+
if buckets <= 0:
|
|
207
|
+
raise ValueError('# of buckets must be > 0')
|
|
208
|
+
step = diff / buckets
|
|
209
|
+
bucket_counts = [0 for x in range(buckets)]
|
|
210
|
+
for x in range(buckets):
|
|
211
|
+
boundaries.append(min_v + (step * (x + 1)))
|
|
212
|
+
|
|
213
|
+
skipped = 0
|
|
214
|
+
samples = 0
|
|
215
|
+
mvsd = MVSD()
|
|
216
|
+
accepted_data = []
|
|
217
|
+
for record in data:
|
|
218
|
+
samples += record.count
|
|
219
|
+
if options.mvsd:
|
|
220
|
+
mvsd.add(record.value, record.count)
|
|
221
|
+
accepted_data.append(record)
|
|
222
|
+
# find the bucket this goes in
|
|
223
|
+
if record.value < min_v or record.value > max_v:
|
|
224
|
+
skipped += record.count
|
|
225
|
+
continue
|
|
226
|
+
for bucket_postion, boundary in enumerate(boundaries):
|
|
227
|
+
if record.value <= boundary:
|
|
228
|
+
bucket_counts[bucket_postion] += record.count
|
|
229
|
+
break
|
|
230
|
+
|
|
231
|
+
# auto-pick the hash scale
|
|
232
|
+
if max(bucket_counts) > 75:
|
|
233
|
+
bucket_scale = int(max(bucket_counts) / 75)
|
|
234
|
+
|
|
235
|
+
print(("# NumSamples = %d; Min = %0.2f; Max = %0.2f" %
|
|
236
|
+
(samples, min_v, max_v)))
|
|
237
|
+
if skipped:
|
|
238
|
+
print(("# %d value%s outside of min/max" %
|
|
239
|
+
(skipped, skipped > 1 and 's' or '')))
|
|
240
|
+
if options.mvsd:
|
|
241
|
+
print(("# Mean = %f; Variance = %f; SD = %f; Median %f" %
|
|
242
|
+
(mvsd.mean(), mvsd.var(), mvsd.sd(),
|
|
243
|
+
median(accepted_data, key=lambda x: x.value))))
|
|
244
|
+
print(("# each " + options.dot + " represents a count of %d" % bucket_scale))
|
|
245
|
+
|
|
246
|
+
bucket_min = min_v
|
|
247
|
+
bucket_max = min_v
|
|
248
|
+
percentage = ""
|
|
249
|
+
format_string = options.format + ' - ' + options.format + ' [%6d]: %s%s'
|
|
250
|
+
for bucket in range(buckets):
|
|
251
|
+
bucket_min = bucket_max
|
|
252
|
+
bucket_max = boundaries[bucket]
|
|
253
|
+
bucket_count = bucket_counts[bucket]
|
|
254
|
+
star_count = 0
|
|
255
|
+
if bucket_count:
|
|
256
|
+
star_count = int(bucket_count / bucket_scale)
|
|
257
|
+
if options.percentage:
|
|
258
|
+
percentage = " (%0.2f%%)" % (100 * Decimal(bucket_count) /
|
|
259
|
+
Decimal(samples))
|
|
260
|
+
print((format_string % (bucket_min, bucket_max, bucket_count, options.dot * star_count, percentage)))
|
|
261
|
+
|
|
262
|
+
|
|
263
|
+
if __name__ == "__main__":
|
|
264
|
+
parser = OptionParser()
|
|
265
|
+
parser.usage = "cat data | %prog [options]"
|
|
266
|
+
parser.add_option("-a", "--agg", dest="agg_value_key", default=False,
|
|
267
|
+
action="store_true", help="Two column input format, " +
|
|
268
|
+
"space seperated with value<space>key")
|
|
269
|
+
parser.add_option("-A", "--agg-key-value", dest="agg_key_value",
|
|
270
|
+
default=False, action="store_true", help="Two column " +
|
|
271
|
+
"input format, space seperated with key<space>value")
|
|
272
|
+
parser.add_option("-m", "--min", dest="min",
|
|
273
|
+
help="minimum value for graph")
|
|
274
|
+
parser.add_option("-x", "--max", dest="max",
|
|
275
|
+
help="maximum value for graph")
|
|
276
|
+
parser.add_option("-b", "--buckets", dest="buckets",
|
|
277
|
+
help="Number of buckets to use for the histogram")
|
|
278
|
+
parser.add_option("-l", "--logscale", dest="logscale", default=False,
|
|
279
|
+
action="store_true",
|
|
280
|
+
help="Buckets grow in logarithmic scale")
|
|
281
|
+
parser.add_option("-B", "--custom-buckets", dest="custbuckets",
|
|
282
|
+
help="Comma seperated list of bucket " +
|
|
283
|
+
"edges for the histogram")
|
|
284
|
+
parser.add_option("--no-mvsd", dest="mvsd", action="store_false",
|
|
285
|
+
default=True, help="Disable the calculation of Mean, " +
|
|
286
|
+
"Variance and SD (improves performance)")
|
|
287
|
+
parser.add_option("-f", "--bucket-format", dest="format", default="%10.4f",
|
|
288
|
+
help="format for bucket numbers")
|
|
289
|
+
parser.add_option("-p", "--percentage", dest="percentage", default=False,
|
|
290
|
+
action="store_true", help="List percentage for each bar")
|
|
291
|
+
parser.add_option("--dot", dest="dot", default='∎', help="Dot representation")
|
|
292
|
+
|
|
293
|
+
(options, args) = parser.parse_args()
|
|
294
|
+
if sys.stdin.isatty():
|
|
295
|
+
# if isatty() that means it's run without anything piped into it
|
|
296
|
+
parser.print_usage()
|
|
297
|
+
print("for more help use --help")
|
|
298
|
+
sys.exit(1)
|
|
299
|
+
histogram(load_stream(sys.stdin, options.agg_value_key,
|
|
300
|
+
options.agg_key_value), options)
|
|
@@ -0,0 +1,59 @@
|
|
|
1
|
+
#!python
|
|
2
|
+
#
|
|
3
|
+
# Copyright 2010 Bitly
|
|
4
|
+
#
|
|
5
|
+
# Licensed under the Apache License, Version 2.0 (the "License"); you may
|
|
6
|
+
# not use this file except in compliance with the License. You may obtain
|
|
7
|
+
# a copy of the License at
|
|
8
|
+
#
|
|
9
|
+
# http://www.apache.org/licenses/LICENSE-2.0
|
|
10
|
+
#
|
|
11
|
+
# Unless required by applicable law or agreed to in writing, software
|
|
12
|
+
# distributed under the License is distributed on an "AS IS" BASIS, WITHOUT
|
|
13
|
+
# WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. See the
|
|
14
|
+
# License for the specific language governing permissions and limitations
|
|
15
|
+
# under the License.
|
|
16
|
+
|
|
17
|
+
"""
|
|
18
|
+
Calculate the 95% time from a list of times given on stdin
|
|
19
|
+
|
|
20
|
+
https://github.com/bitly/data_hacks
|
|
21
|
+
"""
|
|
22
|
+
|
|
23
|
+
import sys
|
|
24
|
+
import os
|
|
25
|
+
from decimal import Decimal
|
|
26
|
+
|
|
27
|
+
def run():
|
|
28
|
+
count = 0
|
|
29
|
+
data = {}
|
|
30
|
+
for line in sys.stdin:
|
|
31
|
+
line = line.strip()
|
|
32
|
+
if not line:
|
|
33
|
+
# skip empty lines (ie: newlines)
|
|
34
|
+
continue
|
|
35
|
+
try:
|
|
36
|
+
t = Decimal(line)
|
|
37
|
+
count +=1
|
|
38
|
+
data[t] = data.get(t, 0) + 1
|
|
39
|
+
except:
|
|
40
|
+
print("invalid line %r" % line, file=sys.stderr)
|
|
41
|
+
print(calc_95(data, count))
|
|
42
|
+
|
|
43
|
+
def calc_95(data, count):
|
|
44
|
+
# find the time it took for x entry, where x is the threshold
|
|
45
|
+
threshold = Decimal(count) * Decimal('.95')
|
|
46
|
+
start = Decimal(0)
|
|
47
|
+
times = list(data.keys())
|
|
48
|
+
times.sort()
|
|
49
|
+
for t in times:
|
|
50
|
+
# increment our count by the # of items in this time bucket
|
|
51
|
+
start += data[t]
|
|
52
|
+
if start > threshold:
|
|
53
|
+
return t
|
|
54
|
+
|
|
55
|
+
if __name__ == "__main__":
|
|
56
|
+
if sys.stdin.isatty() or '--help' in sys.argv or '-h' in sys.argv:
|
|
57
|
+
print("Usage: cat data | %s" % os.path.basename(sys.argv[0]))
|
|
58
|
+
sys.exit(1)
|
|
59
|
+
run()
|
|
@@ -0,0 +1,62 @@
|
|
|
1
|
+
#!python
|
|
2
|
+
#
|
|
3
|
+
# Copyright 2010 Bitly
|
|
4
|
+
#
|
|
5
|
+
# Licensed under the Apache License, Version 2.0 (the "License"); you may
|
|
6
|
+
# not use this file except in compliance with the License. You may obtain
|
|
7
|
+
# a copy of the License at
|
|
8
|
+
#
|
|
9
|
+
# http://www.apache.org/licenses/LICENSE-2.0
|
|
10
|
+
#
|
|
11
|
+
# Unless required by applicable law or agreed to in writing, software
|
|
12
|
+
# distributed under the License is distributed on an "AS IS" BASIS, WITHOUT
|
|
13
|
+
# WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. See the
|
|
14
|
+
# License for the specific language governing permissions and limitations
|
|
15
|
+
# under the License.
|
|
16
|
+
|
|
17
|
+
"""
|
|
18
|
+
Pass through data for a specified amount of time
|
|
19
|
+
|
|
20
|
+
https://github.com/bitly/data_hacks
|
|
21
|
+
"""
|
|
22
|
+
|
|
23
|
+
import time
|
|
24
|
+
import sys
|
|
25
|
+
import os
|
|
26
|
+
|
|
27
|
+
def getruntime(arg):
|
|
28
|
+
if not arg:
|
|
29
|
+
return
|
|
30
|
+
suffix = arg[-1]
|
|
31
|
+
base = int(arg[:-1])
|
|
32
|
+
if suffix == "s":
|
|
33
|
+
return base
|
|
34
|
+
elif suffix == "m":
|
|
35
|
+
return base * 60
|
|
36
|
+
elif suffix == "h":
|
|
37
|
+
return base * 60 * 60
|
|
38
|
+
elif suffix == "d":
|
|
39
|
+
return base * 60 * 60 * 24
|
|
40
|
+
else:
|
|
41
|
+
print("invalid time suffix %r. must be one of s,m,h,d" % arg, file=sys.stderr)
|
|
42
|
+
|
|
43
|
+
def run(runtime):
|
|
44
|
+
end = time.time() + runtime
|
|
45
|
+
for line in sys.stdin:
|
|
46
|
+
sys.stdout.write(line)
|
|
47
|
+
if time.time() > end:
|
|
48
|
+
return
|
|
49
|
+
|
|
50
|
+
if __name__ == "__main__":
|
|
51
|
+
usage = "Usage: tail -f access.log | %s [time] | ..." % os.path.basename(sys.argv[0])
|
|
52
|
+
help = "time can be in the format 10s, 10m, 10h, etc"
|
|
53
|
+
if sys.stdin.isatty():
|
|
54
|
+
print(usage)
|
|
55
|
+
print(help)
|
|
56
|
+
sys.exit(1)
|
|
57
|
+
|
|
58
|
+
runtime = getruntime(sys.argv[-1])
|
|
59
|
+
if not runtime:
|
|
60
|
+
print(usage)
|
|
61
|
+
sys.exit(1)
|
|
62
|
+
run(runtime)
|
|
@@ -0,0 +1,65 @@
|
|
|
1
|
+
#!python
|
|
2
|
+
#
|
|
3
|
+
# Copyright 2010 Bitly
|
|
4
|
+
#
|
|
5
|
+
# Licensed under the Apache License, Version 2.0 (the "License"); you may
|
|
6
|
+
# not use this file except in compliance with the License. You may obtain
|
|
7
|
+
# a copy of the License at
|
|
8
|
+
#
|
|
9
|
+
# http://www.apache.org/licenses/LICENSE-2.0
|
|
10
|
+
#
|
|
11
|
+
# Unless required by applicable law or agreed to in writing, software
|
|
12
|
+
# distributed under the License is distributed on an "AS IS" BASIS, WITHOUT
|
|
13
|
+
# WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. See the
|
|
14
|
+
# License for the specific language governing permissions and limitations
|
|
15
|
+
# under the License.
|
|
16
|
+
|
|
17
|
+
"""
|
|
18
|
+
Pass through a sampled percentage of data
|
|
19
|
+
|
|
20
|
+
https://github.com/bitly/data_hacks
|
|
21
|
+
"""
|
|
22
|
+
|
|
23
|
+
import sys
|
|
24
|
+
import random
|
|
25
|
+
from optparse import OptionParser
|
|
26
|
+
from decimal import Decimal
|
|
27
|
+
|
|
28
|
+
def run(sample_rate):
|
|
29
|
+
input_stream = sys.stdin
|
|
30
|
+
for line in input_stream:
|
|
31
|
+
if random.randint(1,100) <= sample_rate:
|
|
32
|
+
sys.stdout.write(line)
|
|
33
|
+
|
|
34
|
+
def get_sample_rate(rate_string):
|
|
35
|
+
""" return a rate as a percentage"""
|
|
36
|
+
if rate_string.endswith("%"):
|
|
37
|
+
rate = int(rate_string[:-1])
|
|
38
|
+
elif '/' in rate_string:
|
|
39
|
+
x, y = rate_string.split('/')
|
|
40
|
+
rate = Decimal(x) / (Decimal(y) * Decimal('1.0'))
|
|
41
|
+
rate = int(rate * 100)
|
|
42
|
+
else:
|
|
43
|
+
raise ValueError("rate %r is invalid rate format must be '10%%' or '1/10'" % rate_string)
|
|
44
|
+
if rate < 1 or rate > 100:
|
|
45
|
+
raise ValueError('rate %r must be 1%% <= rate <= 100%% ' % rate_string)
|
|
46
|
+
return rate
|
|
47
|
+
|
|
48
|
+
if __name__ == "__main__":
|
|
49
|
+
parser = OptionParser(usage="cat data | %prog [options] [sample_rate]")
|
|
50
|
+
parser.add_option("--verbose", dest="verbose", default=False, action="store_true")
|
|
51
|
+
(options, args) = parser.parse_args()
|
|
52
|
+
|
|
53
|
+
if not args or sys.stdin.isatty():
|
|
54
|
+
parser.print_usage()
|
|
55
|
+
sys.exit(1)
|
|
56
|
+
|
|
57
|
+
try:
|
|
58
|
+
sample_rate = get_sample_rate(sys.argv[-1])
|
|
59
|
+
except ValueError as e:
|
|
60
|
+
print(e, file=sys.stderr)
|
|
61
|
+
parser.print_usage()
|
|
62
|
+
sys.exit(1)
|
|
63
|
+
if options.verbose:
|
|
64
|
+
print("Sample rate is %d%%" % sample_rate, file=sys.stderr)
|
|
65
|
+
run(sample_rate)
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: data_hacks3
|
|
3
|
+
Version: 0.0.4
|
|
4
|
+
Summary: Command line utilities for data analysis
|
|
5
|
+
Home-page: https://github.com/nicholasren/data_hacks3
|
|
6
|
+
Download-URL: http://github.com/downloads/nicholasren/data_hacks3/data_hacks3-0.0.4.tar.gz
|
|
7
|
+
Author: ['Xiaojun Ren']
|
|
8
|
+
Author-email: nicholas.x.ren@gmail.com
|
|
9
|
+
Classifier: Development Status :: 4 - Beta
|
|
10
|
+
Classifier: Programming Language :: Python
|
|
11
|
+
Classifier: Intended Audience :: System Administrators
|
|
12
|
+
Classifier: Topic :: Terminals
|
|
13
|
+
Dynamic: author
|
|
14
|
+
Dynamic: author-email
|
|
15
|
+
Dynamic: classifier
|
|
16
|
+
Dynamic: download-url
|
|
17
|
+
Dynamic: home-page
|
|
18
|
+
Dynamic: summary
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
data_hacks/bar_chart.py,sha256=2gW0Ln2l4XG73RgCbPCD0pb0AD5MyH3RCkvSsBb89TA,4524
|
|
2
|
+
data_hacks/histogram.py,sha256=gO8gngVZBOL1_ai_0X7MEMQFNHdSaANV70XOslvWm1A,10743
|
|
3
|
+
data_hacks/ninety_five_percent.py,sha256=gXP6lob_CDKFhe5o56Tbhjoo2I4-M7X7uAQdUHGnHwU,1728
|
|
4
|
+
data_hacks/run_for.py,sha256=emvHWo3BXsAlRZbOsVhq_i_Zx5xMHRJtu6QAD4apiw0,1663
|
|
5
|
+
data_hacks/sample.py,sha256=mV3q3xIGkwUrWmTsS5QzJ_Cw7j_Vd4pQ2rcXLUKYLZc,2092
|
|
6
|
+
data_hacks3-0.0.4.data/scripts/bar_chart.py,sha256=BXk65pdgmhjLHVZzAze9YP3lbUJ4xCjfRyAMImzYsLQ,4511
|
|
7
|
+
data_hacks3-0.0.4.data/scripts/histogram.py,sha256=MxaLuX04KFRGaQ-ou9mAaqIabkwtDgBRSxaD2h94KPk,10730
|
|
8
|
+
data_hacks3-0.0.4.data/scripts/ninety_five_percent.py,sha256=L-WuCWqxyeN2earIqetS5aOSV9hx94TKXNakOm1B2TQ,1715
|
|
9
|
+
data_hacks3-0.0.4.data/scripts/run_for.py,sha256=0hWIlec0J6LQ4v1LeI0WN_XjeB2uyEzAKxTNdTx7Qno,1650
|
|
10
|
+
data_hacks3-0.0.4.data/scripts/sample.py,sha256=qFhRU2r2FqUB-KlnqmaTHXY3_Gk5vSc2gLtM1HYS-4Y,2079
|
|
11
|
+
data_hacks3-0.0.4.dist-info/METADATA,sha256=yd5zpEZtBqG_Qrm0OXReGCe_MDI_xANaKO3wVV60axg,601
|
|
12
|
+
data_hacks3-0.0.4.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
|
|
13
|
+
data_hacks3-0.0.4.dist-info/top_level.txt,sha256=80xgV1LClXEWXrY9e3yPfstyNGj-hXFhCQsi50HCQrU,11
|
|
14
|
+
data_hacks3-0.0.4.dist-info/RECORD,,
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
data_hacks
|