datasketches 0.5.1 → 0.5.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +9 -0
- data/ext/datasketches/cpc_wrapper.cpp +8 -9
- data/ext/datasketches/fi_wrapper.cpp +4 -5
- data/ext/datasketches/hll_wrapper.cpp +9 -10
- data/ext/datasketches/kll_wrapper.cpp +13 -12
- data/ext/datasketches/theta_wrapper.cpp +11 -13
- data/ext/datasketches/vo_wrapper.cpp +1 -2
- data/lib/datasketches/version.rb +1 -1
- data/vendor/datasketches-cpp/CMakeLists.txt +5 -3
- data/vendor/datasketches-cpp/CODE_OF_CONDUCT.md +1 -1
- data/vendor/datasketches-cpp/LICENSE +14 -0
- data/vendor/datasketches-cpp/README.md +67 -73
- data/vendor/datasketches-cpp/benchmarks/CMakeLists.txt +52 -0
- data/vendor/datasketches-cpp/benchmarks/benchmark_count_min_sketch.cpp +153 -0
- data/vendor/datasketches-cpp/benchmarks/benchmark_count_min_sketch_serialization.cpp +161 -0
- data/vendor/datasketches-cpp/common/CMakeLists.txt +1 -0
- data/vendor/datasketches-cpp/common/include/binomial_bounds.hpp +2 -2
- data/vendor/datasketches-cpp/common/include/fdlibm_log.hpp +101 -0
- data/vendor/datasketches-cpp/common/include/serde.hpp +6 -0
- data/vendor/datasketches-cpp/common/test/CMakeLists.txt +44 -2
- data/vendor/datasketches-cpp/common/test/binomial_bounds_test.cpp +279 -0
- data/vendor/datasketches-cpp/common/test/deserialize_hardening_test.cpp +188 -0
- data/vendor/datasketches-cpp/count/include/count_min.hpp +17 -4
- data/vendor/datasketches-cpp/count/include/count_min_impl.hpp +65 -83
- data/vendor/datasketches-cpp/count/test/count_min_test.cpp +63 -7
- data/vendor/datasketches-cpp/cpc/include/compression_data.hpp +2 -0
- data/vendor/datasketches-cpp/cpc/include/cpc_compressor_impl.hpp +16 -7
- data/vendor/datasketches-cpp/cpc/include/cpc_sketch.hpp +1 -0
- data/vendor/datasketches-cpp/cpc/include/cpc_sketch_impl.hpp +54 -36
- data/vendor/datasketches-cpp/cpc/include/cpc_union_impl.hpp +27 -27
- data/vendor/datasketches-cpp/cpc/include/cpc_util.hpp +7 -7
- data/vendor/datasketches-cpp/cpc/include/icon_estimator.hpp +5 -5
- data/vendor/datasketches-cpp/cpc/include/u32_table_impl.hpp +16 -16
- data/vendor/datasketches-cpp/cpc/test/cpc_sketch_test.cpp +35 -0
- data/vendor/datasketches-cpp/fi/include/frequent_items_sketch.hpp +28 -3
- data/vendor/datasketches-cpp/fi/include/frequent_items_sketch_impl.hpp +41 -25
- data/vendor/datasketches-cpp/fi/include/reverse_purge_hash_map.hpp +3 -1
- data/vendor/datasketches-cpp/fi/include/reverse_purge_hash_map_impl.hpp +10 -5
- data/vendor/datasketches-cpp/fi/test/frequent_items_sketch_serialize_for_java.cpp +24 -0
- data/vendor/datasketches-cpp/fi/test/frequent_items_sketch_test.cpp +116 -0
- data/vendor/datasketches-cpp/filters/include/bloom_filter.hpp +1 -1
- data/vendor/datasketches-cpp/filters/include/bloom_filter_impl.hpp +32 -12
- data/vendor/datasketches-cpp/filters/test/bloom_filter_test.cpp +28 -1
- data/vendor/datasketches-cpp/hll/include/CouponHashSet-internal.hpp +1 -2
- data/vendor/datasketches-cpp/hll/include/CouponList-internal.hpp +19 -5
- data/vendor/datasketches-cpp/hll/include/CubicInterpolation-internal.hpp +4 -4
- data/vendor/datasketches-cpp/hll/include/HarmonicNumbers-internal.hpp +2 -1
- data/vendor/datasketches-cpp/hll/include/Hll4Array-internal.hpp +4 -4
- data/vendor/datasketches-cpp/hll/include/Hll4Array.hpp +1 -1
- data/vendor/datasketches-cpp/hll/include/Hll6Array-internal.hpp +3 -3
- data/vendor/datasketches-cpp/hll/include/Hll6Array.hpp +1 -1
- data/vendor/datasketches-cpp/hll/include/Hll8Array-internal.hpp +6 -3
- data/vendor/datasketches-cpp/hll/include/Hll8Array.hpp +1 -1
- data/vendor/datasketches-cpp/hll/include/HllArray-internal.hpp +35 -29
- data/vendor/datasketches-cpp/hll/include/HllArray.hpp +1 -1
- data/vendor/datasketches-cpp/hll/include/HllSketch-internal.hpp +3 -5
- data/vendor/datasketches-cpp/hll/include/HllSketchImpl-internal.hpp +5 -11
- data/vendor/datasketches-cpp/hll/include/HllSketchImpl.hpp +2 -4
- data/vendor/datasketches-cpp/hll/include/HllSketchImplFactory.hpp +11 -10
- data/vendor/datasketches-cpp/hll/include/HllUnion-internal.hpp +17 -8
- data/vendor/datasketches-cpp/hll/include/HllUtil.hpp +12 -3
- data/vendor/datasketches-cpp/hll/include/coupon_iterator-internal.hpp +2 -2
- data/vendor/datasketches-cpp/hll/include/coupon_iterator.hpp +3 -0
- data/vendor/datasketches-cpp/hll/include/hll.hpp +11 -4
- data/vendor/datasketches-cpp/hll/include/hll.private.hpp +19 -0
- data/vendor/datasketches-cpp/hll/test/CMakeLists.txt +2 -0
- data/vendor/datasketches-cpp/hll/test/CouponListTest.cpp +64 -0
- data/vendor/datasketches-cpp/hll/test/HllFullSizeTest.cpp +137 -0
- data/vendor/datasketches-cpp/hll/test/HllKxqRebuildTest.cpp +150 -0
- data/vendor/datasketches-cpp/hll/test/HllSketchTest.cpp +3 -3
- data/vendor/datasketches-cpp/hll/test/HllUnionTest.cpp +71 -1
- data/vendor/datasketches-cpp/kll/include/kll_helper_impl.hpp +10 -10
- data/vendor/datasketches-cpp/kll/include/kll_sketch.hpp +10 -1
- data/vendor/datasketches-cpp/kll/include/kll_sketch_impl.hpp +33 -24
- data/vendor/datasketches-cpp/kll/test/kll_sketch_deserialize_from_java_test.cpp +24 -0
- data/vendor/datasketches-cpp/kll/test/kll_sketch_serialize_for_java.cpp +10 -0
- data/vendor/datasketches-cpp/quantiles/include/quantiles_sketch.hpp +11 -2
- data/vendor/datasketches-cpp/quantiles/include/quantiles_sketch_impl.hpp +24 -15
- data/vendor/datasketches-cpp/req/include/req_sketch.hpp +9 -0
- data/vendor/datasketches-cpp/req/include/req_sketch_impl.hpp +24 -15
- data/vendor/datasketches-cpp/req/test/req_sketch_deserialize_from_java_test.cpp +46 -0
- data/vendor/datasketches-cpp/req/test/req_sketch_serialize_for_java.cpp +20 -0
- data/vendor/datasketches-cpp/req/test/req_sketch_test.cpp +70 -0
- data/vendor/datasketches-cpp/sampling/include/ebpps_sample_impl.hpp +17 -8
- data/vendor/datasketches-cpp/sampling/include/ebpps_sketch.hpp +13 -0
- data/vendor/datasketches-cpp/sampling/include/var_opt_sketch.hpp +10 -1
- data/vendor/datasketches-cpp/sampling/include/var_opt_sketch_impl.hpp +6 -7
- data/vendor/datasketches-cpp/sampling/include/var_opt_union.hpp +5 -1
- data/vendor/datasketches-cpp/sampling/include/var_opt_union_impl.hpp +2 -1
- data/vendor/datasketches-cpp/sampling/test/ebpps_allocation_test.cpp +1 -1
- data/vendor/datasketches-cpp/sampling/test/ebpps_sketch_test.cpp +2 -2
- data/vendor/datasketches-cpp/sampling/test/var_opt_allocation_test.cpp +1 -1
- data/vendor/datasketches-cpp/sampling/test/var_opt_sketch_test.cpp +10 -4
- data/vendor/datasketches-cpp/sampling/test/var_opt_union_test.cpp +12 -0
- data/vendor/datasketches-cpp/tdigest/include/tdigest.hpp +38 -2
- data/vendor/datasketches-cpp/tdigest/include/tdigest_impl.hpp +168 -9
- data/vendor/datasketches-cpp/tdigest/test/CMakeLists.txt +1 -0
- data/vendor/datasketches-cpp/tdigest/test/tdigest_iterator_test.cpp +274 -0
- data/vendor/datasketches-cpp/tdigest/test/tdigest_test.cpp +275 -0
- data/vendor/datasketches-cpp/theta/include/compact_theta_sketch_parser.hpp +2 -0
- data/vendor/datasketches-cpp/theta/include/compact_theta_sketch_parser_impl.hpp +24 -3
- data/vendor/datasketches-cpp/theta/include/theta_constants.hpp +4 -2
- data/vendor/datasketches-cpp/theta/include/theta_helpers.hpp +32 -0
- data/vendor/datasketches-cpp/theta/include/theta_set_difference_base_impl.hpp +4 -2
- data/vendor/datasketches-cpp/theta/include/theta_sketch.hpp +22 -4
- data/vendor/datasketches-cpp/theta/include/theta_sketch_impl.hpp +60 -38
- data/vendor/datasketches-cpp/theta/include/theta_union_base_impl.hpp +2 -6
- data/vendor/datasketches-cpp/theta/include/theta_update_sketch_base_impl.hpp +2 -2
- data/vendor/datasketches-cpp/theta/test/bit_packing_test.cpp +50 -0
- data/vendor/datasketches-cpp/theta/test/theta_a_not_b_test.cpp +22 -0
- data/vendor/datasketches-cpp/theta/test/theta_sketch_test.cpp +315 -0
- data/vendor/datasketches-cpp/tools/rat-check.sh +68 -0
- data/vendor/datasketches-cpp/tuple/include/array_tuple_sketch.hpp +35 -4
- data/vendor/datasketches-cpp/tuple/include/array_tuple_sketch_impl.hpp +2 -2
- data/vendor/datasketches-cpp/tuple/include/tuple_sketch.hpp +41 -0
- data/vendor/datasketches-cpp/tuple/include/tuple_sketch_impl.hpp +5 -4
- data/vendor/datasketches-cpp/tuple/test/tuple_sketch_test.cpp +59 -0
- data/vendor/datasketches-cpp/version.cfg.in +1 -1
- metadata +12 -2
|
@@ -0,0 +1,153 @@
|
|
|
1
|
+
/*
|
|
2
|
+
* Licensed to the Apache Software Foundation (ASF) under one
|
|
3
|
+
* or more contributor license agreements. See the NOTICE file
|
|
4
|
+
* distributed with this work for additional information
|
|
5
|
+
* regarding copyright ownership. The ASF licenses this file
|
|
6
|
+
* to you under the Apache License, Version 2.0 (the
|
|
7
|
+
* "License"); you may not use this file except in compliance
|
|
8
|
+
* with the License. You may obtain a copy of the License at
|
|
9
|
+
*
|
|
10
|
+
* http://www.apache.org/licenses/LICENSE-2.0
|
|
11
|
+
*
|
|
12
|
+
* Unless required by applicable law or agreed to in writing,
|
|
13
|
+
* software distributed under the License is distributed on an
|
|
14
|
+
* "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
|
|
15
|
+
* KIND, either express or implied. See the License for the
|
|
16
|
+
* specific language governing permissions and limitations
|
|
17
|
+
* under the License.
|
|
18
|
+
*/
|
|
19
|
+
|
|
20
|
+
#include <benchmark/benchmark.h>
|
|
21
|
+
#include <count_min.hpp>
|
|
22
|
+
|
|
23
|
+
#include <cstddef>
|
|
24
|
+
#include <cstdint>
|
|
25
|
+
#include <string>
|
|
26
|
+
#include <vector>
|
|
27
|
+
|
|
28
|
+
namespace
|
|
29
|
+
{
|
|
30
|
+
|
|
31
|
+
using Sketch = datasketches::count_min_sketch<uint64_t>;
|
|
32
|
+
|
|
33
|
+
// Roughly 99.9% confidence and 0.1% relative error:
|
|
34
|
+
// suggest_num_hashes(0.999) == 7 and suggest_num_buckets(0.001) == 2719.
|
|
35
|
+
constexpr uint8_t NUM_HASHES = 7;
|
|
36
|
+
constexpr uint32_t NUM_BUCKETS = 2719;
|
|
37
|
+
|
|
38
|
+
std::vector<uint64_t> makeUInt64Keys(size_t size)
|
|
39
|
+
{
|
|
40
|
+
std::vector<uint64_t> keys;
|
|
41
|
+
keys.reserve(size);
|
|
42
|
+
for (size_t i = 0; i < size; ++i)
|
|
43
|
+
keys.push_back(static_cast<uint64_t>(i * 0x9e3779b97f4a7c15ULL));
|
|
44
|
+
return keys;
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
std::vector<std::string> makeStringKeys(size_t size)
|
|
48
|
+
{
|
|
49
|
+
std::vector<std::string> keys;
|
|
50
|
+
keys.reserve(size);
|
|
51
|
+
for (size_t i = 0; i < size; ++i)
|
|
52
|
+
keys.push_back("countmin-key-" + std::to_string(i * 2654435761ULL));
|
|
53
|
+
return keys;
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
int64_t totalStringBytes(const std::vector<std::string> & keys)
|
|
57
|
+
{
|
|
58
|
+
int64_t bytes = 0;
|
|
59
|
+
for (const auto & key : keys)
|
|
60
|
+
bytes += static_cast<int64_t>(key.size());
|
|
61
|
+
return bytes;
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
void BM_CountMinUpdateUInt64(benchmark::State & state)
|
|
65
|
+
{
|
|
66
|
+
const auto keys = makeUInt64Keys(static_cast<size_t>(state.range(0)));
|
|
67
|
+
for (auto _ : state)
|
|
68
|
+
{
|
|
69
|
+
state.PauseTiming();
|
|
70
|
+
Sketch sketch(NUM_HASHES, NUM_BUCKETS);
|
|
71
|
+
state.ResumeTiming();
|
|
72
|
+
|
|
73
|
+
for (const auto key : keys)
|
|
74
|
+
sketch.update(&key, sizeof(key), 1);
|
|
75
|
+
|
|
76
|
+
benchmark::ClobberMemory();
|
|
77
|
+
benchmark::DoNotOptimize(sketch.get_total_weight());
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
state.SetItemsProcessed(state.iterations() * static_cast<int64_t>(keys.size()));
|
|
81
|
+
state.SetBytesProcessed(state.iterations() * static_cast<int64_t>(keys.size() * sizeof(uint64_t)));
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
void BM_CountMinUpdateStringBytes(benchmark::State & state)
|
|
85
|
+
{
|
|
86
|
+
const auto keys = makeStringKeys(static_cast<size_t>(state.range(0)));
|
|
87
|
+
const auto bytes_per_iteration = totalStringBytes(keys);
|
|
88
|
+
|
|
89
|
+
for (auto _ : state)
|
|
90
|
+
{
|
|
91
|
+
state.PauseTiming();
|
|
92
|
+
Sketch sketch(NUM_HASHES, NUM_BUCKETS);
|
|
93
|
+
state.ResumeTiming();
|
|
94
|
+
|
|
95
|
+
for (const auto & key : keys)
|
|
96
|
+
sketch.update(key.data(), key.size(), 1);
|
|
97
|
+
|
|
98
|
+
benchmark::ClobberMemory();
|
|
99
|
+
benchmark::DoNotOptimize(sketch.get_total_weight());
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
state.SetItemsProcessed(state.iterations() * static_cast<int64_t>(keys.size()));
|
|
103
|
+
state.SetBytesProcessed(state.iterations() * bytes_per_iteration);
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
void BM_CountMinEstimateUInt64(benchmark::State & state)
|
|
107
|
+
{
|
|
108
|
+
const auto keys = makeUInt64Keys(static_cast<size_t>(state.range(0)));
|
|
109
|
+
Sketch sketch(NUM_HASHES, NUM_BUCKETS);
|
|
110
|
+
for (const auto key : keys)
|
|
111
|
+
sketch.update(&key, sizeof(key), 1);
|
|
112
|
+
|
|
113
|
+
uint64_t sum = 0;
|
|
114
|
+
for (auto _ : state)
|
|
115
|
+
{
|
|
116
|
+
for (const auto key : keys)
|
|
117
|
+
sum += sketch.get_estimate(&key, sizeof(key));
|
|
118
|
+
|
|
119
|
+
benchmark::DoNotOptimize(sum);
|
|
120
|
+
}
|
|
121
|
+
|
|
122
|
+
state.SetItemsProcessed(state.iterations() * static_cast<int64_t>(keys.size()));
|
|
123
|
+
state.SetBytesProcessed(state.iterations() * static_cast<int64_t>(keys.size() * sizeof(uint64_t)));
|
|
124
|
+
}
|
|
125
|
+
|
|
126
|
+
void BM_CountMinEstimateStringBytes(benchmark::State & state)
|
|
127
|
+
{
|
|
128
|
+
const auto keys = makeStringKeys(static_cast<size_t>(state.range(0)));
|
|
129
|
+
const auto bytes_per_iteration = totalStringBytes(keys);
|
|
130
|
+
|
|
131
|
+
Sketch sketch(NUM_HASHES, NUM_BUCKETS);
|
|
132
|
+
for (const auto & key : keys)
|
|
133
|
+
sketch.update(key.data(), key.size(), 1);
|
|
134
|
+
|
|
135
|
+
uint64_t sum = 0;
|
|
136
|
+
for (auto _ : state)
|
|
137
|
+
{
|
|
138
|
+
for (const auto & key : keys)
|
|
139
|
+
sum += sketch.get_estimate(key.data(), key.size());
|
|
140
|
+
|
|
141
|
+
benchmark::DoNotOptimize(sum);
|
|
142
|
+
}
|
|
143
|
+
|
|
144
|
+
state.SetItemsProcessed(state.iterations() * static_cast<int64_t>(keys.size()));
|
|
145
|
+
state.SetBytesProcessed(state.iterations() * bytes_per_iteration);
|
|
146
|
+
}
|
|
147
|
+
|
|
148
|
+
}
|
|
149
|
+
|
|
150
|
+
BENCHMARK(BM_CountMinUpdateUInt64)->RangeMultiplier(8)->Range(1024, 65536);
|
|
151
|
+
BENCHMARK(BM_CountMinUpdateStringBytes)->RangeMultiplier(8)->Range(1024, 65536);
|
|
152
|
+
BENCHMARK(BM_CountMinEstimateUInt64)->RangeMultiplier(8)->Range(1024, 65536);
|
|
153
|
+
BENCHMARK(BM_CountMinEstimateStringBytes)->RangeMultiplier(8)->Range(1024, 65536);
|
|
@@ -0,0 +1,161 @@
|
|
|
1
|
+
/*
|
|
2
|
+
* Licensed to the Apache Software Foundation (ASF) under one
|
|
3
|
+
* or more contributor license agreements. See the NOTICE file
|
|
4
|
+
* distributed with this work for additional information
|
|
5
|
+
* regarding copyright ownership. The ASF licenses this file
|
|
6
|
+
* to you under the Apache License, Version 2.0 (the
|
|
7
|
+
* "License"); you may not use this file except in compliance
|
|
8
|
+
* with the License. You may obtain a copy of the License at
|
|
9
|
+
*
|
|
10
|
+
* http://www.apache.org/licenses/LICENSE-2.0
|
|
11
|
+
*
|
|
12
|
+
* Unless required by applicable law or agreed to in writing,
|
|
13
|
+
* software distributed under the License is distributed on an
|
|
14
|
+
* "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
|
|
15
|
+
* KIND, either express or implied. See the License for the
|
|
16
|
+
* specific language governing permissions and limitations
|
|
17
|
+
* under the License.
|
|
18
|
+
*/
|
|
19
|
+
|
|
20
|
+
#include <benchmark/benchmark.h>
|
|
21
|
+
#include <count_min.hpp>
|
|
22
|
+
|
|
23
|
+
#include <algorithm>
|
|
24
|
+
#include <cstddef>
|
|
25
|
+
#include <cstdint>
|
|
26
|
+
#include <cstring>
|
|
27
|
+
#include <ostream>
|
|
28
|
+
#include <streambuf>
|
|
29
|
+
#include <vector>
|
|
30
|
+
|
|
31
|
+
namespace
|
|
32
|
+
{
|
|
33
|
+
|
|
34
|
+
using Sketch = datasketches::count_min_sketch<uint64_t>;
|
|
35
|
+
|
|
36
|
+
// Roughly 99.9% confidence. Serialization cost then scales with the bucket count.
|
|
37
|
+
constexpr uint8_t NUM_HASHES = 7;
|
|
38
|
+
|
|
39
|
+
Sketch makeSketch(uint32_t num_buckets)
|
|
40
|
+
{
|
|
41
|
+
Sketch sketch(NUM_HASHES, num_buckets);
|
|
42
|
+
for (uint64_t i = 0; i < NUM_HASHES; ++i)
|
|
43
|
+
sketch.update(i, i + 1);
|
|
44
|
+
return sketch;
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
class fixed_buffer_streambuf final: public std::streambuf
|
|
48
|
+
{
|
|
49
|
+
public:
|
|
50
|
+
explicit fixed_buffer_streambuf(std::vector<uint8_t>& buffer): buffer_(buffer), position_(0) {}
|
|
51
|
+
|
|
52
|
+
void reset()
|
|
53
|
+
{
|
|
54
|
+
position_ = 0;
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
size_t bytes_written() const
|
|
58
|
+
{
|
|
59
|
+
return position_;
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
protected:
|
|
63
|
+
std::streamsize xsputn(const char* data, std::streamsize size) override
|
|
64
|
+
{
|
|
65
|
+
const size_t requested = static_cast<size_t>(size);
|
|
66
|
+
const size_t available = buffer_.size() - position_;
|
|
67
|
+
const size_t bytes_to_write = std::min(requested, available);
|
|
68
|
+
if (bytes_to_write > 0) {
|
|
69
|
+
std::memcpy(buffer_.data() + position_, data, bytes_to_write);
|
|
70
|
+
position_ += bytes_to_write;
|
|
71
|
+
}
|
|
72
|
+
return static_cast<std::streamsize>(bytes_to_write);
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
int_type overflow(int_type ch) override
|
|
76
|
+
{
|
|
77
|
+
if (traits_type::eq_int_type(ch, traits_type::eof()))
|
|
78
|
+
return traits_type::not_eof(ch);
|
|
79
|
+
if (position_ == buffer_.size())
|
|
80
|
+
return traits_type::eof();
|
|
81
|
+
buffer_[position_++] = static_cast<uint8_t>(traits_type::to_char_type(ch));
|
|
82
|
+
return ch;
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
private:
|
|
86
|
+
std::vector<uint8_t>& buffer_;
|
|
87
|
+
size_t position_;
|
|
88
|
+
};
|
|
89
|
+
|
|
90
|
+
void BM_CountMinGetSerializedSizeBytes(benchmark::State& state)
|
|
91
|
+
{
|
|
92
|
+
const auto sketch = makeSketch(static_cast<uint32_t>(state.range(0)));
|
|
93
|
+
|
|
94
|
+
for (auto _ : state) {
|
|
95
|
+
const auto size = sketch.get_serialized_size_bytes();
|
|
96
|
+
benchmark::DoNotOptimize(size);
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
state.SetItemsProcessed(state.iterations());
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
void BM_CountMinSerializeVector(benchmark::State& state)
|
|
103
|
+
{
|
|
104
|
+
const auto sketch = makeSketch(static_cast<uint32_t>(state.range(0)));
|
|
105
|
+
const auto serialized_size = sketch.get_serialized_size_bytes();
|
|
106
|
+
|
|
107
|
+
for (auto _ : state) {
|
|
108
|
+
auto bytes = sketch.serialize();
|
|
109
|
+
benchmark::DoNotOptimize(bytes.data());
|
|
110
|
+
benchmark::DoNotOptimize(bytes.size());
|
|
111
|
+
benchmark::ClobberMemory();
|
|
112
|
+
}
|
|
113
|
+
|
|
114
|
+
state.SetBytesProcessed(state.iterations() * static_cast<int64_t>(serialized_size));
|
|
115
|
+
}
|
|
116
|
+
|
|
117
|
+
void BM_CountMinSerializeOStream(benchmark::State& state)
|
|
118
|
+
{
|
|
119
|
+
const auto sketch = makeSketch(static_cast<uint32_t>(state.range(0)));
|
|
120
|
+
const auto serialized_size = sketch.get_serialized_size_bytes();
|
|
121
|
+
std::vector<uint8_t> bytes(serialized_size);
|
|
122
|
+
fixed_buffer_streambuf stream_buffer(bytes);
|
|
123
|
+
std::ostream os(&stream_buffer);
|
|
124
|
+
|
|
125
|
+
for (auto _ : state) {
|
|
126
|
+
stream_buffer.reset();
|
|
127
|
+
os.clear();
|
|
128
|
+
sketch.serialize(os);
|
|
129
|
+
benchmark::DoNotOptimize(stream_buffer.bytes_written());
|
|
130
|
+
benchmark::ClobberMemory();
|
|
131
|
+
}
|
|
132
|
+
|
|
133
|
+
state.SetBytesProcessed(state.iterations() * static_cast<int64_t>(serialized_size));
|
|
134
|
+
}
|
|
135
|
+
|
|
136
|
+
void BM_CountMinSerializeToSink(benchmark::State& state)
|
|
137
|
+
{
|
|
138
|
+
const auto sketch = makeSketch(static_cast<uint32_t>(state.range(0)));
|
|
139
|
+
const auto serialized_size = sketch.get_serialized_size_bytes();
|
|
140
|
+
std::vector<uint8_t> bytes(serialized_size);
|
|
141
|
+
|
|
142
|
+
for (auto _ : state) {
|
|
143
|
+
uint8_t* ptr = bytes.data();
|
|
144
|
+
const auto bytes_written = sketch.serialize_to([&ptr](const void* data, size_t size) {
|
|
145
|
+
std::memcpy(ptr, data, size);
|
|
146
|
+
ptr += size;
|
|
147
|
+
});
|
|
148
|
+
benchmark::DoNotOptimize(ptr);
|
|
149
|
+
benchmark::DoNotOptimize(bytes_written);
|
|
150
|
+
benchmark::ClobberMemory();
|
|
151
|
+
}
|
|
152
|
+
|
|
153
|
+
state.SetBytesProcessed(state.iterations() * static_cast<int64_t>(serialized_size));
|
|
154
|
+
}
|
|
155
|
+
|
|
156
|
+
}
|
|
157
|
+
|
|
158
|
+
BENCHMARK(BM_CountMinGetSerializedSizeBytes)->RangeMultiplier(8)->Range(1024, 65536);
|
|
159
|
+
BENCHMARK(BM_CountMinSerializeVector)->RangeMultiplier(8)->Range(1024, 65536);
|
|
160
|
+
BENCHMARK(BM_CountMinSerializeOStream)->RangeMultiplier(8)->Range(1024, 65536);
|
|
161
|
+
BENCHMARK(BM_CountMinSerializeToSink)->RangeMultiplier(8)->Range(1024, 65536);
|
|
@@ -441,8 +441,8 @@ private:
|
|
|
441
441
|
}
|
|
442
442
|
|
|
443
443
|
static void check_theta(double theta) {
|
|
444
|
-
if (theta
|
|
445
|
-
throw std::invalid_argument("theta must be in
|
|
444
|
+
if (theta <= 0 || theta > 1) {
|
|
445
|
+
throw std::invalid_argument("theta must be in (0, 1]");
|
|
446
446
|
}
|
|
447
447
|
}
|
|
448
448
|
|
|
@@ -0,0 +1,101 @@
|
|
|
1
|
+
// fdlibm __ieee754_log, used by Java's StrictMath.log (and by Math.log on most JVMs).
|
|
2
|
+
// Derived from FDLIBM 5.3, Copyright (C) 1993 by Sun Microsystems, Inc.
|
|
3
|
+
// "Permission to use, copy, modify, and distribute this software is freely granted,
|
|
4
|
+
// provided that this notice is preserved."
|
|
5
|
+
#ifndef FDLIBM_LOG_HPP
|
|
6
|
+
#define FDLIBM_LOG_HPP
|
|
7
|
+
#include <cstdint>
|
|
8
|
+
#include <cstring>
|
|
9
|
+
#include <limits>
|
|
10
|
+
|
|
11
|
+
namespace fdlibm {
|
|
12
|
+
|
|
13
|
+
inline int32_t hi_word(double x){ uint64_t u; std::memcpy(&u,&x,8); return (int32_t)(uint32_t)(u>>32); }
|
|
14
|
+
inline uint32_t lo_word(double x){ uint64_t u; std::memcpy(&u,&x,8); return (uint32_t)u; }
|
|
15
|
+
inline void set_hi_word(double& x, uint32_t hi){ uint64_t u; std::memcpy(&u,&x,8);
|
|
16
|
+
u = (u & 0x00000000ffffffffULL) | ((uint64_t)hi<<32); std::memcpy(&x,&u,8); }
|
|
17
|
+
|
|
18
|
+
// Forces a value to be rounded to a double before it is used again. fdlibm needs strict
|
|
19
|
+
// IEEE-754 evaluation: a fused multiply-add anywhere in the polynomial below changes the
|
|
20
|
+
// result. Pragmas are overridden by an explicit -ffp-contract=fast, so use a volatile
|
|
21
|
+
// round-trip, which the standard requires the compiler to honour.
|
|
22
|
+
inline double rnd(double v) { volatile double t = v; return t; }
|
|
23
|
+
|
|
24
|
+
inline double log(double x) {
|
|
25
|
+
// fdlibm depends on strict IEEE-754 evaluation: a fused multiply-add would change the result
|
|
26
|
+
// of the polynomial evaluation below, so contraction must be off for this function.
|
|
27
|
+
#if defined(__clang__)
|
|
28
|
+
#pragma clang fp contract(off)
|
|
29
|
+
#endif
|
|
30
|
+
static const double
|
|
31
|
+
ln2_hi = 6.93147180369123816490e-01, /* 3fe62e42 fee00000 */
|
|
32
|
+
ln2_lo = 1.90821492927058770002e-10, /* 3dea39ef 35793c76 */
|
|
33
|
+
two54 = 1.80143985094819840000e+16, /* 43500000 00000000 */
|
|
34
|
+
Lg1 = 6.666666666666735130e-01, /* 3FE55555 55555593 */
|
|
35
|
+
Lg2 = 3.999999999940941908e-01, /* 3FD99999 9997FA04 */
|
|
36
|
+
Lg3 = 2.857142874366239149e-01, /* 3FD24924 94229359 */
|
|
37
|
+
Lg4 = 2.222219843214978396e-01, /* 3FCC71C5 1D8E78AF */
|
|
38
|
+
Lg5 = 1.818357216161805012e-01, /* 3FC74664 96CB03DE */
|
|
39
|
+
Lg6 = 1.531383769920937332e-01, /* 3FC39A09 D078C69F */
|
|
40
|
+
Lg7 = 1.479819860511658591e-01, /* 3FC2F112 DF3E5244 */
|
|
41
|
+
zero = 0.0;
|
|
42
|
+
|
|
43
|
+
double hfsq,f,s,z,R,w,t1,t2,dk;
|
|
44
|
+
int32_t k,hx,i,j;
|
|
45
|
+
uint32_t lx;
|
|
46
|
+
|
|
47
|
+
hx = hi_word(x);
|
|
48
|
+
lx = lo_word(x);
|
|
49
|
+
|
|
50
|
+
k = 0;
|
|
51
|
+
if (hx < 0x00100000) { /* x < 2**-1022 */
|
|
52
|
+
// fdlibm writes these as -two54/zero and (x-x)/zero, which also raise the divide-by-zero
|
|
53
|
+
// and invalid flags. MSVC rejects a compile-time division by a zero constant (C2124), so
|
|
54
|
+
// return the same values directly. The estimators never call log() with these inputs.
|
|
55
|
+
if (((hx & 0x7fffffff) | lx) == 0) { /* log(+-0) = -inf */
|
|
56
|
+
return -std::numeric_limits<double>::infinity();
|
|
57
|
+
}
|
|
58
|
+
if (hx < 0) { /* log(-#) = NaN */
|
|
59
|
+
return std::numeric_limits<double>::quiet_NaN();
|
|
60
|
+
}
|
|
61
|
+
k -= 54; x *= two54; /* subnormal: scale up */
|
|
62
|
+
hx = hi_word(x);
|
|
63
|
+
}
|
|
64
|
+
if (hx >= 0x7ff00000) { return x+x; }
|
|
65
|
+
k += (hx>>20) - 1023;
|
|
66
|
+
hx &= 0x000fffff;
|
|
67
|
+
i = (hx + 0x95f64) & 0x100000;
|
|
68
|
+
set_hi_word(x, (uint32_t)(hx | (i ^ 0x3ff00000))); /* normalize x or x/2 */
|
|
69
|
+
k += (i>>20);
|
|
70
|
+
f = x - 1.0;
|
|
71
|
+
if ((0x000fffff & (2+hx)) < 3) { /* |f| < 2**-20 */
|
|
72
|
+
if (f == zero) {
|
|
73
|
+
if (k == 0) { return zero; }
|
|
74
|
+
dk = (double)k; return rnd(dk*ln2_hi) + rnd(dk*ln2_lo);
|
|
75
|
+
}
|
|
76
|
+
R = rnd(rnd(f*f)*rnd(0.5 - rnd(0.33333333333333333*f)));
|
|
77
|
+
if (k == 0) { return f-R; }
|
|
78
|
+
dk = (double)k; return rnd(dk*ln2_hi) - (rnd(R - rnd(dk*ln2_lo)) - f);
|
|
79
|
+
}
|
|
80
|
+
s = f/(2.0+f);
|
|
81
|
+
dk = (double)k;
|
|
82
|
+
z = s*s;
|
|
83
|
+
i = hx - 0x6147a;
|
|
84
|
+
w = z*z;
|
|
85
|
+
j = 0x6b851 - hx;
|
|
86
|
+
t1 = rnd(w*rnd(Lg2 + rnd(w*rnd(Lg4 + rnd(w*Lg6)))));
|
|
87
|
+
t2 = rnd(z*rnd(Lg1 + rnd(w*rnd(Lg3 + rnd(w*rnd(Lg5 + rnd(w*Lg7)))))));
|
|
88
|
+
i |= j;
|
|
89
|
+
R = t2 + t1;
|
|
90
|
+
if (i > 0) {
|
|
91
|
+
hfsq = rnd(0.5*f)*f;
|
|
92
|
+
if (k == 0) { return f - rnd(hfsq - rnd(s*(hfsq+R))); }
|
|
93
|
+
return rnd(dk*ln2_hi) - (rnd(hfsq - rnd(rnd(s*(hfsq+R)) + rnd(dk*ln2_lo))) - f);
|
|
94
|
+
} else {
|
|
95
|
+
if (k == 0) { return f - rnd(s*(f-R)); }
|
|
96
|
+
return rnd(dk*ln2_hi) - (rnd(rnd(s*(f-R)) - rnd(dk*ln2_lo)) - f);
|
|
97
|
+
}
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
} // namespace fdlibm
|
|
101
|
+
#endif
|
|
@@ -20,6 +20,7 @@
|
|
|
20
20
|
#ifndef DATASKETCHES_SERDE_HPP_
|
|
21
21
|
#define DATASKETCHES_SERDE_HPP_
|
|
22
22
|
|
|
23
|
+
#include <cstdint>
|
|
23
24
|
#include <cstring>
|
|
24
25
|
#include <iostream>
|
|
25
26
|
#include <memory>
|
|
@@ -132,6 +133,11 @@ struct serde<T, typename std::enable_if<std::is_arithmetic<T>::value>::type> {
|
|
|
132
133
|
/// ItemsSketch<String> with ArrayOfStringsSerDe in Java.
|
|
133
134
|
/// The length of each string is stored as a 32-bit integer (historically),
|
|
134
135
|
/// which may be too wasteful. Treat this as an example.
|
|
136
|
+
///
|
|
137
|
+
/// This implementation treats std::string as an arbitrary byte container.
|
|
138
|
+
/// It does not check whether string contents are valid UTF-8.
|
|
139
|
+
///
|
|
140
|
+
/// Use a UTF-8-validating SerDe when cross-language portability is required.
|
|
135
141
|
template<>
|
|
136
142
|
struct serde<std::string> {
|
|
137
143
|
/// @copydoc serde::serialize
|
|
@@ -69,15 +69,21 @@ target_sources(common_test
|
|
|
69
69
|
PRIVATE
|
|
70
70
|
quantiles_sorted_view_test.cpp
|
|
71
71
|
optional_test.cpp
|
|
72
|
+
binomial_bounds_test.cpp
|
|
72
73
|
)
|
|
73
74
|
|
|
74
75
|
# now the integration test part
|
|
75
76
|
add_executable(integration_test)
|
|
76
77
|
|
|
77
|
-
target_link_libraries(integration_test count cpc density fi hll kll req sampling theta tuple common_test_lib)
|
|
78
|
+
target_link_libraries(integration_test count cpc density fi hll kll req sampling theta tuple quantiles common_test_lib)
|
|
78
79
|
|
|
80
|
+
# Use CMAKE_CXX_STANDARD if defined, otherwise C++11
|
|
81
|
+
set(_integration_cxx_standard 11)
|
|
82
|
+
if(DEFINED CMAKE_CXX_STANDARD)
|
|
83
|
+
set(_integration_cxx_standard ${CMAKE_CXX_STANDARD})
|
|
84
|
+
endif()
|
|
79
85
|
set_target_properties(integration_test PROPERTIES
|
|
80
|
-
CXX_STANDARD
|
|
86
|
+
CXX_STANDARD ${_integration_cxx_standard}
|
|
81
87
|
CXX_STANDARD_REQUIRED YES
|
|
82
88
|
)
|
|
83
89
|
|
|
@@ -90,3 +96,39 @@ target_sources(integration_test
|
|
|
90
96
|
PRIVATE
|
|
91
97
|
integration_test.cpp
|
|
92
98
|
)
|
|
99
|
+
|
|
100
|
+
# Separate hardening test executable (header-only, no pre-compiled libs)
|
|
101
|
+
# This ensures the sketch code is compiled with C++17 + hardening
|
|
102
|
+
# Always build this target - it will use CMAKE_CXX_STANDARD if set (and >= 17), otherwise C++17
|
|
103
|
+
|
|
104
|
+
add_executable(hardening_test)
|
|
105
|
+
target_link_libraries(hardening_test common common_test_lib)
|
|
106
|
+
|
|
107
|
+
# Include directories for header-only sketch implementations
|
|
108
|
+
target_include_directories(hardening_test PRIVATE
|
|
109
|
+
${CMAKE_SOURCE_DIR}/quantiles/include
|
|
110
|
+
${CMAKE_SOURCE_DIR}/kll/include
|
|
111
|
+
${CMAKE_SOURCE_DIR}/req/include
|
|
112
|
+
${CMAKE_SOURCE_DIR}/common/include
|
|
113
|
+
)
|
|
114
|
+
|
|
115
|
+
# Use C++17 minimum for hardening tests
|
|
116
|
+
set(_hardening_cxx_standard 17)
|
|
117
|
+
if(DEFINED CMAKE_CXX_STANDARD AND CMAKE_CXX_STANDARD GREATER_EQUAL 17)
|
|
118
|
+
set(_hardening_cxx_standard ${CMAKE_CXX_STANDARD})
|
|
119
|
+
endif()
|
|
120
|
+
set_target_properties(hardening_test PROPERTIES
|
|
121
|
+
CXX_STANDARD ${_hardening_cxx_standard}
|
|
122
|
+
CXX_STANDARD_REQUIRED YES
|
|
123
|
+
)
|
|
124
|
+
message(STATUS "hardening_test will use C++${_hardening_cxx_standard}")
|
|
125
|
+
|
|
126
|
+
add_test(
|
|
127
|
+
NAME hardening_test
|
|
128
|
+
COMMAND hardening_test "[deserialize_hardening]"
|
|
129
|
+
)
|
|
130
|
+
|
|
131
|
+
target_sources(hardening_test
|
|
132
|
+
PRIVATE
|
|
133
|
+
deserialize_hardening_test.cpp
|
|
134
|
+
)
|