datasketches 0.5.2 → 0.5.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +4 -0
- data/ext/datasketches/theta_wrapper.cpp +1 -1
- data/lib/datasketches/version.rb +1 -1
- data/vendor/datasketches-cpp/CMakeLists.txt +5 -3
- data/vendor/datasketches-cpp/CODE_OF_CONDUCT.md +1 -1
- data/vendor/datasketches-cpp/LICENSE +14 -0
- data/vendor/datasketches-cpp/README.md +67 -73
- data/vendor/datasketches-cpp/benchmarks/CMakeLists.txt +52 -0
- data/vendor/datasketches-cpp/benchmarks/benchmark_count_min_sketch.cpp +153 -0
- data/vendor/datasketches-cpp/benchmarks/benchmark_count_min_sketch_serialization.cpp +161 -0
- data/vendor/datasketches-cpp/common/CMakeLists.txt +1 -0
- data/vendor/datasketches-cpp/common/include/binomial_bounds.hpp +2 -2
- data/vendor/datasketches-cpp/common/include/fdlibm_log.hpp +101 -0
- data/vendor/datasketches-cpp/common/include/serde.hpp +6 -0
- data/vendor/datasketches-cpp/common/test/CMakeLists.txt +44 -2
- data/vendor/datasketches-cpp/common/test/binomial_bounds_test.cpp +279 -0
- data/vendor/datasketches-cpp/common/test/deserialize_hardening_test.cpp +188 -0
- data/vendor/datasketches-cpp/count/include/count_min.hpp +17 -4
- data/vendor/datasketches-cpp/count/include/count_min_impl.hpp +65 -83
- data/vendor/datasketches-cpp/count/test/count_min_test.cpp +63 -7
- data/vendor/datasketches-cpp/cpc/include/compression_data.hpp +2 -0
- data/vendor/datasketches-cpp/cpc/include/cpc_compressor_impl.hpp +16 -7
- data/vendor/datasketches-cpp/cpc/include/cpc_sketch.hpp +1 -0
- data/vendor/datasketches-cpp/cpc/include/cpc_sketch_impl.hpp +54 -36
- data/vendor/datasketches-cpp/cpc/include/cpc_union_impl.hpp +27 -27
- data/vendor/datasketches-cpp/cpc/include/cpc_util.hpp +7 -7
- data/vendor/datasketches-cpp/cpc/include/icon_estimator.hpp +5 -5
- data/vendor/datasketches-cpp/cpc/include/u32_table_impl.hpp +16 -16
- data/vendor/datasketches-cpp/cpc/test/cpc_sketch_test.cpp +35 -0
- data/vendor/datasketches-cpp/fi/include/frequent_items_sketch.hpp +28 -3
- data/vendor/datasketches-cpp/fi/include/frequent_items_sketch_impl.hpp +41 -25
- data/vendor/datasketches-cpp/fi/include/reverse_purge_hash_map.hpp +3 -1
- data/vendor/datasketches-cpp/fi/include/reverse_purge_hash_map_impl.hpp +10 -5
- data/vendor/datasketches-cpp/fi/test/frequent_items_sketch_serialize_for_java.cpp +24 -0
- data/vendor/datasketches-cpp/fi/test/frequent_items_sketch_test.cpp +116 -0
- data/vendor/datasketches-cpp/filters/include/bloom_filter.hpp +1 -1
- data/vendor/datasketches-cpp/filters/include/bloom_filter_impl.hpp +32 -12
- data/vendor/datasketches-cpp/filters/test/bloom_filter_test.cpp +28 -1
- data/vendor/datasketches-cpp/hll/include/CouponHashSet-internal.hpp +1 -2
- data/vendor/datasketches-cpp/hll/include/CouponList-internal.hpp +19 -5
- data/vendor/datasketches-cpp/hll/include/CubicInterpolation-internal.hpp +4 -4
- data/vendor/datasketches-cpp/hll/include/HarmonicNumbers-internal.hpp +2 -1
- data/vendor/datasketches-cpp/hll/include/Hll4Array-internal.hpp +4 -4
- data/vendor/datasketches-cpp/hll/include/Hll4Array.hpp +1 -1
- data/vendor/datasketches-cpp/hll/include/Hll6Array-internal.hpp +3 -3
- data/vendor/datasketches-cpp/hll/include/Hll6Array.hpp +1 -1
- data/vendor/datasketches-cpp/hll/include/Hll8Array-internal.hpp +6 -3
- data/vendor/datasketches-cpp/hll/include/Hll8Array.hpp +1 -1
- data/vendor/datasketches-cpp/hll/include/HllArray-internal.hpp +35 -29
- data/vendor/datasketches-cpp/hll/include/HllArray.hpp +1 -1
- data/vendor/datasketches-cpp/hll/include/HllSketch-internal.hpp +3 -5
- data/vendor/datasketches-cpp/hll/include/HllSketchImpl-internal.hpp +5 -11
- data/vendor/datasketches-cpp/hll/include/HllSketchImpl.hpp +2 -4
- data/vendor/datasketches-cpp/hll/include/HllSketchImplFactory.hpp +11 -10
- data/vendor/datasketches-cpp/hll/include/HllUnion-internal.hpp +17 -8
- data/vendor/datasketches-cpp/hll/include/HllUtil.hpp +12 -3
- data/vendor/datasketches-cpp/hll/include/coupon_iterator-internal.hpp +2 -2
- data/vendor/datasketches-cpp/hll/include/coupon_iterator.hpp +3 -0
- data/vendor/datasketches-cpp/hll/include/hll.hpp +11 -4
- data/vendor/datasketches-cpp/hll/include/hll.private.hpp +19 -0
- data/vendor/datasketches-cpp/hll/test/CMakeLists.txt +2 -0
- data/vendor/datasketches-cpp/hll/test/CouponListTest.cpp +64 -0
- data/vendor/datasketches-cpp/hll/test/HllFullSizeTest.cpp +137 -0
- data/vendor/datasketches-cpp/hll/test/HllKxqRebuildTest.cpp +150 -0
- data/vendor/datasketches-cpp/hll/test/HllSketchTest.cpp +3 -3
- data/vendor/datasketches-cpp/hll/test/HllUnionTest.cpp +71 -1
- data/vendor/datasketches-cpp/kll/include/kll_helper_impl.hpp +10 -10
- data/vendor/datasketches-cpp/kll/include/kll_sketch.hpp +10 -1
- data/vendor/datasketches-cpp/kll/include/kll_sketch_impl.hpp +33 -24
- data/vendor/datasketches-cpp/kll/test/kll_sketch_deserialize_from_java_test.cpp +24 -0
- data/vendor/datasketches-cpp/kll/test/kll_sketch_serialize_for_java.cpp +10 -0
- data/vendor/datasketches-cpp/quantiles/include/quantiles_sketch.hpp +11 -2
- data/vendor/datasketches-cpp/quantiles/include/quantiles_sketch_impl.hpp +24 -15
- data/vendor/datasketches-cpp/req/include/req_sketch.hpp +9 -0
- data/vendor/datasketches-cpp/req/include/req_sketch_impl.hpp +24 -15
- data/vendor/datasketches-cpp/req/test/req_sketch_deserialize_from_java_test.cpp +46 -0
- data/vendor/datasketches-cpp/req/test/req_sketch_serialize_for_java.cpp +20 -0
- data/vendor/datasketches-cpp/req/test/req_sketch_test.cpp +70 -0
- data/vendor/datasketches-cpp/sampling/include/ebpps_sample_impl.hpp +17 -8
- data/vendor/datasketches-cpp/sampling/include/ebpps_sketch.hpp +13 -0
- data/vendor/datasketches-cpp/sampling/include/var_opt_sketch.hpp +10 -1
- data/vendor/datasketches-cpp/sampling/include/var_opt_sketch_impl.hpp +6 -7
- data/vendor/datasketches-cpp/sampling/include/var_opt_union.hpp +5 -1
- data/vendor/datasketches-cpp/sampling/include/var_opt_union_impl.hpp +2 -1
- data/vendor/datasketches-cpp/sampling/test/ebpps_allocation_test.cpp +1 -1
- data/vendor/datasketches-cpp/sampling/test/ebpps_sketch_test.cpp +2 -2
- data/vendor/datasketches-cpp/sampling/test/var_opt_allocation_test.cpp +1 -1
- data/vendor/datasketches-cpp/sampling/test/var_opt_sketch_test.cpp +10 -4
- data/vendor/datasketches-cpp/sampling/test/var_opt_union_test.cpp +12 -0
- data/vendor/datasketches-cpp/tdigest/include/tdigest.hpp +38 -2
- data/vendor/datasketches-cpp/tdigest/include/tdigest_impl.hpp +168 -9
- data/vendor/datasketches-cpp/tdigest/test/CMakeLists.txt +1 -0
- data/vendor/datasketches-cpp/tdigest/test/tdigest_iterator_test.cpp +274 -0
- data/vendor/datasketches-cpp/tdigest/test/tdigest_test.cpp +275 -0
- data/vendor/datasketches-cpp/theta/include/compact_theta_sketch_parser.hpp +2 -0
- data/vendor/datasketches-cpp/theta/include/compact_theta_sketch_parser_impl.hpp +24 -3
- data/vendor/datasketches-cpp/theta/include/theta_constants.hpp +4 -2
- data/vendor/datasketches-cpp/theta/include/theta_helpers.hpp +32 -0
- data/vendor/datasketches-cpp/theta/include/theta_set_difference_base_impl.hpp +4 -2
- data/vendor/datasketches-cpp/theta/include/theta_sketch.hpp +22 -4
- data/vendor/datasketches-cpp/theta/include/theta_sketch_impl.hpp +60 -38
- data/vendor/datasketches-cpp/theta/include/theta_union_base_impl.hpp +2 -6
- data/vendor/datasketches-cpp/theta/include/theta_update_sketch_base_impl.hpp +2 -2
- data/vendor/datasketches-cpp/theta/test/bit_packing_test.cpp +50 -0
- data/vendor/datasketches-cpp/theta/test/theta_a_not_b_test.cpp +22 -0
- data/vendor/datasketches-cpp/theta/test/theta_sketch_test.cpp +315 -0
- data/vendor/datasketches-cpp/tools/rat-check.sh +68 -0
- data/vendor/datasketches-cpp/tuple/include/array_tuple_sketch.hpp +35 -4
- data/vendor/datasketches-cpp/tuple/include/array_tuple_sketch_impl.hpp +2 -2
- data/vendor/datasketches-cpp/tuple/include/tuple_sketch.hpp +41 -0
- data/vendor/datasketches-cpp/tuple/include/tuple_sketch_impl.hpp +5 -4
- data/vendor/datasketches-cpp/tuple/test/tuple_sketch_test.cpp +59 -0
- data/vendor/datasketches-cpp/version.cfg.in +1 -1
- metadata +12 -2
|
@@ -73,7 +73,7 @@ bool cpc_sketch_alloc<A>::is_empty() const {
|
|
|
73
73
|
|
|
74
74
|
template<typename A>
|
|
75
75
|
double cpc_sketch_alloc<A>::get_estimate() const {
|
|
76
|
-
if (!was_merged) return get_hip_estimate();
|
|
76
|
+
if (!was_merged) { return get_hip_estimate(); }
|
|
77
77
|
return get_icon_estimate();
|
|
78
78
|
}
|
|
79
79
|
|
|
@@ -92,7 +92,7 @@ double cpc_sketch_alloc<A>::get_lower_bound(unsigned kappa) const {
|
|
|
92
92
|
if (kappa < 1 || kappa > 3) {
|
|
93
93
|
throw std::invalid_argument("kappa must be 1, 2 or 3");
|
|
94
94
|
}
|
|
95
|
-
if (!was_merged) return get_hip_confidence_lb<A>(*this, kappa);
|
|
95
|
+
if (!was_merged) { return get_hip_confidence_lb<A>(*this, kappa); }
|
|
96
96
|
return get_icon_confidence_lb<A>(*this, kappa);
|
|
97
97
|
}
|
|
98
98
|
|
|
@@ -101,13 +101,13 @@ double cpc_sketch_alloc<A>::get_upper_bound(unsigned kappa) const {
|
|
|
101
101
|
if (kappa < 1 || kappa > 3) {
|
|
102
102
|
throw std::invalid_argument("kappa must be 1, 2 or 3");
|
|
103
103
|
}
|
|
104
|
-
if (!was_merged) return get_hip_confidence_ub<A>(*this, kappa);
|
|
104
|
+
if (!was_merged) { return get_hip_confidence_ub<A>(*this, kappa); }
|
|
105
105
|
return get_icon_confidence_ub<A>(*this, kappa);
|
|
106
106
|
}
|
|
107
107
|
|
|
108
108
|
template<typename A>
|
|
109
109
|
void cpc_sketch_alloc<A>::update(const std::string& value) {
|
|
110
|
-
if (value.empty()) return;
|
|
110
|
+
if (value.empty()) { return; }
|
|
111
111
|
update(value.c_str(), value.length());
|
|
112
112
|
}
|
|
113
113
|
|
|
@@ -173,15 +173,15 @@ void cpc_sketch_alloc<A>::update(float value) {
|
|
|
173
173
|
}
|
|
174
174
|
|
|
175
175
|
static inline uint32_t row_col_from_two_hashes(uint64_t hash0, uint64_t hash1, uint8_t lg_k) {
|
|
176
|
-
if (lg_k > 26) throw std::logic_error("lg_k > 26");
|
|
176
|
+
if (lg_k > 26) { throw std::logic_error("lg_k > 26"); }
|
|
177
177
|
const uint32_t k = 1 << lg_k;
|
|
178
178
|
uint8_t col = count_leading_zeros_in_u64(hash1); // 0 <= col <= 64
|
|
179
|
-
if (col > 63) col = 63; // clip so that 0 <= col <= 63
|
|
179
|
+
if (col > 63) { col = 63; } // clip so that 0 <= col <= 63
|
|
180
180
|
const uint32_t row = hash0 & (k - 1);
|
|
181
181
|
uint32_t row_col = (row << 6) | col;
|
|
182
182
|
// To avoid the hash table's "empty" value, we change the row of the following pair.
|
|
183
183
|
// This case is extremely unlikely, but we might as well handle it.
|
|
184
|
-
if (row_col == UINT32_MAX) row_col ^= 1 << 6;
|
|
184
|
+
if (row_col == UINT32_MAX) { row_col ^= 1 << 6; }
|
|
185
185
|
return row_col;
|
|
186
186
|
}
|
|
187
187
|
|
|
@@ -195,7 +195,7 @@ void cpc_sketch_alloc<A>::update(const void* value, size_t size) {
|
|
|
195
195
|
template<typename A>
|
|
196
196
|
void cpc_sketch_alloc<A>::row_col_update(uint32_t row_col) {
|
|
197
197
|
const uint8_t col = row_col & 63;
|
|
198
|
-
if (col < first_interesting_column) return; // important speed optimization
|
|
198
|
+
if (col < first_interesting_column) { return; } // important speed optimization
|
|
199
199
|
// window size is 0 until sketch is promoted from sparse to windowed
|
|
200
200
|
if (sliding_window.size() == 0) {
|
|
201
201
|
update_sparse(row_col);
|
|
@@ -208,26 +208,26 @@ template<typename A>
|
|
|
208
208
|
void cpc_sketch_alloc<A>::update_sparse(uint32_t row_col) {
|
|
209
209
|
const uint32_t k = 1 << lg_k;
|
|
210
210
|
const uint64_t c32pre = static_cast<uint64_t>(num_coupons) << 5;
|
|
211
|
-
if (c32pre >= 3 * k) throw std::logic_error("c32pre >= 3 * k"); // C < 3K/32, in other words flavor == SPARSE
|
|
211
|
+
if (c32pre >= 3 * k) { throw std::logic_error("c32pre >= 3 * k"); } // C < 3K/32, in other words flavor == SPARSE
|
|
212
212
|
bool is_novel = surprising_value_table.maybe_insert(row_col);
|
|
213
213
|
if (is_novel) {
|
|
214
214
|
num_coupons++;
|
|
215
215
|
update_hip(row_col);
|
|
216
216
|
const uint64_t c32post = static_cast<uint64_t>(num_coupons) << 5;
|
|
217
|
-
if (c32post >= 3 * k) promote_sparse_to_windowed(); // C >= 3K/32
|
|
217
|
+
if (c32post >= 3 * k) { promote_sparse_to_windowed(); } // C >= 3K/32
|
|
218
218
|
}
|
|
219
219
|
}
|
|
220
220
|
|
|
221
221
|
// the flavor is HYBRID, PINNED, or SLIDING
|
|
222
222
|
template<typename A>
|
|
223
223
|
void cpc_sketch_alloc<A>::update_windowed(uint32_t row_col) {
|
|
224
|
-
if (window_offset > 56) throw std::logic_error("wrong window offset");
|
|
224
|
+
if (window_offset > 56) { throw std::logic_error("wrong window offset"); }
|
|
225
225
|
const uint32_t k = 1 << lg_k;
|
|
226
226
|
const uint64_t c32pre = static_cast<uint64_t>(num_coupons) << 5;
|
|
227
|
-
if (c32pre < 3 * k) throw std::logic_error("c32pre < 3 * k"); // C < 3K/32, in other words flavor >= HYBRID
|
|
227
|
+
if (c32pre < 3 * k) { throw std::logic_error("c32pre < 3 * k"); } // C < 3K/32, in other words flavor >= HYBRID
|
|
228
228
|
const uint64_t c8pre = static_cast<uint64_t>(num_coupons) << 3;
|
|
229
229
|
const uint64_t w8pre = static_cast<uint64_t>(window_offset) << 3;
|
|
230
|
-
if (c8pre >= (27 + w8pre) * k) throw std::logic_error("c8pre is wrong"); // C < (K * 27/8) + (K * window_offset)
|
|
230
|
+
if (c8pre >= (27 + w8pre) * k) { throw std::logic_error("c8pre is wrong"); } // C < (K * 27/8) + (K * window_offset)
|
|
231
231
|
|
|
232
232
|
bool is_novel = false;
|
|
233
233
|
const uint8_t col = row_col & 63;
|
|
@@ -235,7 +235,7 @@ void cpc_sketch_alloc<A>::update_windowed(uint32_t row_col) {
|
|
|
235
235
|
if (col < window_offset) { // track the surprising 0's "before" the window
|
|
236
236
|
is_novel = surprising_value_table.maybe_delete(row_col); // inverted logic
|
|
237
237
|
} else if (col < window_offset + 8) { // track the 8 bits inside the window
|
|
238
|
-
if (col < window_offset) throw std::logic_error("col < window_offset");
|
|
238
|
+
if (col < window_offset) { throw std::logic_error("col < window_offset"); }
|
|
239
239
|
const uint32_t row = row_col >> 6;
|
|
240
240
|
const uint8_t old_bits = sliding_window[row];
|
|
241
241
|
const uint8_t new_bits = old_bits | (1 << (col - window_offset));
|
|
@@ -244,7 +244,7 @@ void cpc_sketch_alloc<A>::update_windowed(uint32_t row_col) {
|
|
|
244
244
|
is_novel = true;
|
|
245
245
|
}
|
|
246
246
|
} else { // track the surprising 1's "after" the window
|
|
247
|
-
if (col < window_offset + 8) throw std::logic_error("col < window_offset + 8");
|
|
247
|
+
if (col < window_offset + 8) { throw std::logic_error("col < window_offset + 8"); }
|
|
248
248
|
is_novel = surprising_value_table.maybe_insert(row_col); // normal logic
|
|
249
249
|
}
|
|
250
250
|
|
|
@@ -254,9 +254,9 @@ void cpc_sketch_alloc<A>::update_windowed(uint32_t row_col) {
|
|
|
254
254
|
const uint64_t c8post = static_cast<uint64_t>(num_coupons) << 3;
|
|
255
255
|
if (c8post >= (27 + w8pre) * k) {
|
|
256
256
|
move_window();
|
|
257
|
-
if (window_offset < 1 || window_offset > 56) throw std::logic_error("wrong window offset");
|
|
257
|
+
if (window_offset < 1 || window_offset > 56) { throw std::logic_error("wrong window offset"); }
|
|
258
258
|
const uint64_t w8post = static_cast<uint64_t>(window_offset) << 3;
|
|
259
|
-
if (c8post >= (27 + w8post) * k) throw std::logic_error("c8pre is wrong"); // C < (K * 27/8) + (K * window_offset)
|
|
259
|
+
if (c8post >= (27 + w8post) * k) { throw std::logic_error("c8pre is wrong"); } // C < (K * 27/8) + (K * window_offset)
|
|
260
260
|
}
|
|
261
261
|
}
|
|
262
262
|
}
|
|
@@ -276,7 +276,7 @@ template<typename A>
|
|
|
276
276
|
void cpc_sketch_alloc<A>::promote_sparse_to_windowed() {
|
|
277
277
|
const uint32_t k = 1 << lg_k;
|
|
278
278
|
const uint64_t c32 = static_cast<uint64_t>(num_coupons) << 5;
|
|
279
|
-
if (!(c32 == 3 * k || (lg_k == 4 && c32 > 3 * k))) throw std::logic_error("wrong c32");
|
|
279
|
+
if (!(c32 == 3 * k || (lg_k == 4 && c32 > 3 * k))) { throw std::logic_error("wrong c32"); }
|
|
280
280
|
|
|
281
281
|
sliding_window.resize(k, 0); // zero the memory (because we will be OR'ing into it)
|
|
282
282
|
|
|
@@ -285,7 +285,7 @@ void cpc_sketch_alloc<A>::promote_sparse_to_windowed() {
|
|
|
285
285
|
const uint32_t* old_slots = surprising_value_table.get_slots();
|
|
286
286
|
const uint32_t old_num_slots = 1 << surprising_value_table.get_lg_size();
|
|
287
287
|
|
|
288
|
-
if (window_offset != 0) throw std::logic_error("window_offset != 0");
|
|
288
|
+
if (window_offset != 0) { throw std::logic_error("window_offset != 0"); }
|
|
289
289
|
|
|
290
290
|
for (uint32_t i = 0; i < old_num_slots; i++) {
|
|
291
291
|
const uint32_t row_col = old_slots[i];
|
|
@@ -297,7 +297,7 @@ void cpc_sketch_alloc<A>::promote_sparse_to_windowed() {
|
|
|
297
297
|
} else {
|
|
298
298
|
// cannot use u32_table::must_insert(), because it doesn't provide for growth
|
|
299
299
|
const bool is_novel = new_table.maybe_insert(row_col);
|
|
300
|
-
if (!is_novel) throw std::logic_error("is_novel != true");
|
|
300
|
+
if (!is_novel) { throw std::logic_error("is_novel != true"); }
|
|
301
301
|
}
|
|
302
302
|
}
|
|
303
303
|
}
|
|
@@ -308,17 +308,17 @@ void cpc_sketch_alloc<A>::promote_sparse_to_windowed() {
|
|
|
308
308
|
template<typename A>
|
|
309
309
|
void cpc_sketch_alloc<A>::move_window() {
|
|
310
310
|
const uint8_t new_offset = window_offset + 1;
|
|
311
|
-
if (new_offset > 56) throw std::logic_error("new_offset > 56");
|
|
312
|
-
if (new_offset != determine_correct_offset(lg_k, num_coupons)) throw std::logic_error("new_offset is wrong");
|
|
311
|
+
if (new_offset > 56) { throw std::logic_error("new_offset > 56"); }
|
|
312
|
+
if (new_offset != determine_correct_offset(lg_k, num_coupons)) { throw std::logic_error("new_offset is wrong"); }
|
|
313
313
|
|
|
314
|
-
if (sliding_window.size() == 0) throw std::logic_error("no sliding window");
|
|
314
|
+
if (sliding_window.size() == 0) { throw std::logic_error("no sliding window"); }
|
|
315
315
|
const uint32_t k = 1 << lg_k;
|
|
316
316
|
|
|
317
317
|
// Construct the full-sized bit matrix that corresponds to the sketch
|
|
318
318
|
vector_u64 bit_matrix = build_bit_matrix();
|
|
319
319
|
|
|
320
320
|
// refresh the KXP register on every 8th window shift.
|
|
321
|
-
if ((new_offset & 0x7) == 0) refresh_kxp(bit_matrix.data());
|
|
321
|
+
if ((new_offset & 0x7) == 0) { refresh_kxp(bit_matrix.data()); }
|
|
322
322
|
|
|
323
323
|
surprising_value_table.clear(); // the new number of surprises will be about the same
|
|
324
324
|
|
|
@@ -339,14 +339,14 @@ void cpc_sketch_alloc<A>::move_window() {
|
|
|
339
339
|
pattern = pattern ^ (static_cast<uint64_t>(1) << col); // erase the 1
|
|
340
340
|
const uint32_t row_col = (i << 6) | col;
|
|
341
341
|
const bool is_novel = surprising_value_table.maybe_insert(row_col);
|
|
342
|
-
if (!is_novel) throw std::logic_error("is_novel != true");
|
|
342
|
+
if (!is_novel) { throw std::logic_error("is_novel != true"); }
|
|
343
343
|
}
|
|
344
344
|
}
|
|
345
345
|
|
|
346
346
|
window_offset = new_offset;
|
|
347
347
|
|
|
348
348
|
first_interesting_column = count_trailing_zeros_in_u64(all_surprises_ored);
|
|
349
|
-
if (first_interesting_column > new_offset) first_interesting_column = new_offset; // corner case
|
|
349
|
+
if (first_interesting_column > new_offset) { first_interesting_column = new_offset; } // corner case
|
|
350
350
|
}
|
|
351
351
|
|
|
352
352
|
// The KXP register is a double with roughly 50 bits of precision, but
|
|
@@ -438,7 +438,7 @@ void cpc_sketch_alloc<A>::serialize(std::ostream& os) const {
|
|
|
438
438
|
write(os, compressed.table_num_entries);
|
|
439
439
|
// HIP values can be in two different places in the sequence of fields
|
|
440
440
|
// this is the first HIP decision point
|
|
441
|
-
if (has_hip) write_hip(os);
|
|
441
|
+
if (has_hip) { write_hip(os); }
|
|
442
442
|
}
|
|
443
443
|
if (has_table) {
|
|
444
444
|
write(os, compressed.table_data_words);
|
|
@@ -447,7 +447,7 @@ void cpc_sketch_alloc<A>::serialize(std::ostream& os) const {
|
|
|
447
447
|
write(os, compressed.window_data_words);
|
|
448
448
|
}
|
|
449
449
|
// this is the second HIP decision point
|
|
450
|
-
if (has_hip && !(has_table && has_window)) write_hip(os);
|
|
450
|
+
if (has_hip && !(has_table && has_window)) { write_hip(os); }
|
|
451
451
|
if (has_window) {
|
|
452
452
|
write(os, compressed.window_data.data(), compressed.window_data_words * sizeof(uint32_t));
|
|
453
453
|
}
|
|
@@ -494,7 +494,7 @@ auto cpc_sketch_alloc<A>::serialize(unsigned header_size_bytes) const -> vector_
|
|
|
494
494
|
ptr += copy_to_mem(compressed.table_num_entries, ptr);
|
|
495
495
|
// HIP values can be in two different places in the sequence of fields
|
|
496
496
|
// this is the first HIP decision point
|
|
497
|
-
if (has_hip) ptr += copy_hip_to_mem(ptr);
|
|
497
|
+
if (has_hip) { ptr += copy_hip_to_mem(ptr); }
|
|
498
498
|
}
|
|
499
499
|
if (has_table) {
|
|
500
500
|
ptr += copy_to_mem(compressed.table_data_words, ptr);
|
|
@@ -503,7 +503,7 @@ auto cpc_sketch_alloc<A>::serialize(unsigned header_size_bytes) const -> vector_
|
|
|
503
503
|
ptr += copy_to_mem(compressed.window_data_words, ptr);
|
|
504
504
|
}
|
|
505
505
|
// this is the second HIP decision point
|
|
506
|
-
if (has_hip && !(has_table && has_window)) ptr += copy_hip_to_mem(ptr);
|
|
506
|
+
if (has_hip && !(has_table && has_window)) { ptr += copy_hip_to_mem(ptr); }
|
|
507
507
|
if (has_window) {
|
|
508
508
|
ptr += copy_to_mem(compressed.window_data.data(), ptr, compressed.window_data_words * sizeof(uint32_t));
|
|
509
509
|
}
|
|
@@ -511,7 +511,7 @@ auto cpc_sketch_alloc<A>::serialize(unsigned header_size_bytes) const -> vector_
|
|
|
511
511
|
ptr += copy_to_mem(compressed.table_data.data(), ptr, compressed.table_data_words * sizeof(uint32_t));
|
|
512
512
|
}
|
|
513
513
|
}
|
|
514
|
-
if (ptr != bytes.data() + size) throw std::logic_error("serialized size mismatch");
|
|
514
|
+
if (ptr != bytes.data() + size) { throw std::logic_error("serialized size mismatch"); }
|
|
515
515
|
return bytes;
|
|
516
516
|
}
|
|
517
517
|
|
|
@@ -527,6 +527,7 @@ cpc_sketch_alloc<A> cpc_sketch_alloc<A>::deserialize(std::istream& is, uint64_t
|
|
|
527
527
|
const bool has_hip = flags_byte & (1 << flags::HAS_HIP);
|
|
528
528
|
const bool has_table = flags_byte & (1 << flags::HAS_TABLE);
|
|
529
529
|
const bool has_window = flags_byte & (1 << flags::HAS_WINDOW);
|
|
530
|
+
check_lg_k(lg_k);
|
|
530
531
|
compressed_state<A> compressed(allocator);
|
|
531
532
|
compressed.table_data_words = 0;
|
|
532
533
|
compressed.table_num_entries = 0;
|
|
@@ -561,7 +562,7 @@ cpc_sketch_alloc<A> cpc_sketch_alloc<A>::deserialize(std::istream& is, uint64_t
|
|
|
561
562
|
compressed.table_data.resize(compressed.table_data_words);
|
|
562
563
|
read(is, compressed.table_data.data(), compressed.table_data_words * sizeof(uint32_t));
|
|
563
564
|
}
|
|
564
|
-
if (!has_window) compressed.table_num_entries = num_coupons;
|
|
565
|
+
if (!has_window) { compressed.table_num_entries = num_coupons; }
|
|
565
566
|
}
|
|
566
567
|
|
|
567
568
|
uint8_t expected_preamble_ints = get_preamble_ints(num_coupons, has_hip, has_table, has_window);
|
|
@@ -581,10 +582,12 @@ cpc_sketch_alloc<A> cpc_sketch_alloc<A>::deserialize(std::istream& is, uint64_t
|
|
|
581
582
|
throw std::invalid_argument("Incompatible seed hashes: " + std::to_string(seed_hash) + ", "
|
|
582
583
|
+ std::to_string(compute_seed_hash(seed)));
|
|
583
584
|
}
|
|
585
|
+
check_num_coupons(lg_k, num_coupons, compressed.table_num_entries);
|
|
584
586
|
uncompressed_state<A> uncompressed(allocator);
|
|
585
587
|
get_compressor<A>().uncompress(compressed, uncompressed, lg_k, num_coupons);
|
|
586
|
-
if (!is.good())
|
|
587
|
-
throw std::runtime_error("error reading from std::istream");
|
|
588
|
+
if (!is.good()) {
|
|
589
|
+
throw std::runtime_error("error reading from std::istream");
|
|
590
|
+
}
|
|
588
591
|
return cpc_sketch_alloc(lg_k, num_coupons, first_interesting_column, std::move(uncompressed.table),
|
|
589
592
|
std::move(uncompressed.window), has_hip, kxp, hip_est_accum, seed);
|
|
590
593
|
}
|
|
@@ -611,6 +614,7 @@ cpc_sketch_alloc<A> cpc_sketch_alloc<A>::deserialize(const void* bytes, size_t s
|
|
|
611
614
|
const bool has_hip = flags_byte & (1 << flags::HAS_HIP);
|
|
612
615
|
const bool has_table = flags_byte & (1 << flags::HAS_TABLE);
|
|
613
616
|
const bool has_window = flags_byte & (1 << flags::HAS_WINDOW);
|
|
617
|
+
check_lg_k(lg_k);
|
|
614
618
|
ensure_minimum_memory(size, preamble_ints << 2);
|
|
615
619
|
compressed_state<A> compressed(allocator);
|
|
616
620
|
compressed.table_data_words = 0;
|
|
@@ -645,13 +649,13 @@ cpc_sketch_alloc<A> cpc_sketch_alloc<A>::deserialize(const void* bytes, size_t s
|
|
|
645
649
|
ptr += copy_from_mem(ptr, hip_est_accum);
|
|
646
650
|
}
|
|
647
651
|
if (has_window) {
|
|
648
|
-
compressed.window_data.resize(compressed.window_data_words);
|
|
649
652
|
check_memory_size(ptr - base + (compressed.window_data_words * sizeof(uint32_t)), size);
|
|
653
|
+
compressed.window_data.resize(compressed.window_data_words);
|
|
650
654
|
ptr += copy_from_mem(ptr, compressed.window_data.data(), compressed.window_data_words * sizeof(uint32_t));
|
|
651
655
|
}
|
|
652
656
|
if (has_table) {
|
|
653
|
-
compressed.table_data.resize(compressed.table_data_words);
|
|
654
657
|
check_memory_size(ptr - base + (compressed.table_data_words * sizeof(uint32_t)), size);
|
|
658
|
+
compressed.table_data.resize(compressed.table_data_words);
|
|
655
659
|
ptr += copy_from_mem(ptr, compressed.table_data.data(), compressed.table_data_words * sizeof(uint32_t));
|
|
656
660
|
}
|
|
657
661
|
if (!has_window) compressed.table_num_entries = num_coupons;
|
|
@@ -675,6 +679,7 @@ cpc_sketch_alloc<A> cpc_sketch_alloc<A>::deserialize(const void* bytes, size_t s
|
|
|
675
679
|
throw std::invalid_argument("Incompatible seed hashes: " + std::to_string(seed_hash) + ", "
|
|
676
680
|
+ std::to_string(compute_seed_hash(seed)));
|
|
677
681
|
}
|
|
682
|
+
check_num_coupons(lg_k, num_coupons, compressed.table_num_entries);
|
|
678
683
|
uncompressed_state<A> uncompressed(allocator);
|
|
679
684
|
get_compressor<A>().uncompress(compressed, uncompressed, lg_k, num_coupons);
|
|
680
685
|
return cpc_sketch_alloc(lg_k, num_coupons, first_interesting_column, std::move(uncompressed.table),
|
|
@@ -719,6 +724,19 @@ size_t cpc_sketch_alloc<A>::get_max_serialized_size_bytes(uint8_t lg_k) {
|
|
|
719
724
|
return (int) (CPC_EMPIRICAL_MAX_SIZE_FACTOR * k) + CPC_MAX_PREAMBLE_SIZE_BYTES;
|
|
720
725
|
}
|
|
721
726
|
|
|
727
|
+
template<typename A>
|
|
728
|
+
void cpc_sketch_alloc<A>::check_num_coupons(uint8_t lg_k, uint32_t num_coupons, uint32_t num_pairs) {
|
|
729
|
+
// at most one coupon per bit of the k x 64 bit matrix, and surprising values are a subset of coupons
|
|
730
|
+
if (num_coupons > (static_cast<uint64_t>(1) << lg_k) * 64) {
|
|
731
|
+
throw std::invalid_argument("Possible corruption: num_coupons " + std::to_string(num_coupons)
|
|
732
|
+
+ " exceeds the capacity for lg_k " + std::to_string(lg_k));
|
|
733
|
+
}
|
|
734
|
+
if (num_pairs > num_coupons) {
|
|
735
|
+
throw std::invalid_argument("Possible corruption: table entries " + std::to_string(num_pairs)
|
|
736
|
+
+ " exceed num_coupons " + std::to_string(num_coupons));
|
|
737
|
+
}
|
|
738
|
+
}
|
|
739
|
+
|
|
722
740
|
template<typename A>
|
|
723
741
|
void cpc_sketch_alloc<A>::check_lg_k(uint8_t lg_k) {
|
|
724
742
|
if (lg_k < cpc_constants::MIN_LG_K || lg_k > cpc_constants::MAX_LG_K) {
|
|
@@ -109,15 +109,15 @@ void cpc_union_alloc<A>::internal_update(S&& sketch) {
|
|
|
109
109
|
+ std::to_string(seed_hash_sketch));
|
|
110
110
|
}
|
|
111
111
|
const auto src_flavor = sketch.determine_flavor();
|
|
112
|
-
if (cpc_sketch_alloc<A>::flavor::EMPTY == src_flavor) return;
|
|
112
|
+
if (cpc_sketch_alloc<A>::flavor::EMPTY == src_flavor) { return; }
|
|
113
113
|
|
|
114
|
-
if (sketch.get_lg_k() < lg_k) reduce_k(sketch.get_lg_k());
|
|
115
|
-
if (sketch.get_lg_k() < lg_k) throw std::logic_error("sketch lg_k < union lg_k");
|
|
114
|
+
if (sketch.get_lg_k() < lg_k) { reduce_k(sketch.get_lg_k()); }
|
|
115
|
+
if (sketch.get_lg_k() < lg_k) { throw std::logic_error("sketch lg_k < union lg_k"); }
|
|
116
116
|
|
|
117
|
-
if (accumulator == nullptr && bit_matrix.size() == 0) throw std::logic_error("both accumulator and bit matrix are absent");
|
|
117
|
+
if (accumulator == nullptr && bit_matrix.size() == 0) { throw std::logic_error("both accumulator and bit matrix are absent"); }
|
|
118
118
|
|
|
119
119
|
if (cpc_sketch_alloc<A>::flavor::SPARSE == src_flavor && accumulator != nullptr) { // Case A
|
|
120
|
-
if (bit_matrix.size() > 0) throw std::logic_error("union bit_matrix is not expected");
|
|
120
|
+
if (bit_matrix.size() > 0) { throw std::logic_error("union bit_matrix is not expected"); }
|
|
121
121
|
const auto initial_dest_flavor = accumulator->determine_flavor();
|
|
122
122
|
if (cpc_sketch_alloc<A>::flavor::EMPTY != initial_dest_flavor &&
|
|
123
123
|
cpc_sketch_alloc<A>::flavor::SPARSE != initial_dest_flavor) throw std::logic_error("wrong flavor");
|
|
@@ -138,24 +138,24 @@ void cpc_union_alloc<A>::internal_update(S&& sketch) {
|
|
|
138
138
|
}
|
|
139
139
|
|
|
140
140
|
if (cpc_sketch_alloc<A>::flavor::SPARSE == src_flavor && bit_matrix.size() > 0) { // Case B
|
|
141
|
-
if (accumulator != nullptr) throw std::logic_error("union accumulator != null");
|
|
141
|
+
if (accumulator != nullptr) { throw std::logic_error("union accumulator != null"); }
|
|
142
142
|
or_table_into_matrix(sketch.surprising_value_table);
|
|
143
143
|
return;
|
|
144
144
|
}
|
|
145
145
|
|
|
146
146
|
if (cpc_sketch_alloc<A>::flavor::HYBRID != src_flavor && cpc_sketch_alloc<A>::flavor::PINNED != src_flavor
|
|
147
|
-
&& cpc_sketch_alloc<A>::flavor::SLIDING != src_flavor) throw std::logic_error("wrong flavor");
|
|
147
|
+
&& cpc_sketch_alloc<A>::flavor::SLIDING != src_flavor) { throw std::logic_error("wrong flavor"); }
|
|
148
148
|
|
|
149
149
|
// source is past SPARSE mode, so make sure that dest is a bit matrix
|
|
150
150
|
if (accumulator != nullptr) {
|
|
151
|
-
if (bit_matrix.size() > 0) throw std::logic_error("union bit matrix is not expected");
|
|
151
|
+
if (bit_matrix.size() > 0) { throw std::logic_error("union bit matrix is not expected"); }
|
|
152
152
|
const auto dst_flavor = accumulator->determine_flavor();
|
|
153
153
|
if (cpc_sketch_alloc<A>::flavor::EMPTY != dst_flavor && cpc_sketch_alloc<A>::flavor::SPARSE != dst_flavor) {
|
|
154
154
|
throw std::logic_error("wrong flavor");
|
|
155
155
|
}
|
|
156
156
|
switch_to_bit_matrix();
|
|
157
157
|
}
|
|
158
|
-
if (bit_matrix.size() == 0) throw std::logic_error("union bit_matrix is expected");
|
|
158
|
+
if (bit_matrix.size() == 0) { throw std::logic_error("union bit_matrix is expected"); }
|
|
159
159
|
|
|
160
160
|
if (cpc_sketch_alloc<A>::flavor::HYBRID == src_flavor || cpc_sketch_alloc<A>::flavor::PINNED == src_flavor) { // Case C
|
|
161
161
|
or_window_into_matrix(sketch.sliding_window, sketch.window_offset, sketch.get_lg_k());
|
|
@@ -165,7 +165,7 @@ void cpc_union_alloc<A>::internal_update(S&& sketch) {
|
|
|
165
165
|
|
|
166
166
|
// SLIDING mode involves inverted logic, so we can't just walk the source sketch.
|
|
167
167
|
// Instead, we convert it to a bitMatrix that can be OR'ed into the destination.
|
|
168
|
-
if (cpc_sketch_alloc<A>::flavor::SLIDING != src_flavor) throw std::logic_error("wrong flavor"); // Case D
|
|
168
|
+
if (cpc_sketch_alloc<A>::flavor::SLIDING != src_flavor) { throw std::logic_error("wrong flavor"); } // Case D
|
|
169
169
|
vector_u64 src_matrix = sketch.build_bit_matrix();
|
|
170
170
|
or_matrix_into_matrix(src_matrix, sketch.get_lg_k());
|
|
171
171
|
}
|
|
@@ -173,20 +173,20 @@ void cpc_union_alloc<A>::internal_update(S&& sketch) {
|
|
|
173
173
|
template<typename A>
|
|
174
174
|
cpc_sketch_alloc<A> cpc_union_alloc<A>::get_result() const {
|
|
175
175
|
if (accumulator != nullptr) {
|
|
176
|
-
if (bit_matrix.size() > 0) throw std::logic_error("bit_matrix is not expected");
|
|
176
|
+
if (bit_matrix.size() > 0) { throw std::logic_error("bit_matrix is not expected"); }
|
|
177
177
|
return get_result_from_accumulator();
|
|
178
178
|
}
|
|
179
|
-
if (bit_matrix.size() == 0) throw std::logic_error("bit_matrix is expected");
|
|
179
|
+
if (bit_matrix.size() == 0) { throw std::logic_error("bit_matrix is expected"); }
|
|
180
180
|
return get_result_from_bit_matrix();
|
|
181
181
|
}
|
|
182
182
|
|
|
183
183
|
template<typename A>
|
|
184
184
|
cpc_sketch_alloc<A> cpc_union_alloc<A>::get_result_from_accumulator() const {
|
|
185
|
-
if (lg_k != accumulator->get_lg_k()) throw std::logic_error("lg_k != accumulator->lg_k");
|
|
185
|
+
if (lg_k != accumulator->get_lg_k()) { throw std::logic_error("lg_k != accumulator->lg_k"); }
|
|
186
186
|
if (accumulator->get_num_coupons() == 0) {
|
|
187
187
|
return cpc_sketch_alloc<A>(lg_k, seed, accumulator->get_allocator());
|
|
188
188
|
}
|
|
189
|
-
if (accumulator->determine_flavor() != cpc_sketch_alloc<A>::flavor::SPARSE) throw std::logic_error("wrong flavor");
|
|
189
|
+
if (accumulator->determine_flavor() != cpc_sketch_alloc<A>::flavor::SPARSE) { throw std::logic_error("wrong flavor"); }
|
|
190
190
|
cpc_sketch_alloc<A> copy(*accumulator);
|
|
191
191
|
copy.was_merged = true;
|
|
192
192
|
return copy;
|
|
@@ -199,7 +199,7 @@ cpc_sketch_alloc<A> cpc_union_alloc<A>::get_result_from_bit_matrix() const {
|
|
|
199
199
|
|
|
200
200
|
const auto flavor = cpc_sketch_alloc<A>::determine_flavor(lg_k, num_coupons);
|
|
201
201
|
if (flavor != cpc_sketch_alloc<A>::flavor::HYBRID && flavor != cpc_sketch_alloc<A>::flavor::PINNED
|
|
202
|
-
&& flavor != cpc_sketch_alloc<A>::flavor::SLIDING) throw std::logic_error("wrong flavor");
|
|
202
|
+
&& flavor != cpc_sketch_alloc<A>::flavor::SLIDING) { throw std::logic_error("wrong flavor"); }
|
|
203
203
|
|
|
204
204
|
const uint8_t offset = cpc_sketch_alloc<A>::determine_correct_offset(lg_k, num_coupons);
|
|
205
205
|
|
|
@@ -208,7 +208,7 @@ cpc_sketch_alloc<A> cpc_union_alloc<A>::get_result_from_bit_matrix() const {
|
|
|
208
208
|
|
|
209
209
|
// dynamically growing caused snowplow effect
|
|
210
210
|
uint8_t table_lg_size = lg_k - 4; // K/16; in some cases this will end up being oversized
|
|
211
|
-
if (table_lg_size < 2) table_lg_size = 2;
|
|
211
|
+
if (table_lg_size < 2) { table_lg_size = 2; }
|
|
212
212
|
u32_table<A> table(table_lg_size, 6 + lg_k, bit_matrix.get_allocator());
|
|
213
213
|
|
|
214
214
|
// the following should work even when the offset is zero
|
|
@@ -229,14 +229,14 @@ cpc_sketch_alloc<A> cpc_union_alloc<A>::get_result_from_bit_matrix() const {
|
|
|
229
229
|
pattern = pattern ^ (static_cast<uint64_t>(1) << col); // erase the 1
|
|
230
230
|
const uint32_t row_col = (i << 6) | col;
|
|
231
231
|
bool is_novel = table.maybe_insert(row_col);
|
|
232
|
-
if (!is_novel) throw std::logic_error("is_novel != true");
|
|
232
|
+
if (!is_novel) { throw std::logic_error("is_novel != true"); }
|
|
233
233
|
}
|
|
234
234
|
}
|
|
235
235
|
|
|
236
236
|
// at this point we could shrink an oversized hash table, but the relative waste isn't very big
|
|
237
237
|
|
|
238
238
|
uint8_t first_interesting_column = count_trailing_zeros_in_u64(all_surprises_ored);
|
|
239
|
-
if (first_interesting_column > offset) first_interesting_column = offset; // corner case
|
|
239
|
+
if (first_interesting_column > offset) { first_interesting_column = offset; } // corner case
|
|
240
240
|
|
|
241
241
|
// HIP-related fields will contain zeros, and that is okay
|
|
242
242
|
return cpc_sketch_alloc<A>(lg_k, num_coupons, first_interesting_column, std::move(table), std::move(sliding_window), false, 0, 0, seed);
|
|
@@ -260,9 +260,9 @@ void cpc_union_alloc<A>::walk_table_updating_sketch(const u32_table<A>& table) {
|
|
|
260
260
|
// Using a golden ratio stride fixes the snowplow effect.
|
|
261
261
|
const double golden = 0.6180339887498949025;
|
|
262
262
|
uint32_t stride = static_cast<uint32_t>(golden * static_cast<double>(num_slots));
|
|
263
|
-
if (stride < 2) throw std::logic_error("stride < 2");
|
|
264
|
-
if (stride == ((stride >> 1) << 1)) stride += 1; // force the stride to be odd
|
|
265
|
-
if (stride < 3 || stride >= num_slots) throw std::out_of_range("stride out of range");
|
|
263
|
+
if (stride < 2) { throw std::logic_error("stride < 2"); }
|
|
264
|
+
if (stride == ((stride >> 1) << 1)) { stride += 1; } // force the stride to be odd
|
|
265
|
+
if (stride < 3 || stride >= num_slots) { throw std::out_of_range("stride out of range"); }
|
|
266
266
|
|
|
267
267
|
for (uint32_t i = 0, j = 0; i < num_slots; i++, j += stride) {
|
|
268
268
|
j &= num_slots - 1;
|
|
@@ -290,7 +290,7 @@ void cpc_union_alloc<A>::or_table_into_matrix(const u32_table<A>& table) {
|
|
|
290
290
|
|
|
291
291
|
template<typename A>
|
|
292
292
|
void cpc_union_alloc<A>::or_window_into_matrix(const vector_bytes& sliding_window, uint8_t offset, uint8_t src_lg_k) {
|
|
293
|
-
if (lg_k > src_lg_k) throw std::logic_error("dst LgK > src LgK");
|
|
293
|
+
if (lg_k > src_lg_k) { throw std::logic_error("dst LgK > src LgK"); }
|
|
294
294
|
const uint64_t dst_mask = (1 << lg_k) - 1; // downsamples when dst lgK < src LgK
|
|
295
295
|
const uint32_t src_k = 1 << src_lg_k;
|
|
296
296
|
for (uint32_t src_row = 0; src_row < src_k; src_row++) {
|
|
@@ -300,7 +300,7 @@ void cpc_union_alloc<A>::or_window_into_matrix(const vector_bytes& sliding_windo
|
|
|
300
300
|
|
|
301
301
|
template<typename A>
|
|
302
302
|
void cpc_union_alloc<A>::or_matrix_into_matrix(const vector_u64& src_matrix, uint8_t src_lg_k) {
|
|
303
|
-
if (lg_k > src_lg_k) throw std::logic_error("dst LgK > src LgK");
|
|
303
|
+
if (lg_k > src_lg_k) { throw std::logic_error("dst LgK > src LgK"); }
|
|
304
304
|
const uint64_t dst_mask = (1 << lg_k) - 1; // downsamples when dst lgK < src LgK
|
|
305
305
|
const uint32_t src_k = 1 << src_lg_k;
|
|
306
306
|
for (uint32_t src_row = 0; src_row < src_k; src_row++) {
|
|
@@ -310,11 +310,11 @@ void cpc_union_alloc<A>::or_matrix_into_matrix(const vector_u64& src_matrix, uin
|
|
|
310
310
|
|
|
311
311
|
template<typename A>
|
|
312
312
|
void cpc_union_alloc<A>::reduce_k(uint8_t new_lg_k) {
|
|
313
|
-
if (new_lg_k >= lg_k) throw std::logic_error("new LgK >= union lgK");
|
|
314
|
-
if (accumulator == nullptr && bit_matrix.size() == 0) throw std::logic_error("both accumulator and bit_matrix are absent");
|
|
313
|
+
if (new_lg_k >= lg_k) { throw std::logic_error("new LgK >= union lgK"); }
|
|
314
|
+
if (accumulator == nullptr && bit_matrix.size() == 0) { throw std::logic_error("both accumulator and bit_matrix are absent"); }
|
|
315
315
|
|
|
316
316
|
if (bit_matrix.size() > 0) { // downsample the unioner's bit matrix
|
|
317
|
-
if (accumulator != nullptr) throw std::logic_error("accumulator is not null");
|
|
317
|
+
if (accumulator != nullptr) { throw std::logic_error("accumulator is not null"); }
|
|
318
318
|
vector_u64 old_matrix = std::move(bit_matrix);
|
|
319
319
|
const uint8_t old_lg_k = lg_k;
|
|
320
320
|
const uint32_t new_k = 1 << new_lg_k;
|
|
@@ -325,7 +325,7 @@ void cpc_union_alloc<A>::reduce_k(uint8_t new_lg_k) {
|
|
|
325
325
|
}
|
|
326
326
|
|
|
327
327
|
if (accumulator != nullptr) { // downsample the unioner's sketch
|
|
328
|
-
if (bit_matrix.size() > 0) throw std::logic_error("bit_matrix is not expected");
|
|
328
|
+
if (bit_matrix.size() > 0) { throw std::logic_error("bit_matrix is not expected"); }
|
|
329
329
|
if (!accumulator->is_empty()) {
|
|
330
330
|
cpc_sketch_alloc<A> old_accumulator(*accumulator);
|
|
331
331
|
*accumulator = cpc_sketch_alloc<A>(new_lg_k, seed, old_accumulator.get_allocator());
|
|
@@ -25,19 +25,19 @@
|
|
|
25
25
|
namespace datasketches {
|
|
26
26
|
|
|
27
27
|
static inline uint64_t divide_longs_rounding_up(uint64_t x, uint64_t y) {
|
|
28
|
-
if (y == 0) throw std::invalid_argument("divide_longs_rounding_up: bad argument");
|
|
28
|
+
if (y == 0) { throw std::invalid_argument("divide_longs_rounding_up: bad argument"); }
|
|
29
29
|
const uint64_t quotient = x / y;
|
|
30
|
-
if (quotient * y == x) return (quotient);
|
|
31
|
-
else return quotient + 1;
|
|
30
|
+
if (quotient * y == x) { return (quotient); }
|
|
31
|
+
else { return quotient + 1; }
|
|
32
32
|
}
|
|
33
33
|
|
|
34
34
|
static inline uint8_t floor_log2_of_long(uint64_t x) {
|
|
35
|
-
if (x < 1) throw std::invalid_argument("floor_log2_of_long: bad argument");
|
|
35
|
+
if (x < 1) { throw std::invalid_argument("floor_log2_of_long: bad argument"); }
|
|
36
36
|
uint8_t p = 0;
|
|
37
37
|
uint64_t y = 1;
|
|
38
38
|
while (true) {
|
|
39
|
-
if (y == x) return p;
|
|
40
|
-
if (y > x) return p - 1;
|
|
39
|
+
if (y == x) { return p; }
|
|
40
|
+
if (y > x) { return p - 1; }
|
|
41
41
|
p += 1;
|
|
42
42
|
y <<= 1;
|
|
43
43
|
}
|
|
@@ -98,7 +98,7 @@ static inline uint32_t warren_count_bits_set_in_matrix(const uint64_t* array, ui
|
|
|
98
98
|
}
|
|
99
99
|
|
|
100
100
|
static inline uint32_t count_bits_set_in_matrix(const uint64_t* a, uint32_t length) {
|
|
101
|
-
if ((length & 0x7) != 0) throw std::invalid_argument("the length of the array must be a multiple of 8");
|
|
101
|
+
if ((length & 0x7) != 0) { throw std::invalid_argument("the length of the array must be a multiple of 8"); }
|
|
102
102
|
uint32_t total = 0;
|
|
103
103
|
uint64_t ones, twos, twos_a, twos_b, fours, fours_a, fours_b, eights;
|
|
104
104
|
fours = twos = ones = 0;
|
|
@@ -246,14 +246,14 @@ static inline double icon_exponential_approximation(double k, double c) {
|
|
|
246
246
|
}
|
|
247
247
|
|
|
248
248
|
static inline double compute_icon_estimate(uint8_t lg_k, uint32_t c) {
|
|
249
|
-
if (lg_k < ICON_MIN_LOG_K || lg_k > ICON_MAX_LOG_K) throw std::out_of_range("lg_k out of range");
|
|
250
|
-
if (c < 2) return ((c == 0) ? 0.0 : 1.0);
|
|
249
|
+
if (lg_k < ICON_MIN_LOG_K || lg_k > ICON_MAX_LOG_K) { throw std::out_of_range("lg_k out of range"); }
|
|
250
|
+
if (c < 2) { return ((c == 0) ? 0.0 : 1.0); }
|
|
251
251
|
const uint32_t k = 1 << lg_k;
|
|
252
252
|
const double double_k = static_cast<double>(k);
|
|
253
253
|
const double double_c = static_cast<double>(c);
|
|
254
254
|
// Differing thresholds ensure that the approximated estimator is monotonically increasing.
|
|
255
255
|
const double threshold_factor = ((lg_k < 14) ? 5.7 : 5.6);
|
|
256
|
-
if (double_c > (threshold_factor * double_k)) return icon_exponential_approximation(double_k, double_c);
|
|
256
|
+
if (double_c > (threshold_factor * double_k)) { return icon_exponential_approximation(double_k, double_c); }
|
|
257
257
|
const double factor = evaluate_polynomial(
|
|
258
258
|
ICON_POLYNOMIAL_COEFFICIENTS,
|
|
259
259
|
ICON_POLYNOMIAL_NUM_COEFFICIENTS * (lg_k - ICON_MIN_LOG_K),
|
|
@@ -265,8 +265,8 @@ static inline double compute_icon_estimate(uint8_t lg_k, uint32_t c) {
|
|
|
265
265
|
// The somewhat arbitrary constant 66.774757 is baked into the table ICON_POLYNOMIAL_COEFFICIENTS
|
|
266
266
|
const double term = 1.0 + (ratio * ratio * ratio / 66.774757);
|
|
267
267
|
const double result = double_c * factor * term;
|
|
268
|
-
if (result >= double_c) return result;
|
|
269
|
-
else return double_c;
|
|
268
|
+
if (result >= double_c) { return result; }
|
|
269
|
+
else { return double_c; }
|
|
270
270
|
}
|
|
271
271
|
|
|
272
272
|
} /* namespace datasketches */
|
|
@@ -43,8 +43,8 @@ num_valid_bits(num_valid_bits),
|
|
|
43
43
|
num_items(0),
|
|
44
44
|
slots(1ULL << lg_size, UINT32_MAX, allocator)
|
|
45
45
|
{
|
|
46
|
-
if (lg_size < 2) throw std::invalid_argument("lg_size must be >= 2");
|
|
47
|
-
if (num_valid_bits < 1 || num_valid_bits > 32) throw std::invalid_argument("num_valid_bits must be between 1 and 32");
|
|
46
|
+
if (lg_size < 2) { throw std::invalid_argument("lg_size must be >= 2"); }
|
|
47
|
+
if (num_valid_bits < 1 || num_valid_bits > 32) { throw std::invalid_argument("num_valid_bits must be between 1 and 32"); }
|
|
48
48
|
}
|
|
49
49
|
|
|
50
50
|
template<typename A>
|
|
@@ -71,8 +71,8 @@ void u32_table<A>::clear() {
|
|
|
71
71
|
template<typename A>
|
|
72
72
|
bool u32_table<A>::maybe_insert(uint32_t item) {
|
|
73
73
|
const uint32_t index = lookup(item);
|
|
74
|
-
if (slots[index] == item) return false;
|
|
75
|
-
if (slots[index] != UINT32_MAX) throw std::logic_error("could not insert");
|
|
74
|
+
if (slots[index] == item) { return false; }
|
|
75
|
+
if (slots[index] != UINT32_MAX) { throw std::logic_error("could not insert"); }
|
|
76
76
|
slots[index] = item;
|
|
77
77
|
num_items++;
|
|
78
78
|
if (U32_TABLE_UPSIZE_DENOM * num_items > U32_TABLE_UPSIZE_NUMER * (1 << lg_size)) {
|
|
@@ -84,9 +84,9 @@ bool u32_table<A>::maybe_insert(uint32_t item) {
|
|
|
84
84
|
template<typename A>
|
|
85
85
|
bool u32_table<A>::maybe_delete(uint32_t item) {
|
|
86
86
|
const uint32_t index = lookup(item);
|
|
87
|
-
if (slots[index] == UINT32_MAX) return false;
|
|
88
|
-
if (slots[index] != item) throw std::logic_error("item does not exist");
|
|
89
|
-
if (num_items == 0) throw std::logic_error("delete error");
|
|
87
|
+
if (slots[index] == UINT32_MAX) { return false; }
|
|
88
|
+
if (slots[index] != item) { throw std::logic_error("item does not exist"); }
|
|
89
|
+
if (num_items == 0) { throw std::logic_error("delete error"); }
|
|
90
90
|
// delete the item
|
|
91
91
|
slots[index] = UINT32_MAX;
|
|
92
92
|
num_items--;
|
|
@@ -129,7 +129,7 @@ uint32_t u32_table<A>::lookup(uint32_t item) const {
|
|
|
129
129
|
const uint32_t mask = size - 1;
|
|
130
130
|
const uint8_t shift = num_valid_bits - lg_size;
|
|
131
131
|
uint32_t probe = item >> shift;
|
|
132
|
-
if (probe > mask) throw std::logic_error("probe out of range");
|
|
132
|
+
if (probe > mask) { throw std::logic_error("probe out of range"); }
|
|
133
133
|
while (slots[probe] != item && slots[probe] != UINT32_MAX) {
|
|
134
134
|
probe = (probe + 1) & mask;
|
|
135
135
|
}
|
|
@@ -140,17 +140,17 @@ uint32_t u32_table<A>::lookup(uint32_t item) const {
|
|
|
140
140
|
template<typename A>
|
|
141
141
|
void u32_table<A>::must_insert(uint32_t item) {
|
|
142
142
|
const uint32_t index = lookup(item);
|
|
143
|
-
if (slots[index] == item) throw std::logic_error("item exists");
|
|
144
|
-
if (slots[index] != UINT32_MAX) throw std::logic_error("could not insert");
|
|
143
|
+
if (slots[index] == item) { throw std::logic_error("item exists"); }
|
|
144
|
+
if (slots[index] != UINT32_MAX) { throw std::logic_error("could not insert"); }
|
|
145
145
|
slots[index] = item;
|
|
146
146
|
}
|
|
147
147
|
|
|
148
148
|
template<typename A>
|
|
149
149
|
void u32_table<A>::rebuild(uint8_t new_lg_size) {
|
|
150
|
-
if (new_lg_size < 2) throw std::logic_error("lg_size must be >= 2");
|
|
150
|
+
if (new_lg_size < 2) { throw std::logic_error("lg_size must be >= 2"); }
|
|
151
151
|
const uint32_t old_size = 1 << lg_size;
|
|
152
152
|
const uint32_t new_size = 1 << new_lg_size;
|
|
153
|
-
if (new_size <= num_items) throw std::logic_error("new_size <= num_items");
|
|
153
|
+
if (new_size <= num_items) { throw std::logic_error("new_size <= num_items"); }
|
|
154
154
|
vector_u32 old_slots = std::move(slots);
|
|
155
155
|
slots = vector_u32(new_size, UINT32_MAX, old_slots.get_allocator());
|
|
156
156
|
lg_size = new_lg_size;
|
|
@@ -169,7 +169,7 @@ void u32_table<A>::rebuild(uint8_t new_lg_size) {
|
|
|
169
169
|
// The result is nearly sorted, so make sure to use an efficient sort for that case
|
|
170
170
|
template<typename A>
|
|
171
171
|
auto u32_table<A>::unwrapping_get_items() const -> vector_u32 {
|
|
172
|
-
if (num_items == 0) return vector_u32(slots.get_allocator());
|
|
172
|
+
if (num_items == 0) { return vector_u32(slots.get_allocator()); }
|
|
173
173
|
const uint32_t table_size = 1 << lg_size;
|
|
174
174
|
vector_u32 result(num_items, 0, slots.get_allocator());
|
|
175
175
|
size_t i = 0;
|
|
@@ -187,9 +187,9 @@ auto u32_table<A>::unwrapping_get_items() const -> vector_u32 {
|
|
|
187
187
|
// the rest of the table is processed normally
|
|
188
188
|
while (i < table_size) {
|
|
189
189
|
const uint32_t item = slots[i++];
|
|
190
|
-
if (item != UINT32_MAX) result[l++] = item;
|
|
190
|
+
if (item != UINT32_MAX) { result[l++] = item; }
|
|
191
191
|
}
|
|
192
|
-
if (l != r + 1) throw std::logic_error("unwrapping error");
|
|
192
|
+
if (l != r + 1) { throw std::logic_error("unwrapping error"); }
|
|
193
193
|
return result;
|
|
194
194
|
}
|
|
195
195
|
|
|
@@ -213,7 +213,7 @@ void u32_table<A>::merge(
|
|
|
213
213
|
else if (arr_a[a] < arr_b[b]) { arr_c[c] = arr_a[a++]; }
|
|
214
214
|
else { arr_c[c] = arr_b[b++]; }
|
|
215
215
|
}
|
|
216
|
-
if (a != lim_a || b != lim_b) throw std::logic_error("merging error");
|
|
216
|
+
if (a != lim_a || b != lim_b) { throw std::logic_error("merging error"); }
|
|
217
217
|
}
|
|
218
218
|
|
|
219
219
|
// In applications where the input array is already nearly sorted,
|