datasketches 0.5.2 → 0.5.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (115) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +4 -0
  3. data/ext/datasketches/theta_wrapper.cpp +1 -1
  4. data/lib/datasketches/version.rb +1 -1
  5. data/vendor/datasketches-cpp/CMakeLists.txt +5 -3
  6. data/vendor/datasketches-cpp/CODE_OF_CONDUCT.md +1 -1
  7. data/vendor/datasketches-cpp/LICENSE +14 -0
  8. data/vendor/datasketches-cpp/README.md +67 -73
  9. data/vendor/datasketches-cpp/benchmarks/CMakeLists.txt +52 -0
  10. data/vendor/datasketches-cpp/benchmarks/benchmark_count_min_sketch.cpp +153 -0
  11. data/vendor/datasketches-cpp/benchmarks/benchmark_count_min_sketch_serialization.cpp +161 -0
  12. data/vendor/datasketches-cpp/common/CMakeLists.txt +1 -0
  13. data/vendor/datasketches-cpp/common/include/binomial_bounds.hpp +2 -2
  14. data/vendor/datasketches-cpp/common/include/fdlibm_log.hpp +101 -0
  15. data/vendor/datasketches-cpp/common/include/serde.hpp +6 -0
  16. data/vendor/datasketches-cpp/common/test/CMakeLists.txt +44 -2
  17. data/vendor/datasketches-cpp/common/test/binomial_bounds_test.cpp +279 -0
  18. data/vendor/datasketches-cpp/common/test/deserialize_hardening_test.cpp +188 -0
  19. data/vendor/datasketches-cpp/count/include/count_min.hpp +17 -4
  20. data/vendor/datasketches-cpp/count/include/count_min_impl.hpp +65 -83
  21. data/vendor/datasketches-cpp/count/test/count_min_test.cpp +63 -7
  22. data/vendor/datasketches-cpp/cpc/include/compression_data.hpp +2 -0
  23. data/vendor/datasketches-cpp/cpc/include/cpc_compressor_impl.hpp +16 -7
  24. data/vendor/datasketches-cpp/cpc/include/cpc_sketch.hpp +1 -0
  25. data/vendor/datasketches-cpp/cpc/include/cpc_sketch_impl.hpp +54 -36
  26. data/vendor/datasketches-cpp/cpc/include/cpc_union_impl.hpp +27 -27
  27. data/vendor/datasketches-cpp/cpc/include/cpc_util.hpp +7 -7
  28. data/vendor/datasketches-cpp/cpc/include/icon_estimator.hpp +5 -5
  29. data/vendor/datasketches-cpp/cpc/include/u32_table_impl.hpp +16 -16
  30. data/vendor/datasketches-cpp/cpc/test/cpc_sketch_test.cpp +35 -0
  31. data/vendor/datasketches-cpp/fi/include/frequent_items_sketch.hpp +28 -3
  32. data/vendor/datasketches-cpp/fi/include/frequent_items_sketch_impl.hpp +41 -25
  33. data/vendor/datasketches-cpp/fi/include/reverse_purge_hash_map.hpp +3 -1
  34. data/vendor/datasketches-cpp/fi/include/reverse_purge_hash_map_impl.hpp +10 -5
  35. data/vendor/datasketches-cpp/fi/test/frequent_items_sketch_serialize_for_java.cpp +24 -0
  36. data/vendor/datasketches-cpp/fi/test/frequent_items_sketch_test.cpp +116 -0
  37. data/vendor/datasketches-cpp/filters/include/bloom_filter.hpp +1 -1
  38. data/vendor/datasketches-cpp/filters/include/bloom_filter_impl.hpp +32 -12
  39. data/vendor/datasketches-cpp/filters/test/bloom_filter_test.cpp +28 -1
  40. data/vendor/datasketches-cpp/hll/include/CouponHashSet-internal.hpp +1 -2
  41. data/vendor/datasketches-cpp/hll/include/CouponList-internal.hpp +19 -5
  42. data/vendor/datasketches-cpp/hll/include/CubicInterpolation-internal.hpp +4 -4
  43. data/vendor/datasketches-cpp/hll/include/HarmonicNumbers-internal.hpp +2 -1
  44. data/vendor/datasketches-cpp/hll/include/Hll4Array-internal.hpp +4 -4
  45. data/vendor/datasketches-cpp/hll/include/Hll4Array.hpp +1 -1
  46. data/vendor/datasketches-cpp/hll/include/Hll6Array-internal.hpp +3 -3
  47. data/vendor/datasketches-cpp/hll/include/Hll6Array.hpp +1 -1
  48. data/vendor/datasketches-cpp/hll/include/Hll8Array-internal.hpp +6 -3
  49. data/vendor/datasketches-cpp/hll/include/Hll8Array.hpp +1 -1
  50. data/vendor/datasketches-cpp/hll/include/HllArray-internal.hpp +35 -29
  51. data/vendor/datasketches-cpp/hll/include/HllArray.hpp +1 -1
  52. data/vendor/datasketches-cpp/hll/include/HllSketch-internal.hpp +3 -5
  53. data/vendor/datasketches-cpp/hll/include/HllSketchImpl-internal.hpp +5 -11
  54. data/vendor/datasketches-cpp/hll/include/HllSketchImpl.hpp +2 -4
  55. data/vendor/datasketches-cpp/hll/include/HllSketchImplFactory.hpp +11 -10
  56. data/vendor/datasketches-cpp/hll/include/HllUnion-internal.hpp +17 -8
  57. data/vendor/datasketches-cpp/hll/include/HllUtil.hpp +12 -3
  58. data/vendor/datasketches-cpp/hll/include/coupon_iterator-internal.hpp +2 -2
  59. data/vendor/datasketches-cpp/hll/include/coupon_iterator.hpp +3 -0
  60. data/vendor/datasketches-cpp/hll/include/hll.hpp +11 -4
  61. data/vendor/datasketches-cpp/hll/include/hll.private.hpp +19 -0
  62. data/vendor/datasketches-cpp/hll/test/CMakeLists.txt +2 -0
  63. data/vendor/datasketches-cpp/hll/test/CouponListTest.cpp +64 -0
  64. data/vendor/datasketches-cpp/hll/test/HllFullSizeTest.cpp +137 -0
  65. data/vendor/datasketches-cpp/hll/test/HllKxqRebuildTest.cpp +150 -0
  66. data/vendor/datasketches-cpp/hll/test/HllSketchTest.cpp +3 -3
  67. data/vendor/datasketches-cpp/hll/test/HllUnionTest.cpp +71 -1
  68. data/vendor/datasketches-cpp/kll/include/kll_helper_impl.hpp +10 -10
  69. data/vendor/datasketches-cpp/kll/include/kll_sketch.hpp +10 -1
  70. data/vendor/datasketches-cpp/kll/include/kll_sketch_impl.hpp +33 -24
  71. data/vendor/datasketches-cpp/kll/test/kll_sketch_deserialize_from_java_test.cpp +24 -0
  72. data/vendor/datasketches-cpp/kll/test/kll_sketch_serialize_for_java.cpp +10 -0
  73. data/vendor/datasketches-cpp/quantiles/include/quantiles_sketch.hpp +11 -2
  74. data/vendor/datasketches-cpp/quantiles/include/quantiles_sketch_impl.hpp +24 -15
  75. data/vendor/datasketches-cpp/req/include/req_sketch.hpp +9 -0
  76. data/vendor/datasketches-cpp/req/include/req_sketch_impl.hpp +24 -15
  77. data/vendor/datasketches-cpp/req/test/req_sketch_deserialize_from_java_test.cpp +46 -0
  78. data/vendor/datasketches-cpp/req/test/req_sketch_serialize_for_java.cpp +20 -0
  79. data/vendor/datasketches-cpp/req/test/req_sketch_test.cpp +70 -0
  80. data/vendor/datasketches-cpp/sampling/include/ebpps_sample_impl.hpp +17 -8
  81. data/vendor/datasketches-cpp/sampling/include/ebpps_sketch.hpp +13 -0
  82. data/vendor/datasketches-cpp/sampling/include/var_opt_sketch.hpp +10 -1
  83. data/vendor/datasketches-cpp/sampling/include/var_opt_sketch_impl.hpp +6 -7
  84. data/vendor/datasketches-cpp/sampling/include/var_opt_union.hpp +5 -1
  85. data/vendor/datasketches-cpp/sampling/include/var_opt_union_impl.hpp +2 -1
  86. data/vendor/datasketches-cpp/sampling/test/ebpps_allocation_test.cpp +1 -1
  87. data/vendor/datasketches-cpp/sampling/test/ebpps_sketch_test.cpp +2 -2
  88. data/vendor/datasketches-cpp/sampling/test/var_opt_allocation_test.cpp +1 -1
  89. data/vendor/datasketches-cpp/sampling/test/var_opt_sketch_test.cpp +10 -4
  90. data/vendor/datasketches-cpp/sampling/test/var_opt_union_test.cpp +12 -0
  91. data/vendor/datasketches-cpp/tdigest/include/tdigest.hpp +38 -2
  92. data/vendor/datasketches-cpp/tdigest/include/tdigest_impl.hpp +168 -9
  93. data/vendor/datasketches-cpp/tdigest/test/CMakeLists.txt +1 -0
  94. data/vendor/datasketches-cpp/tdigest/test/tdigest_iterator_test.cpp +274 -0
  95. data/vendor/datasketches-cpp/tdigest/test/tdigest_test.cpp +275 -0
  96. data/vendor/datasketches-cpp/theta/include/compact_theta_sketch_parser.hpp +2 -0
  97. data/vendor/datasketches-cpp/theta/include/compact_theta_sketch_parser_impl.hpp +24 -3
  98. data/vendor/datasketches-cpp/theta/include/theta_constants.hpp +4 -2
  99. data/vendor/datasketches-cpp/theta/include/theta_helpers.hpp +32 -0
  100. data/vendor/datasketches-cpp/theta/include/theta_set_difference_base_impl.hpp +4 -2
  101. data/vendor/datasketches-cpp/theta/include/theta_sketch.hpp +22 -4
  102. data/vendor/datasketches-cpp/theta/include/theta_sketch_impl.hpp +60 -38
  103. data/vendor/datasketches-cpp/theta/include/theta_union_base_impl.hpp +2 -6
  104. data/vendor/datasketches-cpp/theta/include/theta_update_sketch_base_impl.hpp +2 -2
  105. data/vendor/datasketches-cpp/theta/test/bit_packing_test.cpp +50 -0
  106. data/vendor/datasketches-cpp/theta/test/theta_a_not_b_test.cpp +22 -0
  107. data/vendor/datasketches-cpp/theta/test/theta_sketch_test.cpp +315 -0
  108. data/vendor/datasketches-cpp/tools/rat-check.sh +68 -0
  109. data/vendor/datasketches-cpp/tuple/include/array_tuple_sketch.hpp +35 -4
  110. data/vendor/datasketches-cpp/tuple/include/array_tuple_sketch_impl.hpp +2 -2
  111. data/vendor/datasketches-cpp/tuple/include/tuple_sketch.hpp +41 -0
  112. data/vendor/datasketches-cpp/tuple/include/tuple_sketch_impl.hpp +5 -4
  113. data/vendor/datasketches-cpp/tuple/test/tuple_sketch_test.cpp +59 -0
  114. data/vendor/datasketches-cpp/version.cfg.in +1 -1
  115. metadata +12 -2
@@ -73,7 +73,7 @@ bool cpc_sketch_alloc<A>::is_empty() const {
73
73
 
74
74
  template<typename A>
75
75
  double cpc_sketch_alloc<A>::get_estimate() const {
76
- if (!was_merged) return get_hip_estimate();
76
+ if (!was_merged) { return get_hip_estimate(); }
77
77
  return get_icon_estimate();
78
78
  }
79
79
 
@@ -92,7 +92,7 @@ double cpc_sketch_alloc<A>::get_lower_bound(unsigned kappa) const {
92
92
  if (kappa < 1 || kappa > 3) {
93
93
  throw std::invalid_argument("kappa must be 1, 2 or 3");
94
94
  }
95
- if (!was_merged) return get_hip_confidence_lb<A>(*this, kappa);
95
+ if (!was_merged) { return get_hip_confidence_lb<A>(*this, kappa); }
96
96
  return get_icon_confidence_lb<A>(*this, kappa);
97
97
  }
98
98
 
@@ -101,13 +101,13 @@ double cpc_sketch_alloc<A>::get_upper_bound(unsigned kappa) const {
101
101
  if (kappa < 1 || kappa > 3) {
102
102
  throw std::invalid_argument("kappa must be 1, 2 or 3");
103
103
  }
104
- if (!was_merged) return get_hip_confidence_ub<A>(*this, kappa);
104
+ if (!was_merged) { return get_hip_confidence_ub<A>(*this, kappa); }
105
105
  return get_icon_confidence_ub<A>(*this, kappa);
106
106
  }
107
107
 
108
108
  template<typename A>
109
109
  void cpc_sketch_alloc<A>::update(const std::string& value) {
110
- if (value.empty()) return;
110
+ if (value.empty()) { return; }
111
111
  update(value.c_str(), value.length());
112
112
  }
113
113
 
@@ -173,15 +173,15 @@ void cpc_sketch_alloc<A>::update(float value) {
173
173
  }
174
174
 
175
175
  static inline uint32_t row_col_from_two_hashes(uint64_t hash0, uint64_t hash1, uint8_t lg_k) {
176
- if (lg_k > 26) throw std::logic_error("lg_k > 26");
176
+ if (lg_k > 26) { throw std::logic_error("lg_k > 26"); }
177
177
  const uint32_t k = 1 << lg_k;
178
178
  uint8_t col = count_leading_zeros_in_u64(hash1); // 0 <= col <= 64
179
- if (col > 63) col = 63; // clip so that 0 <= col <= 63
179
+ if (col > 63) { col = 63; } // clip so that 0 <= col <= 63
180
180
  const uint32_t row = hash0 & (k - 1);
181
181
  uint32_t row_col = (row << 6) | col;
182
182
  // To avoid the hash table's "empty" value, we change the row of the following pair.
183
183
  // This case is extremely unlikely, but we might as well handle it.
184
- if (row_col == UINT32_MAX) row_col ^= 1 << 6;
184
+ if (row_col == UINT32_MAX) { row_col ^= 1 << 6; }
185
185
  return row_col;
186
186
  }
187
187
 
@@ -195,7 +195,7 @@ void cpc_sketch_alloc<A>::update(const void* value, size_t size) {
195
195
  template<typename A>
196
196
  void cpc_sketch_alloc<A>::row_col_update(uint32_t row_col) {
197
197
  const uint8_t col = row_col & 63;
198
- if (col < first_interesting_column) return; // important speed optimization
198
+ if (col < first_interesting_column) { return; } // important speed optimization
199
199
  // window size is 0 until sketch is promoted from sparse to windowed
200
200
  if (sliding_window.size() == 0) {
201
201
  update_sparse(row_col);
@@ -208,26 +208,26 @@ template<typename A>
208
208
  void cpc_sketch_alloc<A>::update_sparse(uint32_t row_col) {
209
209
  const uint32_t k = 1 << lg_k;
210
210
  const uint64_t c32pre = static_cast<uint64_t>(num_coupons) << 5;
211
- if (c32pre >= 3 * k) throw std::logic_error("c32pre >= 3 * k"); // C < 3K/32, in other words flavor == SPARSE
211
+ if (c32pre >= 3 * k) { throw std::logic_error("c32pre >= 3 * k"); } // C < 3K/32, in other words flavor == SPARSE
212
212
  bool is_novel = surprising_value_table.maybe_insert(row_col);
213
213
  if (is_novel) {
214
214
  num_coupons++;
215
215
  update_hip(row_col);
216
216
  const uint64_t c32post = static_cast<uint64_t>(num_coupons) << 5;
217
- if (c32post >= 3 * k) promote_sparse_to_windowed(); // C >= 3K/32
217
+ if (c32post >= 3 * k) { promote_sparse_to_windowed(); } // C >= 3K/32
218
218
  }
219
219
  }
220
220
 
221
221
  // the flavor is HYBRID, PINNED, or SLIDING
222
222
  template<typename A>
223
223
  void cpc_sketch_alloc<A>::update_windowed(uint32_t row_col) {
224
- if (window_offset > 56) throw std::logic_error("wrong window offset");
224
+ if (window_offset > 56) { throw std::logic_error("wrong window offset"); }
225
225
  const uint32_t k = 1 << lg_k;
226
226
  const uint64_t c32pre = static_cast<uint64_t>(num_coupons) << 5;
227
- if (c32pre < 3 * k) throw std::logic_error("c32pre < 3 * k"); // C < 3K/32, in other words flavor >= HYBRID
227
+ if (c32pre < 3 * k) { throw std::logic_error("c32pre < 3 * k"); } // C < 3K/32, in other words flavor >= HYBRID
228
228
  const uint64_t c8pre = static_cast<uint64_t>(num_coupons) << 3;
229
229
  const uint64_t w8pre = static_cast<uint64_t>(window_offset) << 3;
230
- if (c8pre >= (27 + w8pre) * k) throw std::logic_error("c8pre is wrong"); // C < (K * 27/8) + (K * window_offset)
230
+ if (c8pre >= (27 + w8pre) * k) { throw std::logic_error("c8pre is wrong"); } // C < (K * 27/8) + (K * window_offset)
231
231
 
232
232
  bool is_novel = false;
233
233
  const uint8_t col = row_col & 63;
@@ -235,7 +235,7 @@ void cpc_sketch_alloc<A>::update_windowed(uint32_t row_col) {
235
235
  if (col < window_offset) { // track the surprising 0's "before" the window
236
236
  is_novel = surprising_value_table.maybe_delete(row_col); // inverted logic
237
237
  } else if (col < window_offset + 8) { // track the 8 bits inside the window
238
- if (col < window_offset) throw std::logic_error("col < window_offset");
238
+ if (col < window_offset) { throw std::logic_error("col < window_offset"); }
239
239
  const uint32_t row = row_col >> 6;
240
240
  const uint8_t old_bits = sliding_window[row];
241
241
  const uint8_t new_bits = old_bits | (1 << (col - window_offset));
@@ -244,7 +244,7 @@ void cpc_sketch_alloc<A>::update_windowed(uint32_t row_col) {
244
244
  is_novel = true;
245
245
  }
246
246
  } else { // track the surprising 1's "after" the window
247
- if (col < window_offset + 8) throw std::logic_error("col < window_offset + 8");
247
+ if (col < window_offset + 8) { throw std::logic_error("col < window_offset + 8"); }
248
248
  is_novel = surprising_value_table.maybe_insert(row_col); // normal logic
249
249
  }
250
250
 
@@ -254,9 +254,9 @@ void cpc_sketch_alloc<A>::update_windowed(uint32_t row_col) {
254
254
  const uint64_t c8post = static_cast<uint64_t>(num_coupons) << 3;
255
255
  if (c8post >= (27 + w8pre) * k) {
256
256
  move_window();
257
- if (window_offset < 1 || window_offset > 56) throw std::logic_error("wrong window offset");
257
+ if (window_offset < 1 || window_offset > 56) { throw std::logic_error("wrong window offset"); }
258
258
  const uint64_t w8post = static_cast<uint64_t>(window_offset) << 3;
259
- if (c8post >= (27 + w8post) * k) throw std::logic_error("c8pre is wrong"); // C < (K * 27/8) + (K * window_offset)
259
+ if (c8post >= (27 + w8post) * k) { throw std::logic_error("c8pre is wrong"); } // C < (K * 27/8) + (K * window_offset)
260
260
  }
261
261
  }
262
262
  }
@@ -276,7 +276,7 @@ template<typename A>
276
276
  void cpc_sketch_alloc<A>::promote_sparse_to_windowed() {
277
277
  const uint32_t k = 1 << lg_k;
278
278
  const uint64_t c32 = static_cast<uint64_t>(num_coupons) << 5;
279
- if (!(c32 == 3 * k || (lg_k == 4 && c32 > 3 * k))) throw std::logic_error("wrong c32");
279
+ if (!(c32 == 3 * k || (lg_k == 4 && c32 > 3 * k))) { throw std::logic_error("wrong c32"); }
280
280
 
281
281
  sliding_window.resize(k, 0); // zero the memory (because we will be OR'ing into it)
282
282
 
@@ -285,7 +285,7 @@ void cpc_sketch_alloc<A>::promote_sparse_to_windowed() {
285
285
  const uint32_t* old_slots = surprising_value_table.get_slots();
286
286
  const uint32_t old_num_slots = 1 << surprising_value_table.get_lg_size();
287
287
 
288
- if (window_offset != 0) throw std::logic_error("window_offset != 0");
288
+ if (window_offset != 0) { throw std::logic_error("window_offset != 0"); }
289
289
 
290
290
  for (uint32_t i = 0; i < old_num_slots; i++) {
291
291
  const uint32_t row_col = old_slots[i];
@@ -297,7 +297,7 @@ void cpc_sketch_alloc<A>::promote_sparse_to_windowed() {
297
297
  } else {
298
298
  // cannot use u32_table::must_insert(), because it doesn't provide for growth
299
299
  const bool is_novel = new_table.maybe_insert(row_col);
300
- if (!is_novel) throw std::logic_error("is_novel != true");
300
+ if (!is_novel) { throw std::logic_error("is_novel != true"); }
301
301
  }
302
302
  }
303
303
  }
@@ -308,17 +308,17 @@ void cpc_sketch_alloc<A>::promote_sparse_to_windowed() {
308
308
  template<typename A>
309
309
  void cpc_sketch_alloc<A>::move_window() {
310
310
  const uint8_t new_offset = window_offset + 1;
311
- if (new_offset > 56) throw std::logic_error("new_offset > 56");
312
- if (new_offset != determine_correct_offset(lg_k, num_coupons)) throw std::logic_error("new_offset is wrong");
311
+ if (new_offset > 56) { throw std::logic_error("new_offset > 56"); }
312
+ if (new_offset != determine_correct_offset(lg_k, num_coupons)) { throw std::logic_error("new_offset is wrong"); }
313
313
 
314
- if (sliding_window.size() == 0) throw std::logic_error("no sliding window");
314
+ if (sliding_window.size() == 0) { throw std::logic_error("no sliding window"); }
315
315
  const uint32_t k = 1 << lg_k;
316
316
 
317
317
  // Construct the full-sized bit matrix that corresponds to the sketch
318
318
  vector_u64 bit_matrix = build_bit_matrix();
319
319
 
320
320
  // refresh the KXP register on every 8th window shift.
321
- if ((new_offset & 0x7) == 0) refresh_kxp(bit_matrix.data());
321
+ if ((new_offset & 0x7) == 0) { refresh_kxp(bit_matrix.data()); }
322
322
 
323
323
  surprising_value_table.clear(); // the new number of surprises will be about the same
324
324
 
@@ -339,14 +339,14 @@ void cpc_sketch_alloc<A>::move_window() {
339
339
  pattern = pattern ^ (static_cast<uint64_t>(1) << col); // erase the 1
340
340
  const uint32_t row_col = (i << 6) | col;
341
341
  const bool is_novel = surprising_value_table.maybe_insert(row_col);
342
- if (!is_novel) throw std::logic_error("is_novel != true");
342
+ if (!is_novel) { throw std::logic_error("is_novel != true"); }
343
343
  }
344
344
  }
345
345
 
346
346
  window_offset = new_offset;
347
347
 
348
348
  first_interesting_column = count_trailing_zeros_in_u64(all_surprises_ored);
349
- if (first_interesting_column > new_offset) first_interesting_column = new_offset; // corner case
349
+ if (first_interesting_column > new_offset) { first_interesting_column = new_offset; } // corner case
350
350
  }
351
351
 
352
352
  // The KXP register is a double with roughly 50 bits of precision, but
@@ -438,7 +438,7 @@ void cpc_sketch_alloc<A>::serialize(std::ostream& os) const {
438
438
  write(os, compressed.table_num_entries);
439
439
  // HIP values can be in two different places in the sequence of fields
440
440
  // this is the first HIP decision point
441
- if (has_hip) write_hip(os);
441
+ if (has_hip) { write_hip(os); }
442
442
  }
443
443
  if (has_table) {
444
444
  write(os, compressed.table_data_words);
@@ -447,7 +447,7 @@ void cpc_sketch_alloc<A>::serialize(std::ostream& os) const {
447
447
  write(os, compressed.window_data_words);
448
448
  }
449
449
  // this is the second HIP decision point
450
- if (has_hip && !(has_table && has_window)) write_hip(os);
450
+ if (has_hip && !(has_table && has_window)) { write_hip(os); }
451
451
  if (has_window) {
452
452
  write(os, compressed.window_data.data(), compressed.window_data_words * sizeof(uint32_t));
453
453
  }
@@ -494,7 +494,7 @@ auto cpc_sketch_alloc<A>::serialize(unsigned header_size_bytes) const -> vector_
494
494
  ptr += copy_to_mem(compressed.table_num_entries, ptr);
495
495
  // HIP values can be in two different places in the sequence of fields
496
496
  // this is the first HIP decision point
497
- if (has_hip) ptr += copy_hip_to_mem(ptr);
497
+ if (has_hip) { ptr += copy_hip_to_mem(ptr); }
498
498
  }
499
499
  if (has_table) {
500
500
  ptr += copy_to_mem(compressed.table_data_words, ptr);
@@ -503,7 +503,7 @@ auto cpc_sketch_alloc<A>::serialize(unsigned header_size_bytes) const -> vector_
503
503
  ptr += copy_to_mem(compressed.window_data_words, ptr);
504
504
  }
505
505
  // this is the second HIP decision point
506
- if (has_hip && !(has_table && has_window)) ptr += copy_hip_to_mem(ptr);
506
+ if (has_hip && !(has_table && has_window)) { ptr += copy_hip_to_mem(ptr); }
507
507
  if (has_window) {
508
508
  ptr += copy_to_mem(compressed.window_data.data(), ptr, compressed.window_data_words * sizeof(uint32_t));
509
509
  }
@@ -511,7 +511,7 @@ auto cpc_sketch_alloc<A>::serialize(unsigned header_size_bytes) const -> vector_
511
511
  ptr += copy_to_mem(compressed.table_data.data(), ptr, compressed.table_data_words * sizeof(uint32_t));
512
512
  }
513
513
  }
514
- if (ptr != bytes.data() + size) throw std::logic_error("serialized size mismatch");
514
+ if (ptr != bytes.data() + size) { throw std::logic_error("serialized size mismatch"); }
515
515
  return bytes;
516
516
  }
517
517
 
@@ -527,6 +527,7 @@ cpc_sketch_alloc<A> cpc_sketch_alloc<A>::deserialize(std::istream& is, uint64_t
527
527
  const bool has_hip = flags_byte & (1 << flags::HAS_HIP);
528
528
  const bool has_table = flags_byte & (1 << flags::HAS_TABLE);
529
529
  const bool has_window = flags_byte & (1 << flags::HAS_WINDOW);
530
+ check_lg_k(lg_k);
530
531
  compressed_state<A> compressed(allocator);
531
532
  compressed.table_data_words = 0;
532
533
  compressed.table_num_entries = 0;
@@ -561,7 +562,7 @@ cpc_sketch_alloc<A> cpc_sketch_alloc<A>::deserialize(std::istream& is, uint64_t
561
562
  compressed.table_data.resize(compressed.table_data_words);
562
563
  read(is, compressed.table_data.data(), compressed.table_data_words * sizeof(uint32_t));
563
564
  }
564
- if (!has_window) compressed.table_num_entries = num_coupons;
565
+ if (!has_window) { compressed.table_num_entries = num_coupons; }
565
566
  }
566
567
 
567
568
  uint8_t expected_preamble_ints = get_preamble_ints(num_coupons, has_hip, has_table, has_window);
@@ -581,10 +582,12 @@ cpc_sketch_alloc<A> cpc_sketch_alloc<A>::deserialize(std::istream& is, uint64_t
581
582
  throw std::invalid_argument("Incompatible seed hashes: " + std::to_string(seed_hash) + ", "
582
583
  + std::to_string(compute_seed_hash(seed)));
583
584
  }
585
+ check_num_coupons(lg_k, num_coupons, compressed.table_num_entries);
584
586
  uncompressed_state<A> uncompressed(allocator);
585
587
  get_compressor<A>().uncompress(compressed, uncompressed, lg_k, num_coupons);
586
- if (!is.good())
587
- throw std::runtime_error("error reading from std::istream");
588
+ if (!is.good()) {
589
+ throw std::runtime_error("error reading from std::istream");
590
+ }
588
591
  return cpc_sketch_alloc(lg_k, num_coupons, first_interesting_column, std::move(uncompressed.table),
589
592
  std::move(uncompressed.window), has_hip, kxp, hip_est_accum, seed);
590
593
  }
@@ -611,6 +614,7 @@ cpc_sketch_alloc<A> cpc_sketch_alloc<A>::deserialize(const void* bytes, size_t s
611
614
  const bool has_hip = flags_byte & (1 << flags::HAS_HIP);
612
615
  const bool has_table = flags_byte & (1 << flags::HAS_TABLE);
613
616
  const bool has_window = flags_byte & (1 << flags::HAS_WINDOW);
617
+ check_lg_k(lg_k);
614
618
  ensure_minimum_memory(size, preamble_ints << 2);
615
619
  compressed_state<A> compressed(allocator);
616
620
  compressed.table_data_words = 0;
@@ -645,13 +649,13 @@ cpc_sketch_alloc<A> cpc_sketch_alloc<A>::deserialize(const void* bytes, size_t s
645
649
  ptr += copy_from_mem(ptr, hip_est_accum);
646
650
  }
647
651
  if (has_window) {
648
- compressed.window_data.resize(compressed.window_data_words);
649
652
  check_memory_size(ptr - base + (compressed.window_data_words * sizeof(uint32_t)), size);
653
+ compressed.window_data.resize(compressed.window_data_words);
650
654
  ptr += copy_from_mem(ptr, compressed.window_data.data(), compressed.window_data_words * sizeof(uint32_t));
651
655
  }
652
656
  if (has_table) {
653
- compressed.table_data.resize(compressed.table_data_words);
654
657
  check_memory_size(ptr - base + (compressed.table_data_words * sizeof(uint32_t)), size);
658
+ compressed.table_data.resize(compressed.table_data_words);
655
659
  ptr += copy_from_mem(ptr, compressed.table_data.data(), compressed.table_data_words * sizeof(uint32_t));
656
660
  }
657
661
  if (!has_window) compressed.table_num_entries = num_coupons;
@@ -675,6 +679,7 @@ cpc_sketch_alloc<A> cpc_sketch_alloc<A>::deserialize(const void* bytes, size_t s
675
679
  throw std::invalid_argument("Incompatible seed hashes: " + std::to_string(seed_hash) + ", "
676
680
  + std::to_string(compute_seed_hash(seed)));
677
681
  }
682
+ check_num_coupons(lg_k, num_coupons, compressed.table_num_entries);
678
683
  uncompressed_state<A> uncompressed(allocator);
679
684
  get_compressor<A>().uncompress(compressed, uncompressed, lg_k, num_coupons);
680
685
  return cpc_sketch_alloc(lg_k, num_coupons, first_interesting_column, std::move(uncompressed.table),
@@ -719,6 +724,19 @@ size_t cpc_sketch_alloc<A>::get_max_serialized_size_bytes(uint8_t lg_k) {
719
724
  return (int) (CPC_EMPIRICAL_MAX_SIZE_FACTOR * k) + CPC_MAX_PREAMBLE_SIZE_BYTES;
720
725
  }
721
726
 
727
+ template<typename A>
728
+ void cpc_sketch_alloc<A>::check_num_coupons(uint8_t lg_k, uint32_t num_coupons, uint32_t num_pairs) {
729
+ // at most one coupon per bit of the k x 64 bit matrix, and surprising values are a subset of coupons
730
+ if (num_coupons > (static_cast<uint64_t>(1) << lg_k) * 64) {
731
+ throw std::invalid_argument("Possible corruption: num_coupons " + std::to_string(num_coupons)
732
+ + " exceeds the capacity for lg_k " + std::to_string(lg_k));
733
+ }
734
+ if (num_pairs > num_coupons) {
735
+ throw std::invalid_argument("Possible corruption: table entries " + std::to_string(num_pairs)
736
+ + " exceed num_coupons " + std::to_string(num_coupons));
737
+ }
738
+ }
739
+
722
740
  template<typename A>
723
741
  void cpc_sketch_alloc<A>::check_lg_k(uint8_t lg_k) {
724
742
  if (lg_k < cpc_constants::MIN_LG_K || lg_k > cpc_constants::MAX_LG_K) {
@@ -109,15 +109,15 @@ void cpc_union_alloc<A>::internal_update(S&& sketch) {
109
109
  + std::to_string(seed_hash_sketch));
110
110
  }
111
111
  const auto src_flavor = sketch.determine_flavor();
112
- if (cpc_sketch_alloc<A>::flavor::EMPTY == src_flavor) return;
112
+ if (cpc_sketch_alloc<A>::flavor::EMPTY == src_flavor) { return; }
113
113
 
114
- if (sketch.get_lg_k() < lg_k) reduce_k(sketch.get_lg_k());
115
- if (sketch.get_lg_k() < lg_k) throw std::logic_error("sketch lg_k < union lg_k");
114
+ if (sketch.get_lg_k() < lg_k) { reduce_k(sketch.get_lg_k()); }
115
+ if (sketch.get_lg_k() < lg_k) { throw std::logic_error("sketch lg_k < union lg_k"); }
116
116
 
117
- if (accumulator == nullptr && bit_matrix.size() == 0) throw std::logic_error("both accumulator and bit matrix are absent");
117
+ if (accumulator == nullptr && bit_matrix.size() == 0) { throw std::logic_error("both accumulator and bit matrix are absent"); }
118
118
 
119
119
  if (cpc_sketch_alloc<A>::flavor::SPARSE == src_flavor && accumulator != nullptr) { // Case A
120
- if (bit_matrix.size() > 0) throw std::logic_error("union bit_matrix is not expected");
120
+ if (bit_matrix.size() > 0) { throw std::logic_error("union bit_matrix is not expected"); }
121
121
  const auto initial_dest_flavor = accumulator->determine_flavor();
122
122
  if (cpc_sketch_alloc<A>::flavor::EMPTY != initial_dest_flavor &&
123
123
  cpc_sketch_alloc<A>::flavor::SPARSE != initial_dest_flavor) throw std::logic_error("wrong flavor");
@@ -138,24 +138,24 @@ void cpc_union_alloc<A>::internal_update(S&& sketch) {
138
138
  }
139
139
 
140
140
  if (cpc_sketch_alloc<A>::flavor::SPARSE == src_flavor && bit_matrix.size() > 0) { // Case B
141
- if (accumulator != nullptr) throw std::logic_error("union accumulator != null");
141
+ if (accumulator != nullptr) { throw std::logic_error("union accumulator != null"); }
142
142
  or_table_into_matrix(sketch.surprising_value_table);
143
143
  return;
144
144
  }
145
145
 
146
146
  if (cpc_sketch_alloc<A>::flavor::HYBRID != src_flavor && cpc_sketch_alloc<A>::flavor::PINNED != src_flavor
147
- && cpc_sketch_alloc<A>::flavor::SLIDING != src_flavor) throw std::logic_error("wrong flavor");
147
+ && cpc_sketch_alloc<A>::flavor::SLIDING != src_flavor) { throw std::logic_error("wrong flavor"); }
148
148
 
149
149
  // source is past SPARSE mode, so make sure that dest is a bit matrix
150
150
  if (accumulator != nullptr) {
151
- if (bit_matrix.size() > 0) throw std::logic_error("union bit matrix is not expected");
151
+ if (bit_matrix.size() > 0) { throw std::logic_error("union bit matrix is not expected"); }
152
152
  const auto dst_flavor = accumulator->determine_flavor();
153
153
  if (cpc_sketch_alloc<A>::flavor::EMPTY != dst_flavor && cpc_sketch_alloc<A>::flavor::SPARSE != dst_flavor) {
154
154
  throw std::logic_error("wrong flavor");
155
155
  }
156
156
  switch_to_bit_matrix();
157
157
  }
158
- if (bit_matrix.size() == 0) throw std::logic_error("union bit_matrix is expected");
158
+ if (bit_matrix.size() == 0) { throw std::logic_error("union bit_matrix is expected"); }
159
159
 
160
160
  if (cpc_sketch_alloc<A>::flavor::HYBRID == src_flavor || cpc_sketch_alloc<A>::flavor::PINNED == src_flavor) { // Case C
161
161
  or_window_into_matrix(sketch.sliding_window, sketch.window_offset, sketch.get_lg_k());
@@ -165,7 +165,7 @@ void cpc_union_alloc<A>::internal_update(S&& sketch) {
165
165
 
166
166
  // SLIDING mode involves inverted logic, so we can't just walk the source sketch.
167
167
  // Instead, we convert it to a bitMatrix that can be OR'ed into the destination.
168
- if (cpc_sketch_alloc<A>::flavor::SLIDING != src_flavor) throw std::logic_error("wrong flavor"); // Case D
168
+ if (cpc_sketch_alloc<A>::flavor::SLIDING != src_flavor) { throw std::logic_error("wrong flavor"); } // Case D
169
169
  vector_u64 src_matrix = sketch.build_bit_matrix();
170
170
  or_matrix_into_matrix(src_matrix, sketch.get_lg_k());
171
171
  }
@@ -173,20 +173,20 @@ void cpc_union_alloc<A>::internal_update(S&& sketch) {
173
173
  template<typename A>
174
174
  cpc_sketch_alloc<A> cpc_union_alloc<A>::get_result() const {
175
175
  if (accumulator != nullptr) {
176
- if (bit_matrix.size() > 0) throw std::logic_error("bit_matrix is not expected");
176
+ if (bit_matrix.size() > 0) { throw std::logic_error("bit_matrix is not expected"); }
177
177
  return get_result_from_accumulator();
178
178
  }
179
- if (bit_matrix.size() == 0) throw std::logic_error("bit_matrix is expected");
179
+ if (bit_matrix.size() == 0) { throw std::logic_error("bit_matrix is expected"); }
180
180
  return get_result_from_bit_matrix();
181
181
  }
182
182
 
183
183
  template<typename A>
184
184
  cpc_sketch_alloc<A> cpc_union_alloc<A>::get_result_from_accumulator() const {
185
- if (lg_k != accumulator->get_lg_k()) throw std::logic_error("lg_k != accumulator->lg_k");
185
+ if (lg_k != accumulator->get_lg_k()) { throw std::logic_error("lg_k != accumulator->lg_k"); }
186
186
  if (accumulator->get_num_coupons() == 0) {
187
187
  return cpc_sketch_alloc<A>(lg_k, seed, accumulator->get_allocator());
188
188
  }
189
- if (accumulator->determine_flavor() != cpc_sketch_alloc<A>::flavor::SPARSE) throw std::logic_error("wrong flavor");
189
+ if (accumulator->determine_flavor() != cpc_sketch_alloc<A>::flavor::SPARSE) { throw std::logic_error("wrong flavor"); }
190
190
  cpc_sketch_alloc<A> copy(*accumulator);
191
191
  copy.was_merged = true;
192
192
  return copy;
@@ -199,7 +199,7 @@ cpc_sketch_alloc<A> cpc_union_alloc<A>::get_result_from_bit_matrix() const {
199
199
 
200
200
  const auto flavor = cpc_sketch_alloc<A>::determine_flavor(lg_k, num_coupons);
201
201
  if (flavor != cpc_sketch_alloc<A>::flavor::HYBRID && flavor != cpc_sketch_alloc<A>::flavor::PINNED
202
- && flavor != cpc_sketch_alloc<A>::flavor::SLIDING) throw std::logic_error("wrong flavor");
202
+ && flavor != cpc_sketch_alloc<A>::flavor::SLIDING) { throw std::logic_error("wrong flavor"); }
203
203
 
204
204
  const uint8_t offset = cpc_sketch_alloc<A>::determine_correct_offset(lg_k, num_coupons);
205
205
 
@@ -208,7 +208,7 @@ cpc_sketch_alloc<A> cpc_union_alloc<A>::get_result_from_bit_matrix() const {
208
208
 
209
209
  // dynamically growing caused snowplow effect
210
210
  uint8_t table_lg_size = lg_k - 4; // K/16; in some cases this will end up being oversized
211
- if (table_lg_size < 2) table_lg_size = 2;
211
+ if (table_lg_size < 2) { table_lg_size = 2; }
212
212
  u32_table<A> table(table_lg_size, 6 + lg_k, bit_matrix.get_allocator());
213
213
 
214
214
  // the following should work even when the offset is zero
@@ -229,14 +229,14 @@ cpc_sketch_alloc<A> cpc_union_alloc<A>::get_result_from_bit_matrix() const {
229
229
  pattern = pattern ^ (static_cast<uint64_t>(1) << col); // erase the 1
230
230
  const uint32_t row_col = (i << 6) | col;
231
231
  bool is_novel = table.maybe_insert(row_col);
232
- if (!is_novel) throw std::logic_error("is_novel != true");
232
+ if (!is_novel) { throw std::logic_error("is_novel != true"); }
233
233
  }
234
234
  }
235
235
 
236
236
  // at this point we could shrink an oversized hash table, but the relative waste isn't very big
237
237
 
238
238
  uint8_t first_interesting_column = count_trailing_zeros_in_u64(all_surprises_ored);
239
- if (first_interesting_column > offset) first_interesting_column = offset; // corner case
239
+ if (first_interesting_column > offset) { first_interesting_column = offset; } // corner case
240
240
 
241
241
  // HIP-related fields will contain zeros, and that is okay
242
242
  return cpc_sketch_alloc<A>(lg_k, num_coupons, first_interesting_column, std::move(table), std::move(sliding_window), false, 0, 0, seed);
@@ -260,9 +260,9 @@ void cpc_union_alloc<A>::walk_table_updating_sketch(const u32_table<A>& table) {
260
260
  // Using a golden ratio stride fixes the snowplow effect.
261
261
  const double golden = 0.6180339887498949025;
262
262
  uint32_t stride = static_cast<uint32_t>(golden * static_cast<double>(num_slots));
263
- if (stride < 2) throw std::logic_error("stride < 2");
264
- if (stride == ((stride >> 1) << 1)) stride += 1; // force the stride to be odd
265
- if (stride < 3 || stride >= num_slots) throw std::out_of_range("stride out of range");
263
+ if (stride < 2) { throw std::logic_error("stride < 2"); }
264
+ if (stride == ((stride >> 1) << 1)) { stride += 1; } // force the stride to be odd
265
+ if (stride < 3 || stride >= num_slots) { throw std::out_of_range("stride out of range"); }
266
266
 
267
267
  for (uint32_t i = 0, j = 0; i < num_slots; i++, j += stride) {
268
268
  j &= num_slots - 1;
@@ -290,7 +290,7 @@ void cpc_union_alloc<A>::or_table_into_matrix(const u32_table<A>& table) {
290
290
 
291
291
  template<typename A>
292
292
  void cpc_union_alloc<A>::or_window_into_matrix(const vector_bytes& sliding_window, uint8_t offset, uint8_t src_lg_k) {
293
- if (lg_k > src_lg_k) throw std::logic_error("dst LgK > src LgK");
293
+ if (lg_k > src_lg_k) { throw std::logic_error("dst LgK > src LgK"); }
294
294
  const uint64_t dst_mask = (1 << lg_k) - 1; // downsamples when dst lgK < src LgK
295
295
  const uint32_t src_k = 1 << src_lg_k;
296
296
  for (uint32_t src_row = 0; src_row < src_k; src_row++) {
@@ -300,7 +300,7 @@ void cpc_union_alloc<A>::or_window_into_matrix(const vector_bytes& sliding_windo
300
300
 
301
301
  template<typename A>
302
302
  void cpc_union_alloc<A>::or_matrix_into_matrix(const vector_u64& src_matrix, uint8_t src_lg_k) {
303
- if (lg_k > src_lg_k) throw std::logic_error("dst LgK > src LgK");
303
+ if (lg_k > src_lg_k) { throw std::logic_error("dst LgK > src LgK"); }
304
304
  const uint64_t dst_mask = (1 << lg_k) - 1; // downsamples when dst lgK < src LgK
305
305
  const uint32_t src_k = 1 << src_lg_k;
306
306
  for (uint32_t src_row = 0; src_row < src_k; src_row++) {
@@ -310,11 +310,11 @@ void cpc_union_alloc<A>::or_matrix_into_matrix(const vector_u64& src_matrix, uin
310
310
 
311
311
  template<typename A>
312
312
  void cpc_union_alloc<A>::reduce_k(uint8_t new_lg_k) {
313
- if (new_lg_k >= lg_k) throw std::logic_error("new LgK >= union lgK");
314
- if (accumulator == nullptr && bit_matrix.size() == 0) throw std::logic_error("both accumulator and bit_matrix are absent");
313
+ if (new_lg_k >= lg_k) { throw std::logic_error("new LgK >= union lgK"); }
314
+ if (accumulator == nullptr && bit_matrix.size() == 0) { throw std::logic_error("both accumulator and bit_matrix are absent"); }
315
315
 
316
316
  if (bit_matrix.size() > 0) { // downsample the unioner's bit matrix
317
- if (accumulator != nullptr) throw std::logic_error("accumulator is not null");
317
+ if (accumulator != nullptr) { throw std::logic_error("accumulator is not null"); }
318
318
  vector_u64 old_matrix = std::move(bit_matrix);
319
319
  const uint8_t old_lg_k = lg_k;
320
320
  const uint32_t new_k = 1 << new_lg_k;
@@ -325,7 +325,7 @@ void cpc_union_alloc<A>::reduce_k(uint8_t new_lg_k) {
325
325
  }
326
326
 
327
327
  if (accumulator != nullptr) { // downsample the unioner's sketch
328
- if (bit_matrix.size() > 0) throw std::logic_error("bit_matrix is not expected");
328
+ if (bit_matrix.size() > 0) { throw std::logic_error("bit_matrix is not expected"); }
329
329
  if (!accumulator->is_empty()) {
330
330
  cpc_sketch_alloc<A> old_accumulator(*accumulator);
331
331
  *accumulator = cpc_sketch_alloc<A>(new_lg_k, seed, old_accumulator.get_allocator());
@@ -25,19 +25,19 @@
25
25
  namespace datasketches {
26
26
 
27
27
  static inline uint64_t divide_longs_rounding_up(uint64_t x, uint64_t y) {
28
- if (y == 0) throw std::invalid_argument("divide_longs_rounding_up: bad argument");
28
+ if (y == 0) { throw std::invalid_argument("divide_longs_rounding_up: bad argument"); }
29
29
  const uint64_t quotient = x / y;
30
- if (quotient * y == x) return (quotient);
31
- else return quotient + 1;
30
+ if (quotient * y == x) { return (quotient); }
31
+ else { return quotient + 1; }
32
32
  }
33
33
 
34
34
  static inline uint8_t floor_log2_of_long(uint64_t x) {
35
- if (x < 1) throw std::invalid_argument("floor_log2_of_long: bad argument");
35
+ if (x < 1) { throw std::invalid_argument("floor_log2_of_long: bad argument"); }
36
36
  uint8_t p = 0;
37
37
  uint64_t y = 1;
38
38
  while (true) {
39
- if (y == x) return p;
40
- if (y > x) return p - 1;
39
+ if (y == x) { return p; }
40
+ if (y > x) { return p - 1; }
41
41
  p += 1;
42
42
  y <<= 1;
43
43
  }
@@ -98,7 +98,7 @@ static inline uint32_t warren_count_bits_set_in_matrix(const uint64_t* array, ui
98
98
  }
99
99
 
100
100
  static inline uint32_t count_bits_set_in_matrix(const uint64_t* a, uint32_t length) {
101
- if ((length & 0x7) != 0) throw std::invalid_argument("the length of the array must be a multiple of 8");
101
+ if ((length & 0x7) != 0) { throw std::invalid_argument("the length of the array must be a multiple of 8"); }
102
102
  uint32_t total = 0;
103
103
  uint64_t ones, twos, twos_a, twos_b, fours, fours_a, fours_b, eights;
104
104
  fours = twos = ones = 0;
@@ -246,14 +246,14 @@ static inline double icon_exponential_approximation(double k, double c) {
246
246
  }
247
247
 
248
248
  static inline double compute_icon_estimate(uint8_t lg_k, uint32_t c) {
249
- if (lg_k < ICON_MIN_LOG_K || lg_k > ICON_MAX_LOG_K) throw std::out_of_range("lg_k out of range");
250
- if (c < 2) return ((c == 0) ? 0.0 : 1.0);
249
+ if (lg_k < ICON_MIN_LOG_K || lg_k > ICON_MAX_LOG_K) { throw std::out_of_range("lg_k out of range"); }
250
+ if (c < 2) { return ((c == 0) ? 0.0 : 1.0); }
251
251
  const uint32_t k = 1 << lg_k;
252
252
  const double double_k = static_cast<double>(k);
253
253
  const double double_c = static_cast<double>(c);
254
254
  // Differing thresholds ensure that the approximated estimator is monotonically increasing.
255
255
  const double threshold_factor = ((lg_k < 14) ? 5.7 : 5.6);
256
- if (double_c > (threshold_factor * double_k)) return icon_exponential_approximation(double_k, double_c);
256
+ if (double_c > (threshold_factor * double_k)) { return icon_exponential_approximation(double_k, double_c); }
257
257
  const double factor = evaluate_polynomial(
258
258
  ICON_POLYNOMIAL_COEFFICIENTS,
259
259
  ICON_POLYNOMIAL_NUM_COEFFICIENTS * (lg_k - ICON_MIN_LOG_K),
@@ -265,8 +265,8 @@ static inline double compute_icon_estimate(uint8_t lg_k, uint32_t c) {
265
265
  // The somewhat arbitrary constant 66.774757 is baked into the table ICON_POLYNOMIAL_COEFFICIENTS
266
266
  const double term = 1.0 + (ratio * ratio * ratio / 66.774757);
267
267
  const double result = double_c * factor * term;
268
- if (result >= double_c) return result;
269
- else return double_c;
268
+ if (result >= double_c) { return result; }
269
+ else { return double_c; }
270
270
  }
271
271
 
272
272
  } /* namespace datasketches */
@@ -43,8 +43,8 @@ num_valid_bits(num_valid_bits),
43
43
  num_items(0),
44
44
  slots(1ULL << lg_size, UINT32_MAX, allocator)
45
45
  {
46
- if (lg_size < 2) throw std::invalid_argument("lg_size must be >= 2");
47
- if (num_valid_bits < 1 || num_valid_bits > 32) throw std::invalid_argument("num_valid_bits must be between 1 and 32");
46
+ if (lg_size < 2) { throw std::invalid_argument("lg_size must be >= 2"); }
47
+ if (num_valid_bits < 1 || num_valid_bits > 32) { throw std::invalid_argument("num_valid_bits must be between 1 and 32"); }
48
48
  }
49
49
 
50
50
  template<typename A>
@@ -71,8 +71,8 @@ void u32_table<A>::clear() {
71
71
  template<typename A>
72
72
  bool u32_table<A>::maybe_insert(uint32_t item) {
73
73
  const uint32_t index = lookup(item);
74
- if (slots[index] == item) return false;
75
- if (slots[index] != UINT32_MAX) throw std::logic_error("could not insert");
74
+ if (slots[index] == item) { return false; }
75
+ if (slots[index] != UINT32_MAX) { throw std::logic_error("could not insert"); }
76
76
  slots[index] = item;
77
77
  num_items++;
78
78
  if (U32_TABLE_UPSIZE_DENOM * num_items > U32_TABLE_UPSIZE_NUMER * (1 << lg_size)) {
@@ -84,9 +84,9 @@ bool u32_table<A>::maybe_insert(uint32_t item) {
84
84
  template<typename A>
85
85
  bool u32_table<A>::maybe_delete(uint32_t item) {
86
86
  const uint32_t index = lookup(item);
87
- if (slots[index] == UINT32_MAX) return false;
88
- if (slots[index] != item) throw std::logic_error("item does not exist");
89
- if (num_items == 0) throw std::logic_error("delete error");
87
+ if (slots[index] == UINT32_MAX) { return false; }
88
+ if (slots[index] != item) { throw std::logic_error("item does not exist"); }
89
+ if (num_items == 0) { throw std::logic_error("delete error"); }
90
90
  // delete the item
91
91
  slots[index] = UINT32_MAX;
92
92
  num_items--;
@@ -129,7 +129,7 @@ uint32_t u32_table<A>::lookup(uint32_t item) const {
129
129
  const uint32_t mask = size - 1;
130
130
  const uint8_t shift = num_valid_bits - lg_size;
131
131
  uint32_t probe = item >> shift;
132
- if (probe > mask) throw std::logic_error("probe out of range");
132
+ if (probe > mask) { throw std::logic_error("probe out of range"); }
133
133
  while (slots[probe] != item && slots[probe] != UINT32_MAX) {
134
134
  probe = (probe + 1) & mask;
135
135
  }
@@ -140,17 +140,17 @@ uint32_t u32_table<A>::lookup(uint32_t item) const {
140
140
  template<typename A>
141
141
  void u32_table<A>::must_insert(uint32_t item) {
142
142
  const uint32_t index = lookup(item);
143
- if (slots[index] == item) throw std::logic_error("item exists");
144
- if (slots[index] != UINT32_MAX) throw std::logic_error("could not insert");
143
+ if (slots[index] == item) { throw std::logic_error("item exists"); }
144
+ if (slots[index] != UINT32_MAX) { throw std::logic_error("could not insert"); }
145
145
  slots[index] = item;
146
146
  }
147
147
 
148
148
  template<typename A>
149
149
  void u32_table<A>::rebuild(uint8_t new_lg_size) {
150
- if (new_lg_size < 2) throw std::logic_error("lg_size must be >= 2");
150
+ if (new_lg_size < 2) { throw std::logic_error("lg_size must be >= 2"); }
151
151
  const uint32_t old_size = 1 << lg_size;
152
152
  const uint32_t new_size = 1 << new_lg_size;
153
- if (new_size <= num_items) throw std::logic_error("new_size <= num_items");
153
+ if (new_size <= num_items) { throw std::logic_error("new_size <= num_items"); }
154
154
  vector_u32 old_slots = std::move(slots);
155
155
  slots = vector_u32(new_size, UINT32_MAX, old_slots.get_allocator());
156
156
  lg_size = new_lg_size;
@@ -169,7 +169,7 @@ void u32_table<A>::rebuild(uint8_t new_lg_size) {
169
169
  // The result is nearly sorted, so make sure to use an efficient sort for that case
170
170
  template<typename A>
171
171
  auto u32_table<A>::unwrapping_get_items() const -> vector_u32 {
172
- if (num_items == 0) return vector_u32(slots.get_allocator());
172
+ if (num_items == 0) { return vector_u32(slots.get_allocator()); }
173
173
  const uint32_t table_size = 1 << lg_size;
174
174
  vector_u32 result(num_items, 0, slots.get_allocator());
175
175
  size_t i = 0;
@@ -187,9 +187,9 @@ auto u32_table<A>::unwrapping_get_items() const -> vector_u32 {
187
187
  // the rest of the table is processed normally
188
188
  while (i < table_size) {
189
189
  const uint32_t item = slots[i++];
190
- if (item != UINT32_MAX) result[l++] = item;
190
+ if (item != UINT32_MAX) { result[l++] = item; }
191
191
  }
192
- if (l != r + 1) throw std::logic_error("unwrapping error");
192
+ if (l != r + 1) { throw std::logic_error("unwrapping error"); }
193
193
  return result;
194
194
  }
195
195
 
@@ -213,7 +213,7 @@ void u32_table<A>::merge(
213
213
  else if (arr_a[a] < arr_b[b]) { arr_c[c] = arr_a[a++]; }
214
214
  else { arr_c[c] = arr_b[b++]; }
215
215
  }
216
- if (a != lim_a || b != lim_b) throw std::logic_error("merging error");
216
+ if (a != lim_a || b != lim_b) { throw std::logic_error("merging error"); }
217
217
  }
218
218
 
219
219
  // In applications where the input array is already nearly sorted,