datasketches 0.5.1 → 0.5.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (120) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +9 -0
  3. data/ext/datasketches/cpc_wrapper.cpp +8 -9
  4. data/ext/datasketches/fi_wrapper.cpp +4 -5
  5. data/ext/datasketches/hll_wrapper.cpp +9 -10
  6. data/ext/datasketches/kll_wrapper.cpp +13 -12
  7. data/ext/datasketches/theta_wrapper.cpp +11 -13
  8. data/ext/datasketches/vo_wrapper.cpp +1 -2
  9. data/lib/datasketches/version.rb +1 -1
  10. data/vendor/datasketches-cpp/CMakeLists.txt +5 -3
  11. data/vendor/datasketches-cpp/CODE_OF_CONDUCT.md +1 -1
  12. data/vendor/datasketches-cpp/LICENSE +14 -0
  13. data/vendor/datasketches-cpp/README.md +67 -73
  14. data/vendor/datasketches-cpp/benchmarks/CMakeLists.txt +52 -0
  15. data/vendor/datasketches-cpp/benchmarks/benchmark_count_min_sketch.cpp +153 -0
  16. data/vendor/datasketches-cpp/benchmarks/benchmark_count_min_sketch_serialization.cpp +161 -0
  17. data/vendor/datasketches-cpp/common/CMakeLists.txt +1 -0
  18. data/vendor/datasketches-cpp/common/include/binomial_bounds.hpp +2 -2
  19. data/vendor/datasketches-cpp/common/include/fdlibm_log.hpp +101 -0
  20. data/vendor/datasketches-cpp/common/include/serde.hpp +6 -0
  21. data/vendor/datasketches-cpp/common/test/CMakeLists.txt +44 -2
  22. data/vendor/datasketches-cpp/common/test/binomial_bounds_test.cpp +279 -0
  23. data/vendor/datasketches-cpp/common/test/deserialize_hardening_test.cpp +188 -0
  24. data/vendor/datasketches-cpp/count/include/count_min.hpp +17 -4
  25. data/vendor/datasketches-cpp/count/include/count_min_impl.hpp +65 -83
  26. data/vendor/datasketches-cpp/count/test/count_min_test.cpp +63 -7
  27. data/vendor/datasketches-cpp/cpc/include/compression_data.hpp +2 -0
  28. data/vendor/datasketches-cpp/cpc/include/cpc_compressor_impl.hpp +16 -7
  29. data/vendor/datasketches-cpp/cpc/include/cpc_sketch.hpp +1 -0
  30. data/vendor/datasketches-cpp/cpc/include/cpc_sketch_impl.hpp +54 -36
  31. data/vendor/datasketches-cpp/cpc/include/cpc_union_impl.hpp +27 -27
  32. data/vendor/datasketches-cpp/cpc/include/cpc_util.hpp +7 -7
  33. data/vendor/datasketches-cpp/cpc/include/icon_estimator.hpp +5 -5
  34. data/vendor/datasketches-cpp/cpc/include/u32_table_impl.hpp +16 -16
  35. data/vendor/datasketches-cpp/cpc/test/cpc_sketch_test.cpp +35 -0
  36. data/vendor/datasketches-cpp/fi/include/frequent_items_sketch.hpp +28 -3
  37. data/vendor/datasketches-cpp/fi/include/frequent_items_sketch_impl.hpp +41 -25
  38. data/vendor/datasketches-cpp/fi/include/reverse_purge_hash_map.hpp +3 -1
  39. data/vendor/datasketches-cpp/fi/include/reverse_purge_hash_map_impl.hpp +10 -5
  40. data/vendor/datasketches-cpp/fi/test/frequent_items_sketch_serialize_for_java.cpp +24 -0
  41. data/vendor/datasketches-cpp/fi/test/frequent_items_sketch_test.cpp +116 -0
  42. data/vendor/datasketches-cpp/filters/include/bloom_filter.hpp +1 -1
  43. data/vendor/datasketches-cpp/filters/include/bloom_filter_impl.hpp +32 -12
  44. data/vendor/datasketches-cpp/filters/test/bloom_filter_test.cpp +28 -1
  45. data/vendor/datasketches-cpp/hll/include/CouponHashSet-internal.hpp +1 -2
  46. data/vendor/datasketches-cpp/hll/include/CouponList-internal.hpp +19 -5
  47. data/vendor/datasketches-cpp/hll/include/CubicInterpolation-internal.hpp +4 -4
  48. data/vendor/datasketches-cpp/hll/include/HarmonicNumbers-internal.hpp +2 -1
  49. data/vendor/datasketches-cpp/hll/include/Hll4Array-internal.hpp +4 -4
  50. data/vendor/datasketches-cpp/hll/include/Hll4Array.hpp +1 -1
  51. data/vendor/datasketches-cpp/hll/include/Hll6Array-internal.hpp +3 -3
  52. data/vendor/datasketches-cpp/hll/include/Hll6Array.hpp +1 -1
  53. data/vendor/datasketches-cpp/hll/include/Hll8Array-internal.hpp +6 -3
  54. data/vendor/datasketches-cpp/hll/include/Hll8Array.hpp +1 -1
  55. data/vendor/datasketches-cpp/hll/include/HllArray-internal.hpp +35 -29
  56. data/vendor/datasketches-cpp/hll/include/HllArray.hpp +1 -1
  57. data/vendor/datasketches-cpp/hll/include/HllSketch-internal.hpp +3 -5
  58. data/vendor/datasketches-cpp/hll/include/HllSketchImpl-internal.hpp +5 -11
  59. data/vendor/datasketches-cpp/hll/include/HllSketchImpl.hpp +2 -4
  60. data/vendor/datasketches-cpp/hll/include/HllSketchImplFactory.hpp +11 -10
  61. data/vendor/datasketches-cpp/hll/include/HllUnion-internal.hpp +17 -8
  62. data/vendor/datasketches-cpp/hll/include/HllUtil.hpp +12 -3
  63. data/vendor/datasketches-cpp/hll/include/coupon_iterator-internal.hpp +2 -2
  64. data/vendor/datasketches-cpp/hll/include/coupon_iterator.hpp +3 -0
  65. data/vendor/datasketches-cpp/hll/include/hll.hpp +11 -4
  66. data/vendor/datasketches-cpp/hll/include/hll.private.hpp +19 -0
  67. data/vendor/datasketches-cpp/hll/test/CMakeLists.txt +2 -0
  68. data/vendor/datasketches-cpp/hll/test/CouponListTest.cpp +64 -0
  69. data/vendor/datasketches-cpp/hll/test/HllFullSizeTest.cpp +137 -0
  70. data/vendor/datasketches-cpp/hll/test/HllKxqRebuildTest.cpp +150 -0
  71. data/vendor/datasketches-cpp/hll/test/HllSketchTest.cpp +3 -3
  72. data/vendor/datasketches-cpp/hll/test/HllUnionTest.cpp +71 -1
  73. data/vendor/datasketches-cpp/kll/include/kll_helper_impl.hpp +10 -10
  74. data/vendor/datasketches-cpp/kll/include/kll_sketch.hpp +10 -1
  75. data/vendor/datasketches-cpp/kll/include/kll_sketch_impl.hpp +33 -24
  76. data/vendor/datasketches-cpp/kll/test/kll_sketch_deserialize_from_java_test.cpp +24 -0
  77. data/vendor/datasketches-cpp/kll/test/kll_sketch_serialize_for_java.cpp +10 -0
  78. data/vendor/datasketches-cpp/quantiles/include/quantiles_sketch.hpp +11 -2
  79. data/vendor/datasketches-cpp/quantiles/include/quantiles_sketch_impl.hpp +24 -15
  80. data/vendor/datasketches-cpp/req/include/req_sketch.hpp +9 -0
  81. data/vendor/datasketches-cpp/req/include/req_sketch_impl.hpp +24 -15
  82. data/vendor/datasketches-cpp/req/test/req_sketch_deserialize_from_java_test.cpp +46 -0
  83. data/vendor/datasketches-cpp/req/test/req_sketch_serialize_for_java.cpp +20 -0
  84. data/vendor/datasketches-cpp/req/test/req_sketch_test.cpp +70 -0
  85. data/vendor/datasketches-cpp/sampling/include/ebpps_sample_impl.hpp +17 -8
  86. data/vendor/datasketches-cpp/sampling/include/ebpps_sketch.hpp +13 -0
  87. data/vendor/datasketches-cpp/sampling/include/var_opt_sketch.hpp +10 -1
  88. data/vendor/datasketches-cpp/sampling/include/var_opt_sketch_impl.hpp +6 -7
  89. data/vendor/datasketches-cpp/sampling/include/var_opt_union.hpp +5 -1
  90. data/vendor/datasketches-cpp/sampling/include/var_opt_union_impl.hpp +2 -1
  91. data/vendor/datasketches-cpp/sampling/test/ebpps_allocation_test.cpp +1 -1
  92. data/vendor/datasketches-cpp/sampling/test/ebpps_sketch_test.cpp +2 -2
  93. data/vendor/datasketches-cpp/sampling/test/var_opt_allocation_test.cpp +1 -1
  94. data/vendor/datasketches-cpp/sampling/test/var_opt_sketch_test.cpp +10 -4
  95. data/vendor/datasketches-cpp/sampling/test/var_opt_union_test.cpp +12 -0
  96. data/vendor/datasketches-cpp/tdigest/include/tdigest.hpp +38 -2
  97. data/vendor/datasketches-cpp/tdigest/include/tdigest_impl.hpp +168 -9
  98. data/vendor/datasketches-cpp/tdigest/test/CMakeLists.txt +1 -0
  99. data/vendor/datasketches-cpp/tdigest/test/tdigest_iterator_test.cpp +274 -0
  100. data/vendor/datasketches-cpp/tdigest/test/tdigest_test.cpp +275 -0
  101. data/vendor/datasketches-cpp/theta/include/compact_theta_sketch_parser.hpp +2 -0
  102. data/vendor/datasketches-cpp/theta/include/compact_theta_sketch_parser_impl.hpp +24 -3
  103. data/vendor/datasketches-cpp/theta/include/theta_constants.hpp +4 -2
  104. data/vendor/datasketches-cpp/theta/include/theta_helpers.hpp +32 -0
  105. data/vendor/datasketches-cpp/theta/include/theta_set_difference_base_impl.hpp +4 -2
  106. data/vendor/datasketches-cpp/theta/include/theta_sketch.hpp +22 -4
  107. data/vendor/datasketches-cpp/theta/include/theta_sketch_impl.hpp +60 -38
  108. data/vendor/datasketches-cpp/theta/include/theta_union_base_impl.hpp +2 -6
  109. data/vendor/datasketches-cpp/theta/include/theta_update_sketch_base_impl.hpp +2 -2
  110. data/vendor/datasketches-cpp/theta/test/bit_packing_test.cpp +50 -0
  111. data/vendor/datasketches-cpp/theta/test/theta_a_not_b_test.cpp +22 -0
  112. data/vendor/datasketches-cpp/theta/test/theta_sketch_test.cpp +315 -0
  113. data/vendor/datasketches-cpp/tools/rat-check.sh +68 -0
  114. data/vendor/datasketches-cpp/tuple/include/array_tuple_sketch.hpp +35 -4
  115. data/vendor/datasketches-cpp/tuple/include/array_tuple_sketch_impl.hpp +2 -2
  116. data/vendor/datasketches-cpp/tuple/include/tuple_sketch.hpp +41 -0
  117. data/vendor/datasketches-cpp/tuple/include/tuple_sketch_impl.hpp +5 -4
  118. data/vendor/datasketches-cpp/tuple/test/tuple_sketch_test.cpp +59 -0
  119. data/vendor/datasketches-cpp/version.cfg.in +1 -1
  120. metadata +12 -2
@@ -18,13 +18,40 @@
18
18
  */
19
19
 
20
20
  #include <catch2/catch.hpp>
21
+ #include <cmath>
22
+ #include <cstdint>
23
+ #include <cstring>
24
+ #include <initializer_list>
21
25
  #include <iostream>
22
26
  #include <fstream>
27
+ #include <sstream>
28
+ #include <utility>
23
29
 
24
30
  #include "tdigest.hpp"
25
31
 
26
32
  namespace datasketches {
27
33
 
34
+ namespace {
35
+ constexpr size_t header_size = 8;
36
+ constexpr size_t counts_size = 8;
37
+ constexpr size_t min_offset = header_size + counts_size;
38
+ constexpr size_t max_offset = min_offset + sizeof(double);
39
+ constexpr size_t first_centroid_mean_offset = min_offset + sizeof(double) * 2;
40
+ constexpr size_t first_centroid_weight_offset = first_centroid_mean_offset + sizeof(double);
41
+ constexpr size_t first_buffered_value_offset = first_centroid_mean_offset;
42
+ constexpr size_t single_value_offset = header_size;
43
+
44
+ template <typename T>
45
+ void write_bytes(std::vector<uint8_t>& bytes, size_t offset, T value) {
46
+ std::memcpy(bytes.data() + offset, &value, sizeof(T));
47
+ }
48
+
49
+ template <typename T>
50
+ void write_bytes(std::string& data, size_t offset, T value) {
51
+ std::memcpy(&data[offset], &value, sizeof(T));
52
+ }
53
+ } // namespace
54
+
28
55
  TEST_CASE("empty", "[tdigest]") {
29
56
  tdigest_double td(10);
30
57
  // std::cout << td.to_string();
@@ -145,6 +172,31 @@ TEST_CASE("merge small", "[tdigest]") {
145
172
  REQUIRE(td1.get_rank(3.01) == 1);
146
173
  }
147
174
 
175
+ TEST_CASE("merge preserves deserialized min max with weighted tails", "[tdigest]") {
176
+ tdigest_double source(100);
177
+ source.update(0);
178
+ source.update(50);
179
+ source.update(90);
180
+ auto bytes = source.serialize();
181
+ write_bytes(bytes, min_offset, -1.0);
182
+ write_bytes(bytes, max_offset, 100.0);
183
+ auto other = tdigest_double::deserialize(bytes.data(), bytes.size());
184
+ REQUIRE(other.get_min_value() == -1.0);
185
+ REQUIRE(other.get_max_value() == 100.0);
186
+
187
+ tdigest_double empty(100);
188
+ empty.merge(other);
189
+ REQUIRE(empty.get_min_value() == -1.0);
190
+ REQUIRE(empty.get_max_value() == 100.0);
191
+
192
+ tdigest_double left(100);
193
+ left.update(10);
194
+ left.update(20);
195
+ left.merge(other);
196
+ REQUIRE(left.get_min_value() == -1.0);
197
+ REQUIRE(left.get_max_value() == 100.0);
198
+ }
199
+
148
200
  TEST_CASE("merge large", "[tdigest]") {
149
201
  const size_t n = 10000;
150
202
  tdigest_double td1;
@@ -453,4 +505,227 @@ TEST_CASE("deserialize from reference implementation bytes float", "[tdigest]")
453
505
  REQUIRE(td.get_rank(n) == 1);
454
506
  }
455
507
 
508
+ TEST_CASE("iterate centroids", "[tdigest]") {
509
+ tdigest_double td(100);
510
+ for (int i = 0; i < 10; i++) {
511
+ td.update(i);
512
+ }
513
+
514
+ auto centroid_count = 0;
515
+ uint64_t total_weight = 0;
516
+ for (const auto &centroid: td) {
517
+ centroid_count++;
518
+ total_weight += centroid.second;
519
+ }
520
+ // Ensure that centroids are retrieved for a case where there is buffered values
521
+ REQUIRE(centroid_count == 10);
522
+ REQUIRE(td.get_total_weight() == total_weight);
523
+ }
524
+
525
+ TEST_CASE("update rejects positive infinity", "[tdigest]") {
526
+ tdigest_double td(100);
527
+ td.update(1.0);
528
+ td.update(2.0);
529
+ td.update(std::numeric_limits<double>::infinity());
530
+ REQUIRE(td.get_total_weight() == 2);
531
+ REQUIRE(td.get_max_value() == 2.0);
532
+ }
533
+
534
+ TEST_CASE("update rejects negative infinity", "[tdigest]") {
535
+ tdigest_double td(100);
536
+ td.update(1.0);
537
+ td.update(2.0);
538
+ td.update(-std::numeric_limits<double>::infinity());
539
+ REQUIRE(td.get_total_weight() == 2);
540
+ REQUIRE(td.get_min_value() == 1.0);
541
+ }
542
+
543
+ TEST_CASE("deserialize bytes rejects NaN single value", "[tdigest]") {
544
+ tdigest_double td(100);
545
+ td.update(1.0);
546
+ auto bytes = td.serialize();
547
+ write_bytes(bytes, single_value_offset, std::numeric_limits<double>::quiet_NaN());
548
+ REQUIRE_THROWS_AS(tdigest_double::deserialize(bytes.data(), bytes.size()), std::invalid_argument);
549
+ }
550
+
551
+ TEST_CASE("deserialize stream rejects infinity min", "[tdigest]") {
552
+ tdigest_double td(100);
553
+ td.update(1.0);
554
+ td.update(2.0);
555
+ td.update(3.0);
556
+ auto bytes = td.serialize();
557
+ std::string data(reinterpret_cast<const char*>(bytes.data()), bytes.size());
558
+ write_bytes(data, min_offset, std::numeric_limits<double>::infinity());
559
+ std::istringstream is(data, std::ios::binary);
560
+ REQUIRE_THROWS_AS(tdigest_double::deserialize(is), std::invalid_argument);
561
+ }
562
+
563
+ TEST_CASE("deserialize bytes rejects NaN centroid mean", "[tdigest]") {
564
+ tdigest_double td(100);
565
+ for (int i = 0; i < 10; ++i) td.update(i);
566
+ auto bytes = td.serialize();
567
+ write_bytes(bytes, first_centroid_mean_offset, std::numeric_limits<double>::quiet_NaN());
568
+ REQUIRE_THROWS_AS(tdigest_double::deserialize(bytes.data(), bytes.size()), std::invalid_argument);
569
+ }
570
+
571
+ TEST_CASE("deserialize bytes rejects NaN buffered value", "[tdigest]") {
572
+ tdigest_double td(100);
573
+ td.update(1.0);
574
+ td.update(2.0);
575
+ auto bytes = td.serialize(0, true);
576
+ write_bytes(bytes, first_buffered_value_offset, std::numeric_limits<double>::quiet_NaN());
577
+ REQUIRE_THROWS_AS(tdigest_double::deserialize(bytes.data(), bytes.size()), std::invalid_argument);
578
+ }
579
+
580
+ TEST_CASE("deserialize bytes rejects infinity single value", "[tdigest]") {
581
+ tdigest_double td(100);
582
+ td.update(1.0);
583
+ auto bytes = td.serialize();
584
+ write_bytes(bytes, single_value_offset, std::numeric_limits<double>::infinity());
585
+ REQUIRE_THROWS_AS(tdigest_double::deserialize(bytes.data(), bytes.size()), std::invalid_argument);
586
+ }
587
+
588
+ TEST_CASE("deserialize bytes rejects NaN max", "[tdigest]") {
589
+ tdigest_double td(100);
590
+ td.update(1.0);
591
+ td.update(2.0);
592
+ auto bytes = td.serialize();
593
+ write_bytes(bytes, max_offset, std::numeric_limits<double>::quiet_NaN());
594
+ REQUIRE_THROWS_AS(tdigest_double::deserialize(bytes.data(), bytes.size()), std::invalid_argument);
595
+ }
596
+
597
+ TEST_CASE("deserialize bytes rejects infinity max", "[tdigest]") {
598
+ tdigest_double td(100);
599
+ td.update(1.0);
600
+ td.update(2.0);
601
+ auto bytes = td.serialize();
602
+ write_bytes(bytes, max_offset, std::numeric_limits<double>::infinity());
603
+ REQUIRE_THROWS_AS(tdigest_double::deserialize(bytes.data(), bytes.size()), std::invalid_argument);
604
+ }
605
+
606
+ TEST_CASE("deserialize bytes rejects infinity buffered value", "[tdigest]") {
607
+ tdigest_double td(100);
608
+ td.update(1.0);
609
+ td.update(2.0);
610
+ auto bytes = td.serialize(0, true);
611
+ write_bytes(bytes, first_buffered_value_offset, std::numeric_limits<double>::infinity());
612
+ REQUIRE_THROWS_AS(tdigest_double::deserialize(bytes.data(), bytes.size()), std::invalid_argument);
613
+ }
614
+
615
+ TEST_CASE("deserialize bytes rejects zero centroid weight", "[tdigest]") {
616
+ tdigest_double td(100);
617
+ for (int i = 0; i < 10; ++i) td.update(i);
618
+ auto bytes = td.serialize();
619
+ write_bytes(bytes, first_centroid_weight_offset, static_cast<uint64_t>(0));
620
+ REQUIRE_THROWS_AS(tdigest_double::deserialize(bytes.data(), bytes.size()), std::invalid_argument);
621
+ }
622
+
623
+ TEST_CASE("deserialize stream rejects zero centroid weight", "[tdigest]") {
624
+ tdigest_double td(100);
625
+ for (int i = 0; i < 10; ++i) td.update(i);
626
+ auto bytes = td.serialize();
627
+ std::string data(reinterpret_cast<const char*>(bytes.data()), bytes.size());
628
+ write_bytes(data, first_centroid_weight_offset, static_cast<uint64_t>(0));
629
+ std::istringstream is(data, std::ios::binary);
630
+ REQUIRE_THROWS_AS(tdigest_double::deserialize(is), std::invalid_argument);
631
+ }
632
+
633
+ TEST_CASE("quantiles are monotonic and stay within min and max", "[tdigest]") {
634
+ tdigest_double td(100);
635
+ for (int i = 0; i < 10000; ++i) td.update(i);
636
+ double previous = td.get_min_value();
637
+ for (int i = 0; i <= 1000; ++i) {
638
+ const double quantile = td.get_quantile(i / 1000.0);
639
+ REQUIRE(quantile >= previous);
640
+ REQUIRE(quantile >= td.get_min_value());
641
+ REQUIRE(quantile <= td.get_max_value());
642
+ previous = quantile;
643
+ }
644
+ }
645
+
646
+ TEST_CASE("rank below the first centroid stays normalized", "[tdigest]") {
647
+ // The format allows a first centroid heavier than 1. The left tail of get_rank()
648
+ // must be divided by the total weight, the same way the right tail is.
649
+ tdigest_double source(200);
650
+ for (int i = 0; i < 1000; ++i) source.update(i);
651
+ auto bytes = source.serialize();
652
+ double first_mean = 0;
653
+ std::memcpy(&first_mean, bytes.data() + first_centroid_mean_offset, sizeof(double));
654
+ REQUIRE(first_mean == 0);
655
+ write_bytes(bytes, min_offset, -1.0);
656
+ write_bytes(bytes, first_centroid_weight_offset, static_cast<uint64_t>(100));
657
+ const auto td = tdigest_double::deserialize(bytes.data(), bytes.size());
658
+ const double total_weight = static_cast<double>(td.get_total_weight());
659
+ REQUIRE(td.get_rank(-1) == 0.5 / total_weight);
660
+ REQUIRE(td.get_rank(-0.5) == (1.0 + ((100.0 / 2.0 - 1.0) * 0.5)) / total_weight);
661
+ double previous = 0;
662
+ for (int i = 0; i <= 100; ++i) {
663
+ const double rank = td.get_rank(-1.0 + (i / 100.0));
664
+ REQUIRE(rank >= 0);
665
+ REQUIRE(rank <= 1);
666
+ REQUIRE(rank >= previous);
667
+ previous = rank;
668
+ }
669
+ }
670
+
671
+ TEST_CASE("quantile above the last centroid does not exceed max", "[tdigest]") {
672
+ tdigest_double source(200);
673
+ for (int i = 0; i < 1000; ++i) source.update(i);
674
+ auto bytes = source.serialize();
675
+ uint32_t num_centroids = 0;
676
+ std::memcpy(&num_centroids, bytes.data() + header_size, sizeof(uint32_t));
677
+ REQUIRE(num_centroids > 1);
678
+ const size_t last_weight_offset = first_centroid_weight_offset + (num_centroids - 1) * 16;
679
+ write_bytes(bytes, last_weight_offset, static_cast<uint64_t>(100));
680
+ const auto td = tdigest_double::deserialize(bytes.data(), bytes.size());
681
+ double previous = td.get_min_value();
682
+ for (int i = 0; i <= 1000; ++i) {
683
+ const double quantile = td.get_quantile(i / 1000.0);
684
+ REQUIRE(quantile >= previous);
685
+ REQUIRE(quantile <= td.get_max_value());
686
+ previous = quantile;
687
+ }
688
+ }
689
+
690
+ tdigest_double sketch_from_centroids(double min, double max,
691
+ std::initializer_list<std::pair<double, uint64_t>> centroids) {
692
+ const auto num_centroids = static_cast<uint32_t>(centroids.size());
693
+ std::vector<uint8_t> bytes(first_centroid_mean_offset + static_cast<size_t>(num_centroids) * 16, 0);
694
+ bytes[0] = 2; // preamble longs
695
+ bytes[1] = 1; // serial version
696
+ bytes[2] = 20; // sketch type
697
+ write_bytes(bytes, 3, static_cast<uint16_t>(100));
698
+ write_bytes(bytes, header_size, num_centroids);
699
+ write_bytes(bytes, header_size + sizeof(uint32_t), static_cast<uint32_t>(0));
700
+ write_bytes(bytes, min_offset, min);
701
+ write_bytes(bytes, max_offset, max);
702
+ size_t offset = first_centroid_mean_offset;
703
+ for (const auto& centroid : centroids) {
704
+ write_bytes(bytes, offset, centroid.first);
705
+ write_bytes(bytes, offset + sizeof(double), centroid.second);
706
+ offset += 16;
707
+ }
708
+ return tdigest_double::deserialize(bytes.data(), bytes.size());
709
+ }
710
+
711
+ TEST_CASE("quantile with a last centroid of weight 2 returns max", "[tdigest]") {
712
+ // weight == total - 1 and last weight == 2 is 0/0 unless that case returns max.
713
+ const auto td = sketch_from_centroids(0, 20, {{0, 10}, {10, 2}});
714
+ const double quantile = td.get_quantile(11.0 / 12.0);
715
+ REQUIRE_FALSE(std::isnan(quantile));
716
+ REQUIRE(quantile == 20);
717
+ }
718
+
719
+ TEST_CASE("quantile right tail approaches max from below", "[tdigest]") {
720
+ const auto td = sketch_from_centroids(0, 40, {{10, 100}, {20, 100}, {30, 100}});
721
+ REQUIRE(td.get_quantile(0.9) == Approx(34.081632653061224).epsilon(1e-12));
722
+ }
723
+
724
+ TEST_CASE("quantile interpolation weights the nearer centroid more", "[tdigest]") {
725
+ const auto td = sketch_from_centroids(0, 40, {{10, 100}, {20, 100}, {30, 100}});
726
+ // Target weight 80 sits between 10 and 20, closer to 10.
727
+ // weighted_average(10, 70, 20, 30) == 13; the swapped weights would return 17.
728
+ REQUIRE(td.get_quantile(80.0 / 300.0) == 13);
729
+ }
730
+
456
731
  } /* namespace datasketches */
@@ -38,6 +38,8 @@ public:
38
38
  };
39
39
 
40
40
  static compact_theta_sketch_data parse(const void* ptr, size_t size, uint64_t seed, bool dump_on_error = false);
41
+ static void check_v4_entry_bits(uint8_t entry_bits);
42
+ static void check_v4_num_entries_bytes(uint8_t num_entries_bytes);
41
43
 
42
44
  private:
43
45
  // offsets are in sizeof(type)
@@ -49,6 +49,7 @@ auto compact_theta_sketch_parser<dummy>::parse(const void* ptr, size_t size, uin
49
49
  theta = reinterpret_cast<const uint64_t*>(ptr)[COMPACT_SKETCH_V4_THETA_U64];
50
50
  }
51
51
  const uint8_t num_entries_bytes = reinterpret_cast<const uint8_t*>(ptr)[COMPACT_SKETCH_V4_NUM_ENTRIES_BYTES_BYTE];
52
+ check_v4_num_entries_bytes(num_entries_bytes);
52
53
  size_t data_offset_bytes = has_theta ? COMPACT_SKETCH_V4_PACKED_DATA_ESTIMATION_BYTE : COMPACT_SKETCH_V4_PACKED_DATA_EXACT_BYTE;
53
54
  check_memory_size(ptr, size, data_offset_bytes + num_entries_bytes, dump_on_error);
54
55
  uint32_t num_entries = 0;
@@ -58,7 +59,8 @@ auto compact_theta_sketch_parser<dummy>::parse(const void* ptr, size_t size, uin
58
59
  }
59
60
  data_offset_bytes += num_entries_bytes;
60
61
  const uint8_t entry_bits = reinterpret_cast<const uint8_t*>(ptr)[COMPACT_SKETCH_V4_ENTRY_BITS_BYTE];
61
- const size_t expected_bits = entry_bits * num_entries;
62
+ check_v4_entry_bits(entry_bits);
63
+ const uint64_t expected_bits = static_cast<uint64_t>(entry_bits) * num_entries;
62
64
  const size_t expected_size_bytes = data_offset_bytes + whole_bytes_to_hold_bits(expected_bits);
63
65
  check_memory_size(ptr, size, expected_size_bytes, dump_on_error);
64
66
  return {false, true, seed_hash, num_entries, theta,
@@ -80,8 +82,9 @@ auto compact_theta_sketch_parser<dummy>::parse(const void* ptr, size_t size, uin
80
82
  check_memory_size(ptr, size, 16, dump_on_error);
81
83
  return {false, true, seed_hash, 1, theta, reinterpret_cast<const uint64_t*>(ptr) + COMPACT_SKETCH_SINGLE_ENTRY_U64, 64};
82
84
  }
83
- const uint32_t num_entries = reinterpret_cast<const uint32_t*>(ptr)[COMPACT_SKETCH_NUM_ENTRIES_U32];
84
85
  const size_t entries_start_u64 = has_theta ? COMPACT_SKETCH_ENTRIES_ESTIMATION_U64 : COMPACT_SKETCH_ENTRIES_EXACT_U64;
86
+ check_memory_size(ptr, size, entries_start_u64 * sizeof(uint64_t), dump_on_error);
87
+ const uint32_t num_entries = reinterpret_cast<const uint32_t*>(ptr)[COMPACT_SKETCH_NUM_ENTRIES_U32];
85
88
  const uint64_t* entries = reinterpret_cast<const uint64_t*>(ptr) + entries_start_u64;
86
89
  const size_t expected_size_bytes = (entries_start_u64 + num_entries) * sizeof(uint64_t);
87
90
  check_memory_size(ptr, size, expected_size_bytes, dump_on_error);
@@ -90,6 +93,7 @@ auto compact_theta_sketch_parser<dummy>::parse(const void* ptr, size_t size, uin
90
93
  }
91
94
  case 1: {
92
95
  uint16_t seed_hash = compute_seed_hash(seed);
96
+ check_memory_size(ptr, size, COMPACT_SKETCH_ENTRIES_ESTIMATION_U64 * sizeof(uint64_t), dump_on_error);
93
97
  const uint32_t num_entries = reinterpret_cast<const uint32_t*>(ptr)[COMPACT_SKETCH_NUM_ENTRIES_U32];
94
98
  uint64_t theta = reinterpret_cast<const uint64_t*>(ptr)[COMPACT_SKETCH_THETA_U64];
95
99
  bool is_empty = (num_entries == 0) && (theta == theta_constants::MAX_THETA);
@@ -106,16 +110,18 @@ auto compact_theta_sketch_parser<dummy>::parse(const void* ptr, size_t size, uin
106
110
  if (preamble_size == 1) {
107
111
  return {true, true, seed_hash, 0, theta_constants::MAX_THETA, nullptr, 64};
108
112
  } else if (preamble_size == 2) {
113
+ check_memory_size(ptr, size, COMPACT_SKETCH_ENTRIES_EXACT_U64 * sizeof(uint64_t), dump_on_error);
109
114
  const uint32_t num_entries = reinterpret_cast<const uint32_t*>(ptr)[COMPACT_SKETCH_NUM_ENTRIES_U32];
110
115
  if (num_entries == 0) {
111
116
  return {true, true, seed_hash, 0, theta_constants::MAX_THETA, nullptr, 64};
112
117
  } else {
113
- const size_t expected_size_bytes = (preamble_size + num_entries) << 3;
118
+ const size_t expected_size_bytes = (preamble_size + static_cast<size_t>(num_entries)) << 3;
114
119
  check_memory_size(ptr, size, expected_size_bytes, dump_on_error);
115
120
  const uint64_t* entries = reinterpret_cast<const uint64_t*>(ptr) + COMPACT_SKETCH_ENTRIES_EXACT_U64;
116
121
  return {false, true, seed_hash, num_entries, theta_constants::MAX_THETA, entries, 64};
117
122
  }
118
123
  } else if (preamble_size == 3) {
124
+ check_memory_size(ptr, size, COMPACT_SKETCH_ENTRIES_ESTIMATION_U64 * sizeof(uint64_t), dump_on_error);
119
125
  const uint32_t num_entries = reinterpret_cast<const uint32_t*>(ptr)[COMPACT_SKETCH_NUM_ENTRIES_U32];
120
126
  uint64_t theta = reinterpret_cast<const uint64_t*>(ptr)[COMPACT_SKETCH_THETA_U64];
121
127
  bool is_empty = (num_entries == 0) && (theta == theta_constants::MAX_THETA);
@@ -140,6 +146,21 @@ void compact_theta_sketch_parser<dummy>::check_memory_size(const void* ptr, size
140
146
  + (dump_on_error ? (", sketch dump: " + hex_dump(reinterpret_cast<const uint8_t*>(ptr), actual_bytes)) : ""));
141
147
  }
142
148
 
149
+ template<bool dummy>
150
+ void compact_theta_sketch_parser<dummy>::check_v4_entry_bits(uint8_t entry_bits) {
151
+ // deltas between ordered hashes below 2^63 need 1 to 63 bits
152
+ if (entry_bits == 0 || entry_bits > 63) {
153
+ throw std::invalid_argument("entry bits must be in [1, 63], actual " + std::to_string(entry_bits));
154
+ }
155
+ }
156
+
157
+ template<bool dummy>
158
+ void compact_theta_sketch_parser<dummy>::check_v4_num_entries_bytes(uint8_t num_entries_bytes) {
159
+ if (num_entries_bytes == 0 || num_entries_bytes > sizeof(uint32_t)) {
160
+ throw std::invalid_argument("num entries bytes must be in [1, 4], actual " + std::to_string(num_entries_bytes));
161
+ }
162
+ }
163
+
143
164
  template<bool dummy>
144
165
  std::string compact_theta_sketch_parser<dummy>::hex_dump(const uint8_t* ptr, size_t size) {
145
166
  std::stringstream s;
@@ -34,10 +34,12 @@ namespace theta_constants {
34
34
 
35
35
  /// max theta - signed max for compatibility with Java
36
36
  const uint64_t MAX_THETA = LLONG_MAX;
37
- /// min log2 of K
38
- const uint8_t MIN_LG_K = 5;
37
+ /// min log2 of nominal entries (K)
38
+ const uint8_t MIN_LG_K = 4;
39
39
  /// max log2 of K
40
40
  const uint8_t MAX_LG_K = 26;
41
+ /// min log2 of cache size
42
+ const uint8_t MIN_LG_ARR = 5;
41
43
  /// default log2 of K
42
44
  const uint8_t DEFAULT_LG_K = 12;
43
45
  }
@@ -20,13 +20,45 @@
20
20
  #ifndef THETA_HELPERS_HPP_
21
21
  #define THETA_HELPERS_HPP_
22
22
 
23
+ #include <algorithm>
24
+ #include <cstdint>
23
25
  #include <stdexcept>
24
26
  #include <string>
27
+ #include <vector>
25
28
 
26
29
  #include "theta_constants.hpp"
30
+ #include "theta_comparators.hpp"
27
31
 
28
32
  namespace datasketches {
29
33
 
34
+ /**
35
+ * Trims a vector of theta entries down to at most nominal_size and returns the theta the
36
+ * result must carry.
37
+ *
38
+ * The entry at index nominal_size becomes the new theta and is itself discarded, so every
39
+ * entry kept is strictly below the returned value. That is what keeps the estimator
40
+ * unbiased: theta must be an exclusive upper bound on the retained hashes. Getting this
41
+ * off by one does not fail loudly, it quietly biases every estimate the sketch produces.
42
+ *
43
+ * Capacity is released as well as size. Callers reserve an upper bound before filling, so
44
+ * without the shrink a trimmed result keeps the untrimmed allocation, which for an update
45
+ * sketch just under the rebuild threshold is nearly twice what it reports.
46
+ *
47
+ * @param entries entries to trim in place; reordered even when nothing is removed
48
+ * @param nominal_size the most entries to keep
49
+ * @param theta returned unchanged when there is nothing to trim
50
+ * @return the theta of the trimmed result
51
+ */
52
+ template<typename ExtractKey, typename Entry, typename Allocator>
53
+ static uint64_t trim_to_nominal(std::vector<Entry, Allocator>& entries, uint32_t nominal_size, uint64_t theta) {
54
+ if (entries.size() <= nominal_size) return theta;
55
+ std::nth_element(entries.begin(), entries.begin() + nominal_size, entries.end(), compare_by_key<ExtractKey>());
56
+ const uint64_t new_theta = ExtractKey()(entries[nominal_size]);
57
+ entries.erase(entries.begin() + nominal_size, entries.end());
58
+ entries.shrink_to_fit();
59
+ return new_theta;
60
+ }
61
+
30
62
  template<typename T>
31
63
  static void check_value(T actual, T expected, const char* description) {
32
64
  if (actual != expected) {
@@ -39,7 +39,9 @@ template<typename FwdSketch, typename Sketch>
39
39
  CS theta_set_difference_base<EN, EK, CS, A>::compute(FwdSketch&& a, const Sketch& b, bool ordered) const {
40
40
  if (a.is_empty() || (a.get_num_retained() > 0 && b.is_empty())) return CS(a, ordered);
41
41
  if (a.get_seed_hash() != seed_hash_) throw std::invalid_argument("A seed hash mismatch");
42
- if (b.get_seed_hash() != seed_hash_) throw std::invalid_argument("B seed hash mismatch");
42
+ // an empty sketch has no hashes, so its seed hash is meaningless and must be ignored,
43
+ // consistent with deserialization, theta_union::update() and theta_intersection::update()
44
+ if (!b.is_empty() && b.get_seed_hash() != seed_hash_) throw std::invalid_argument("B seed hash mismatch");
43
45
 
44
46
  const uint64_t theta = std::min(a.get_theta64(), b.get_theta64());
45
47
  std::vector<EN, A> entries(allocator_);
@@ -69,7 +71,7 @@ CS theta_set_difference_base<EN, EK, CS, A>::compute(FwdSketch&& a, const Sketch
69
71
  const uint64_t hash = EK()(entry);
70
72
  if (hash < theta) {
71
73
  auto result = table.find(hash);
72
- if (!result.second) entries.push_back(conditional_forward<FwdSketch>(entry));
74
+ if (!result.second) entries.emplace_back(conditional_forward<FwdSketch>(entry));
73
75
  } else if (a.is_ordered()) {
74
76
  break; // early stop
75
77
  }
@@ -330,11 +330,25 @@ public:
330
330
  void reset();
331
331
 
332
332
  /**
333
- * Converts this sketch to a compact sketch (ordered or unordered).
333
+ * Converts this sketch to a compact sketch (ordered or unordered, trimmed or not trimmed).
334
+ * This does not modify the source update sketch.
334
335
  * @param ordered optional flag to specify if an ordered sketch should be produced
336
+ * @param trim optional flag to reduce the size of the returned sketch to at most
337
+ * the nominal size k, if required. An update sketch retains more than k entries
338
+ * between rebuilds; those extra entries below theta improve the estimate, so the
339
+ * default is to keep them.
340
+ * Trimming is lossy and is never required for correctness. It discards retained
341
+ * entries, and since the relative error scales with 1 / sqrt(retained), it always
342
+ * degrades accuracy and widens the confidence bounds, whatever mode the source is
343
+ * in. Worst case, a sketch grown to just under the rebuild threshold of 15/16 * 2k
344
+ * loses nearly half its entries, widening the bounds by about sqrt(15/8), or
345
+ * roughly 37%. A sketch in exact mode that retains more than k entries loses
346
+ * exactness as well: it is returned in estimation mode, so get_estimate() carries
347
+ * error where it would otherwise have returned an exact count.
348
+ * Only pass true if a bounded result size matters more than that accuracy.
335
349
  * @return compact sketch
336
350
  */
337
- compact_theta_sketch_alloc<Allocator> compact(bool ordered = true) const;
351
+ compact_theta_sketch_alloc<Allocator> compact(bool ordered = true, bool trim = false) const;
338
352
 
339
353
  virtual iterator begin();
340
354
  virtual iterator end();
@@ -492,7 +506,7 @@ public:
492
506
  uint64_t seed = DEFAULT_SEED, const Allocator& allocator = Allocator());
493
507
 
494
508
  private:
495
- enum flags { IS_BIG_ENDIAN, IS_READ_ONLY, IS_EMPTY, IS_COMPACT, IS_ORDERED };
509
+ enum flags { IS_BIG_ENDIAN, IS_READ_ONLY, IS_EMPTY, IS_COMPACT, IS_ORDERED, IS_SINGLE_ITEM };
496
510
 
497
511
  bool is_empty_;
498
512
  bool is_ordered_;
@@ -501,6 +515,7 @@ private:
501
515
  std::vector<uint64_t, Allocator> entries_;
502
516
 
503
517
  uint8_t get_preamble_longs(bool compressed) const;
518
+ bool is_single_item() const;
504
519
  bool is_suitable_for_compression() const;
505
520
  uint8_t compute_entry_bits() const;
506
521
  uint8_t get_num_entries_bytes() const;
@@ -518,6 +533,7 @@ private:
518
533
  template<typename E, typename EK, typename P, typename S, typename CS, typename A> friend class theta_union_base;
519
534
  template<typename E, typename EK, typename P, typename S, typename CS, typename A> friend class theta_intersection_base;
520
535
  template<typename E, typename EK, typename CS, typename A> friend class theta_set_difference_base;
536
+ template<typename A> friend class update_theta_sketch_alloc;
521
537
  compact_theta_sketch_alloc(bool is_empty, bool is_ordered, uint16_t seed_hash, uint64_t theta, std::vector<uint64_t, Allocator>&& entries);
522
538
  };
523
539
 
@@ -609,9 +625,11 @@ private:
609
625
  uint32_t index_;
610
626
  uint64_t previous_;
611
627
  bool is_block_mode_;
612
- uint8_t buf_i_;
613
628
  uint8_t offset_;
614
629
  uint64_t buffer_[8];
630
+
631
+ inline void unpack1();
632
+ inline void unpack8();
615
633
  };
616
634
 
617
635
  } /* namespace datasketches */