datasketches 0.5.2 → 0.5.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +4 -0
- data/ext/datasketches/theta_wrapper.cpp +1 -1
- data/lib/datasketches/version.rb +1 -1
- data/vendor/datasketches-cpp/CMakeLists.txt +5 -3
- data/vendor/datasketches-cpp/CODE_OF_CONDUCT.md +1 -1
- data/vendor/datasketches-cpp/LICENSE +14 -0
- data/vendor/datasketches-cpp/README.md +67 -73
- data/vendor/datasketches-cpp/benchmarks/CMakeLists.txt +52 -0
- data/vendor/datasketches-cpp/benchmarks/benchmark_count_min_sketch.cpp +153 -0
- data/vendor/datasketches-cpp/benchmarks/benchmark_count_min_sketch_serialization.cpp +161 -0
- data/vendor/datasketches-cpp/common/CMakeLists.txt +1 -0
- data/vendor/datasketches-cpp/common/include/binomial_bounds.hpp +2 -2
- data/vendor/datasketches-cpp/common/include/fdlibm_log.hpp +101 -0
- data/vendor/datasketches-cpp/common/include/serde.hpp +6 -0
- data/vendor/datasketches-cpp/common/test/CMakeLists.txt +44 -2
- data/vendor/datasketches-cpp/common/test/binomial_bounds_test.cpp +279 -0
- data/vendor/datasketches-cpp/common/test/deserialize_hardening_test.cpp +188 -0
- data/vendor/datasketches-cpp/count/include/count_min.hpp +17 -4
- data/vendor/datasketches-cpp/count/include/count_min_impl.hpp +65 -83
- data/vendor/datasketches-cpp/count/test/count_min_test.cpp +63 -7
- data/vendor/datasketches-cpp/cpc/include/compression_data.hpp +2 -0
- data/vendor/datasketches-cpp/cpc/include/cpc_compressor_impl.hpp +16 -7
- data/vendor/datasketches-cpp/cpc/include/cpc_sketch.hpp +1 -0
- data/vendor/datasketches-cpp/cpc/include/cpc_sketch_impl.hpp +54 -36
- data/vendor/datasketches-cpp/cpc/include/cpc_union_impl.hpp +27 -27
- data/vendor/datasketches-cpp/cpc/include/cpc_util.hpp +7 -7
- data/vendor/datasketches-cpp/cpc/include/icon_estimator.hpp +5 -5
- data/vendor/datasketches-cpp/cpc/include/u32_table_impl.hpp +16 -16
- data/vendor/datasketches-cpp/cpc/test/cpc_sketch_test.cpp +35 -0
- data/vendor/datasketches-cpp/fi/include/frequent_items_sketch.hpp +28 -3
- data/vendor/datasketches-cpp/fi/include/frequent_items_sketch_impl.hpp +41 -25
- data/vendor/datasketches-cpp/fi/include/reverse_purge_hash_map.hpp +3 -1
- data/vendor/datasketches-cpp/fi/include/reverse_purge_hash_map_impl.hpp +10 -5
- data/vendor/datasketches-cpp/fi/test/frequent_items_sketch_serialize_for_java.cpp +24 -0
- data/vendor/datasketches-cpp/fi/test/frequent_items_sketch_test.cpp +116 -0
- data/vendor/datasketches-cpp/filters/include/bloom_filter.hpp +1 -1
- data/vendor/datasketches-cpp/filters/include/bloom_filter_impl.hpp +32 -12
- data/vendor/datasketches-cpp/filters/test/bloom_filter_test.cpp +28 -1
- data/vendor/datasketches-cpp/hll/include/CouponHashSet-internal.hpp +1 -2
- data/vendor/datasketches-cpp/hll/include/CouponList-internal.hpp +19 -5
- data/vendor/datasketches-cpp/hll/include/CubicInterpolation-internal.hpp +4 -4
- data/vendor/datasketches-cpp/hll/include/HarmonicNumbers-internal.hpp +2 -1
- data/vendor/datasketches-cpp/hll/include/Hll4Array-internal.hpp +4 -4
- data/vendor/datasketches-cpp/hll/include/Hll4Array.hpp +1 -1
- data/vendor/datasketches-cpp/hll/include/Hll6Array-internal.hpp +3 -3
- data/vendor/datasketches-cpp/hll/include/Hll6Array.hpp +1 -1
- data/vendor/datasketches-cpp/hll/include/Hll8Array-internal.hpp +6 -3
- data/vendor/datasketches-cpp/hll/include/Hll8Array.hpp +1 -1
- data/vendor/datasketches-cpp/hll/include/HllArray-internal.hpp +35 -29
- data/vendor/datasketches-cpp/hll/include/HllArray.hpp +1 -1
- data/vendor/datasketches-cpp/hll/include/HllSketch-internal.hpp +3 -5
- data/vendor/datasketches-cpp/hll/include/HllSketchImpl-internal.hpp +5 -11
- data/vendor/datasketches-cpp/hll/include/HllSketchImpl.hpp +2 -4
- data/vendor/datasketches-cpp/hll/include/HllSketchImplFactory.hpp +11 -10
- data/vendor/datasketches-cpp/hll/include/HllUnion-internal.hpp +17 -8
- data/vendor/datasketches-cpp/hll/include/HllUtil.hpp +12 -3
- data/vendor/datasketches-cpp/hll/include/coupon_iterator-internal.hpp +2 -2
- data/vendor/datasketches-cpp/hll/include/coupon_iterator.hpp +3 -0
- data/vendor/datasketches-cpp/hll/include/hll.hpp +11 -4
- data/vendor/datasketches-cpp/hll/include/hll.private.hpp +19 -0
- data/vendor/datasketches-cpp/hll/test/CMakeLists.txt +2 -0
- data/vendor/datasketches-cpp/hll/test/CouponListTest.cpp +64 -0
- data/vendor/datasketches-cpp/hll/test/HllFullSizeTest.cpp +137 -0
- data/vendor/datasketches-cpp/hll/test/HllKxqRebuildTest.cpp +150 -0
- data/vendor/datasketches-cpp/hll/test/HllSketchTest.cpp +3 -3
- data/vendor/datasketches-cpp/hll/test/HllUnionTest.cpp +71 -1
- data/vendor/datasketches-cpp/kll/include/kll_helper_impl.hpp +10 -10
- data/vendor/datasketches-cpp/kll/include/kll_sketch.hpp +10 -1
- data/vendor/datasketches-cpp/kll/include/kll_sketch_impl.hpp +33 -24
- data/vendor/datasketches-cpp/kll/test/kll_sketch_deserialize_from_java_test.cpp +24 -0
- data/vendor/datasketches-cpp/kll/test/kll_sketch_serialize_for_java.cpp +10 -0
- data/vendor/datasketches-cpp/quantiles/include/quantiles_sketch.hpp +11 -2
- data/vendor/datasketches-cpp/quantiles/include/quantiles_sketch_impl.hpp +24 -15
- data/vendor/datasketches-cpp/req/include/req_sketch.hpp +9 -0
- data/vendor/datasketches-cpp/req/include/req_sketch_impl.hpp +24 -15
- data/vendor/datasketches-cpp/req/test/req_sketch_deserialize_from_java_test.cpp +46 -0
- data/vendor/datasketches-cpp/req/test/req_sketch_serialize_for_java.cpp +20 -0
- data/vendor/datasketches-cpp/req/test/req_sketch_test.cpp +70 -0
- data/vendor/datasketches-cpp/sampling/include/ebpps_sample_impl.hpp +17 -8
- data/vendor/datasketches-cpp/sampling/include/ebpps_sketch.hpp +13 -0
- data/vendor/datasketches-cpp/sampling/include/var_opt_sketch.hpp +10 -1
- data/vendor/datasketches-cpp/sampling/include/var_opt_sketch_impl.hpp +6 -7
- data/vendor/datasketches-cpp/sampling/include/var_opt_union.hpp +5 -1
- data/vendor/datasketches-cpp/sampling/include/var_opt_union_impl.hpp +2 -1
- data/vendor/datasketches-cpp/sampling/test/ebpps_allocation_test.cpp +1 -1
- data/vendor/datasketches-cpp/sampling/test/ebpps_sketch_test.cpp +2 -2
- data/vendor/datasketches-cpp/sampling/test/var_opt_allocation_test.cpp +1 -1
- data/vendor/datasketches-cpp/sampling/test/var_opt_sketch_test.cpp +10 -4
- data/vendor/datasketches-cpp/sampling/test/var_opt_union_test.cpp +12 -0
- data/vendor/datasketches-cpp/tdigest/include/tdigest.hpp +38 -2
- data/vendor/datasketches-cpp/tdigest/include/tdigest_impl.hpp +168 -9
- data/vendor/datasketches-cpp/tdigest/test/CMakeLists.txt +1 -0
- data/vendor/datasketches-cpp/tdigest/test/tdigest_iterator_test.cpp +274 -0
- data/vendor/datasketches-cpp/tdigest/test/tdigest_test.cpp +275 -0
- data/vendor/datasketches-cpp/theta/include/compact_theta_sketch_parser.hpp +2 -0
- data/vendor/datasketches-cpp/theta/include/compact_theta_sketch_parser_impl.hpp +24 -3
- data/vendor/datasketches-cpp/theta/include/theta_constants.hpp +4 -2
- data/vendor/datasketches-cpp/theta/include/theta_helpers.hpp +32 -0
- data/vendor/datasketches-cpp/theta/include/theta_set_difference_base_impl.hpp +4 -2
- data/vendor/datasketches-cpp/theta/include/theta_sketch.hpp +22 -4
- data/vendor/datasketches-cpp/theta/include/theta_sketch_impl.hpp +60 -38
- data/vendor/datasketches-cpp/theta/include/theta_union_base_impl.hpp +2 -6
- data/vendor/datasketches-cpp/theta/include/theta_update_sketch_base_impl.hpp +2 -2
- data/vendor/datasketches-cpp/theta/test/bit_packing_test.cpp +50 -0
- data/vendor/datasketches-cpp/theta/test/theta_a_not_b_test.cpp +22 -0
- data/vendor/datasketches-cpp/theta/test/theta_sketch_test.cpp +315 -0
- data/vendor/datasketches-cpp/tools/rat-check.sh +68 -0
- data/vendor/datasketches-cpp/tuple/include/array_tuple_sketch.hpp +35 -4
- data/vendor/datasketches-cpp/tuple/include/array_tuple_sketch_impl.hpp +2 -2
- data/vendor/datasketches-cpp/tuple/include/tuple_sketch.hpp +41 -0
- data/vendor/datasketches-cpp/tuple/include/tuple_sketch_impl.hpp +5 -4
- data/vendor/datasketches-cpp/tuple/test/tuple_sketch_test.cpp +59 -0
- data/vendor/datasketches-cpp/version.cfg.in +1 -1
- metadata +12 -2
|
@@ -18,13 +18,40 @@
|
|
|
18
18
|
*/
|
|
19
19
|
|
|
20
20
|
#include <catch2/catch.hpp>
|
|
21
|
+
#include <cmath>
|
|
22
|
+
#include <cstdint>
|
|
23
|
+
#include <cstring>
|
|
24
|
+
#include <initializer_list>
|
|
21
25
|
#include <iostream>
|
|
22
26
|
#include <fstream>
|
|
27
|
+
#include <sstream>
|
|
28
|
+
#include <utility>
|
|
23
29
|
|
|
24
30
|
#include "tdigest.hpp"
|
|
25
31
|
|
|
26
32
|
namespace datasketches {
|
|
27
33
|
|
|
34
|
+
namespace {
|
|
35
|
+
constexpr size_t header_size = 8;
|
|
36
|
+
constexpr size_t counts_size = 8;
|
|
37
|
+
constexpr size_t min_offset = header_size + counts_size;
|
|
38
|
+
constexpr size_t max_offset = min_offset + sizeof(double);
|
|
39
|
+
constexpr size_t first_centroid_mean_offset = min_offset + sizeof(double) * 2;
|
|
40
|
+
constexpr size_t first_centroid_weight_offset = first_centroid_mean_offset + sizeof(double);
|
|
41
|
+
constexpr size_t first_buffered_value_offset = first_centroid_mean_offset;
|
|
42
|
+
constexpr size_t single_value_offset = header_size;
|
|
43
|
+
|
|
44
|
+
template <typename T>
|
|
45
|
+
void write_bytes(std::vector<uint8_t>& bytes, size_t offset, T value) {
|
|
46
|
+
std::memcpy(bytes.data() + offset, &value, sizeof(T));
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
template <typename T>
|
|
50
|
+
void write_bytes(std::string& data, size_t offset, T value) {
|
|
51
|
+
std::memcpy(&data[offset], &value, sizeof(T));
|
|
52
|
+
}
|
|
53
|
+
} // namespace
|
|
54
|
+
|
|
28
55
|
TEST_CASE("empty", "[tdigest]") {
|
|
29
56
|
tdigest_double td(10);
|
|
30
57
|
// std::cout << td.to_string();
|
|
@@ -145,6 +172,31 @@ TEST_CASE("merge small", "[tdigest]") {
|
|
|
145
172
|
REQUIRE(td1.get_rank(3.01) == 1);
|
|
146
173
|
}
|
|
147
174
|
|
|
175
|
+
TEST_CASE("merge preserves deserialized min max with weighted tails", "[tdigest]") {
|
|
176
|
+
tdigest_double source(100);
|
|
177
|
+
source.update(0);
|
|
178
|
+
source.update(50);
|
|
179
|
+
source.update(90);
|
|
180
|
+
auto bytes = source.serialize();
|
|
181
|
+
write_bytes(bytes, min_offset, -1.0);
|
|
182
|
+
write_bytes(bytes, max_offset, 100.0);
|
|
183
|
+
auto other = tdigest_double::deserialize(bytes.data(), bytes.size());
|
|
184
|
+
REQUIRE(other.get_min_value() == -1.0);
|
|
185
|
+
REQUIRE(other.get_max_value() == 100.0);
|
|
186
|
+
|
|
187
|
+
tdigest_double empty(100);
|
|
188
|
+
empty.merge(other);
|
|
189
|
+
REQUIRE(empty.get_min_value() == -1.0);
|
|
190
|
+
REQUIRE(empty.get_max_value() == 100.0);
|
|
191
|
+
|
|
192
|
+
tdigest_double left(100);
|
|
193
|
+
left.update(10);
|
|
194
|
+
left.update(20);
|
|
195
|
+
left.merge(other);
|
|
196
|
+
REQUIRE(left.get_min_value() == -1.0);
|
|
197
|
+
REQUIRE(left.get_max_value() == 100.0);
|
|
198
|
+
}
|
|
199
|
+
|
|
148
200
|
TEST_CASE("merge large", "[tdigest]") {
|
|
149
201
|
const size_t n = 10000;
|
|
150
202
|
tdigest_double td1;
|
|
@@ -453,4 +505,227 @@ TEST_CASE("deserialize from reference implementation bytes float", "[tdigest]")
|
|
|
453
505
|
REQUIRE(td.get_rank(n) == 1);
|
|
454
506
|
}
|
|
455
507
|
|
|
508
|
+
TEST_CASE("iterate centroids", "[tdigest]") {
|
|
509
|
+
tdigest_double td(100);
|
|
510
|
+
for (int i = 0; i < 10; i++) {
|
|
511
|
+
td.update(i);
|
|
512
|
+
}
|
|
513
|
+
|
|
514
|
+
auto centroid_count = 0;
|
|
515
|
+
uint64_t total_weight = 0;
|
|
516
|
+
for (const auto ¢roid: td) {
|
|
517
|
+
centroid_count++;
|
|
518
|
+
total_weight += centroid.second;
|
|
519
|
+
}
|
|
520
|
+
// Ensure that centroids are retrieved for a case where there is buffered values
|
|
521
|
+
REQUIRE(centroid_count == 10);
|
|
522
|
+
REQUIRE(td.get_total_weight() == total_weight);
|
|
523
|
+
}
|
|
524
|
+
|
|
525
|
+
TEST_CASE("update rejects positive infinity", "[tdigest]") {
|
|
526
|
+
tdigest_double td(100);
|
|
527
|
+
td.update(1.0);
|
|
528
|
+
td.update(2.0);
|
|
529
|
+
td.update(std::numeric_limits<double>::infinity());
|
|
530
|
+
REQUIRE(td.get_total_weight() == 2);
|
|
531
|
+
REQUIRE(td.get_max_value() == 2.0);
|
|
532
|
+
}
|
|
533
|
+
|
|
534
|
+
TEST_CASE("update rejects negative infinity", "[tdigest]") {
|
|
535
|
+
tdigest_double td(100);
|
|
536
|
+
td.update(1.0);
|
|
537
|
+
td.update(2.0);
|
|
538
|
+
td.update(-std::numeric_limits<double>::infinity());
|
|
539
|
+
REQUIRE(td.get_total_weight() == 2);
|
|
540
|
+
REQUIRE(td.get_min_value() == 1.0);
|
|
541
|
+
}
|
|
542
|
+
|
|
543
|
+
TEST_CASE("deserialize bytes rejects NaN single value", "[tdigest]") {
|
|
544
|
+
tdigest_double td(100);
|
|
545
|
+
td.update(1.0);
|
|
546
|
+
auto bytes = td.serialize();
|
|
547
|
+
write_bytes(bytes, single_value_offset, std::numeric_limits<double>::quiet_NaN());
|
|
548
|
+
REQUIRE_THROWS_AS(tdigest_double::deserialize(bytes.data(), bytes.size()), std::invalid_argument);
|
|
549
|
+
}
|
|
550
|
+
|
|
551
|
+
TEST_CASE("deserialize stream rejects infinity min", "[tdigest]") {
|
|
552
|
+
tdigest_double td(100);
|
|
553
|
+
td.update(1.0);
|
|
554
|
+
td.update(2.0);
|
|
555
|
+
td.update(3.0);
|
|
556
|
+
auto bytes = td.serialize();
|
|
557
|
+
std::string data(reinterpret_cast<const char*>(bytes.data()), bytes.size());
|
|
558
|
+
write_bytes(data, min_offset, std::numeric_limits<double>::infinity());
|
|
559
|
+
std::istringstream is(data, std::ios::binary);
|
|
560
|
+
REQUIRE_THROWS_AS(tdigest_double::deserialize(is), std::invalid_argument);
|
|
561
|
+
}
|
|
562
|
+
|
|
563
|
+
TEST_CASE("deserialize bytes rejects NaN centroid mean", "[tdigest]") {
|
|
564
|
+
tdigest_double td(100);
|
|
565
|
+
for (int i = 0; i < 10; ++i) td.update(i);
|
|
566
|
+
auto bytes = td.serialize();
|
|
567
|
+
write_bytes(bytes, first_centroid_mean_offset, std::numeric_limits<double>::quiet_NaN());
|
|
568
|
+
REQUIRE_THROWS_AS(tdigest_double::deserialize(bytes.data(), bytes.size()), std::invalid_argument);
|
|
569
|
+
}
|
|
570
|
+
|
|
571
|
+
TEST_CASE("deserialize bytes rejects NaN buffered value", "[tdigest]") {
|
|
572
|
+
tdigest_double td(100);
|
|
573
|
+
td.update(1.0);
|
|
574
|
+
td.update(2.0);
|
|
575
|
+
auto bytes = td.serialize(0, true);
|
|
576
|
+
write_bytes(bytes, first_buffered_value_offset, std::numeric_limits<double>::quiet_NaN());
|
|
577
|
+
REQUIRE_THROWS_AS(tdigest_double::deserialize(bytes.data(), bytes.size()), std::invalid_argument);
|
|
578
|
+
}
|
|
579
|
+
|
|
580
|
+
TEST_CASE("deserialize bytes rejects infinity single value", "[tdigest]") {
|
|
581
|
+
tdigest_double td(100);
|
|
582
|
+
td.update(1.0);
|
|
583
|
+
auto bytes = td.serialize();
|
|
584
|
+
write_bytes(bytes, single_value_offset, std::numeric_limits<double>::infinity());
|
|
585
|
+
REQUIRE_THROWS_AS(tdigest_double::deserialize(bytes.data(), bytes.size()), std::invalid_argument);
|
|
586
|
+
}
|
|
587
|
+
|
|
588
|
+
TEST_CASE("deserialize bytes rejects NaN max", "[tdigest]") {
|
|
589
|
+
tdigest_double td(100);
|
|
590
|
+
td.update(1.0);
|
|
591
|
+
td.update(2.0);
|
|
592
|
+
auto bytes = td.serialize();
|
|
593
|
+
write_bytes(bytes, max_offset, std::numeric_limits<double>::quiet_NaN());
|
|
594
|
+
REQUIRE_THROWS_AS(tdigest_double::deserialize(bytes.data(), bytes.size()), std::invalid_argument);
|
|
595
|
+
}
|
|
596
|
+
|
|
597
|
+
TEST_CASE("deserialize bytes rejects infinity max", "[tdigest]") {
|
|
598
|
+
tdigest_double td(100);
|
|
599
|
+
td.update(1.0);
|
|
600
|
+
td.update(2.0);
|
|
601
|
+
auto bytes = td.serialize();
|
|
602
|
+
write_bytes(bytes, max_offset, std::numeric_limits<double>::infinity());
|
|
603
|
+
REQUIRE_THROWS_AS(tdigest_double::deserialize(bytes.data(), bytes.size()), std::invalid_argument);
|
|
604
|
+
}
|
|
605
|
+
|
|
606
|
+
TEST_CASE("deserialize bytes rejects infinity buffered value", "[tdigest]") {
|
|
607
|
+
tdigest_double td(100);
|
|
608
|
+
td.update(1.0);
|
|
609
|
+
td.update(2.0);
|
|
610
|
+
auto bytes = td.serialize(0, true);
|
|
611
|
+
write_bytes(bytes, first_buffered_value_offset, std::numeric_limits<double>::infinity());
|
|
612
|
+
REQUIRE_THROWS_AS(tdigest_double::deserialize(bytes.data(), bytes.size()), std::invalid_argument);
|
|
613
|
+
}
|
|
614
|
+
|
|
615
|
+
TEST_CASE("deserialize bytes rejects zero centroid weight", "[tdigest]") {
|
|
616
|
+
tdigest_double td(100);
|
|
617
|
+
for (int i = 0; i < 10; ++i) td.update(i);
|
|
618
|
+
auto bytes = td.serialize();
|
|
619
|
+
write_bytes(bytes, first_centroid_weight_offset, static_cast<uint64_t>(0));
|
|
620
|
+
REQUIRE_THROWS_AS(tdigest_double::deserialize(bytes.data(), bytes.size()), std::invalid_argument);
|
|
621
|
+
}
|
|
622
|
+
|
|
623
|
+
TEST_CASE("deserialize stream rejects zero centroid weight", "[tdigest]") {
|
|
624
|
+
tdigest_double td(100);
|
|
625
|
+
for (int i = 0; i < 10; ++i) td.update(i);
|
|
626
|
+
auto bytes = td.serialize();
|
|
627
|
+
std::string data(reinterpret_cast<const char*>(bytes.data()), bytes.size());
|
|
628
|
+
write_bytes(data, first_centroid_weight_offset, static_cast<uint64_t>(0));
|
|
629
|
+
std::istringstream is(data, std::ios::binary);
|
|
630
|
+
REQUIRE_THROWS_AS(tdigest_double::deserialize(is), std::invalid_argument);
|
|
631
|
+
}
|
|
632
|
+
|
|
633
|
+
TEST_CASE("quantiles are monotonic and stay within min and max", "[tdigest]") {
|
|
634
|
+
tdigest_double td(100);
|
|
635
|
+
for (int i = 0; i < 10000; ++i) td.update(i);
|
|
636
|
+
double previous = td.get_min_value();
|
|
637
|
+
for (int i = 0; i <= 1000; ++i) {
|
|
638
|
+
const double quantile = td.get_quantile(i / 1000.0);
|
|
639
|
+
REQUIRE(quantile >= previous);
|
|
640
|
+
REQUIRE(quantile >= td.get_min_value());
|
|
641
|
+
REQUIRE(quantile <= td.get_max_value());
|
|
642
|
+
previous = quantile;
|
|
643
|
+
}
|
|
644
|
+
}
|
|
645
|
+
|
|
646
|
+
TEST_CASE("rank below the first centroid stays normalized", "[tdigest]") {
|
|
647
|
+
// The format allows a first centroid heavier than 1. The left tail of get_rank()
|
|
648
|
+
// must be divided by the total weight, the same way the right tail is.
|
|
649
|
+
tdigest_double source(200);
|
|
650
|
+
for (int i = 0; i < 1000; ++i) source.update(i);
|
|
651
|
+
auto bytes = source.serialize();
|
|
652
|
+
double first_mean = 0;
|
|
653
|
+
std::memcpy(&first_mean, bytes.data() + first_centroid_mean_offset, sizeof(double));
|
|
654
|
+
REQUIRE(first_mean == 0);
|
|
655
|
+
write_bytes(bytes, min_offset, -1.0);
|
|
656
|
+
write_bytes(bytes, first_centroid_weight_offset, static_cast<uint64_t>(100));
|
|
657
|
+
const auto td = tdigest_double::deserialize(bytes.data(), bytes.size());
|
|
658
|
+
const double total_weight = static_cast<double>(td.get_total_weight());
|
|
659
|
+
REQUIRE(td.get_rank(-1) == 0.5 / total_weight);
|
|
660
|
+
REQUIRE(td.get_rank(-0.5) == (1.0 + ((100.0 / 2.0 - 1.0) * 0.5)) / total_weight);
|
|
661
|
+
double previous = 0;
|
|
662
|
+
for (int i = 0; i <= 100; ++i) {
|
|
663
|
+
const double rank = td.get_rank(-1.0 + (i / 100.0));
|
|
664
|
+
REQUIRE(rank >= 0);
|
|
665
|
+
REQUIRE(rank <= 1);
|
|
666
|
+
REQUIRE(rank >= previous);
|
|
667
|
+
previous = rank;
|
|
668
|
+
}
|
|
669
|
+
}
|
|
670
|
+
|
|
671
|
+
TEST_CASE("quantile above the last centroid does not exceed max", "[tdigest]") {
|
|
672
|
+
tdigest_double source(200);
|
|
673
|
+
for (int i = 0; i < 1000; ++i) source.update(i);
|
|
674
|
+
auto bytes = source.serialize();
|
|
675
|
+
uint32_t num_centroids = 0;
|
|
676
|
+
std::memcpy(&num_centroids, bytes.data() + header_size, sizeof(uint32_t));
|
|
677
|
+
REQUIRE(num_centroids > 1);
|
|
678
|
+
const size_t last_weight_offset = first_centroid_weight_offset + (num_centroids - 1) * 16;
|
|
679
|
+
write_bytes(bytes, last_weight_offset, static_cast<uint64_t>(100));
|
|
680
|
+
const auto td = tdigest_double::deserialize(bytes.data(), bytes.size());
|
|
681
|
+
double previous = td.get_min_value();
|
|
682
|
+
for (int i = 0; i <= 1000; ++i) {
|
|
683
|
+
const double quantile = td.get_quantile(i / 1000.0);
|
|
684
|
+
REQUIRE(quantile >= previous);
|
|
685
|
+
REQUIRE(quantile <= td.get_max_value());
|
|
686
|
+
previous = quantile;
|
|
687
|
+
}
|
|
688
|
+
}
|
|
689
|
+
|
|
690
|
+
tdigest_double sketch_from_centroids(double min, double max,
|
|
691
|
+
std::initializer_list<std::pair<double, uint64_t>> centroids) {
|
|
692
|
+
const auto num_centroids = static_cast<uint32_t>(centroids.size());
|
|
693
|
+
std::vector<uint8_t> bytes(first_centroid_mean_offset + static_cast<size_t>(num_centroids) * 16, 0);
|
|
694
|
+
bytes[0] = 2; // preamble longs
|
|
695
|
+
bytes[1] = 1; // serial version
|
|
696
|
+
bytes[2] = 20; // sketch type
|
|
697
|
+
write_bytes(bytes, 3, static_cast<uint16_t>(100));
|
|
698
|
+
write_bytes(bytes, header_size, num_centroids);
|
|
699
|
+
write_bytes(bytes, header_size + sizeof(uint32_t), static_cast<uint32_t>(0));
|
|
700
|
+
write_bytes(bytes, min_offset, min);
|
|
701
|
+
write_bytes(bytes, max_offset, max);
|
|
702
|
+
size_t offset = first_centroid_mean_offset;
|
|
703
|
+
for (const auto& centroid : centroids) {
|
|
704
|
+
write_bytes(bytes, offset, centroid.first);
|
|
705
|
+
write_bytes(bytes, offset + sizeof(double), centroid.second);
|
|
706
|
+
offset += 16;
|
|
707
|
+
}
|
|
708
|
+
return tdigest_double::deserialize(bytes.data(), bytes.size());
|
|
709
|
+
}
|
|
710
|
+
|
|
711
|
+
TEST_CASE("quantile with a last centroid of weight 2 returns max", "[tdigest]") {
|
|
712
|
+
// weight == total - 1 and last weight == 2 is 0/0 unless that case returns max.
|
|
713
|
+
const auto td = sketch_from_centroids(0, 20, {{0, 10}, {10, 2}});
|
|
714
|
+
const double quantile = td.get_quantile(11.0 / 12.0);
|
|
715
|
+
REQUIRE_FALSE(std::isnan(quantile));
|
|
716
|
+
REQUIRE(quantile == 20);
|
|
717
|
+
}
|
|
718
|
+
|
|
719
|
+
TEST_CASE("quantile right tail approaches max from below", "[tdigest]") {
|
|
720
|
+
const auto td = sketch_from_centroids(0, 40, {{10, 100}, {20, 100}, {30, 100}});
|
|
721
|
+
REQUIRE(td.get_quantile(0.9) == Approx(34.081632653061224).epsilon(1e-12));
|
|
722
|
+
}
|
|
723
|
+
|
|
724
|
+
TEST_CASE("quantile interpolation weights the nearer centroid more", "[tdigest]") {
|
|
725
|
+
const auto td = sketch_from_centroids(0, 40, {{10, 100}, {20, 100}, {30, 100}});
|
|
726
|
+
// Target weight 80 sits between 10 and 20, closer to 10.
|
|
727
|
+
// weighted_average(10, 70, 20, 30) == 13; the swapped weights would return 17.
|
|
728
|
+
REQUIRE(td.get_quantile(80.0 / 300.0) == 13);
|
|
729
|
+
}
|
|
730
|
+
|
|
456
731
|
} /* namespace datasketches */
|
|
@@ -38,6 +38,8 @@ public:
|
|
|
38
38
|
};
|
|
39
39
|
|
|
40
40
|
static compact_theta_sketch_data parse(const void* ptr, size_t size, uint64_t seed, bool dump_on_error = false);
|
|
41
|
+
static void check_v4_entry_bits(uint8_t entry_bits);
|
|
42
|
+
static void check_v4_num_entries_bytes(uint8_t num_entries_bytes);
|
|
41
43
|
|
|
42
44
|
private:
|
|
43
45
|
// offsets are in sizeof(type)
|
|
@@ -49,6 +49,7 @@ auto compact_theta_sketch_parser<dummy>::parse(const void* ptr, size_t size, uin
|
|
|
49
49
|
theta = reinterpret_cast<const uint64_t*>(ptr)[COMPACT_SKETCH_V4_THETA_U64];
|
|
50
50
|
}
|
|
51
51
|
const uint8_t num_entries_bytes = reinterpret_cast<const uint8_t*>(ptr)[COMPACT_SKETCH_V4_NUM_ENTRIES_BYTES_BYTE];
|
|
52
|
+
check_v4_num_entries_bytes(num_entries_bytes);
|
|
52
53
|
size_t data_offset_bytes = has_theta ? COMPACT_SKETCH_V4_PACKED_DATA_ESTIMATION_BYTE : COMPACT_SKETCH_V4_PACKED_DATA_EXACT_BYTE;
|
|
53
54
|
check_memory_size(ptr, size, data_offset_bytes + num_entries_bytes, dump_on_error);
|
|
54
55
|
uint32_t num_entries = 0;
|
|
@@ -58,7 +59,8 @@ auto compact_theta_sketch_parser<dummy>::parse(const void* ptr, size_t size, uin
|
|
|
58
59
|
}
|
|
59
60
|
data_offset_bytes += num_entries_bytes;
|
|
60
61
|
const uint8_t entry_bits = reinterpret_cast<const uint8_t*>(ptr)[COMPACT_SKETCH_V4_ENTRY_BITS_BYTE];
|
|
61
|
-
|
|
62
|
+
check_v4_entry_bits(entry_bits);
|
|
63
|
+
const uint64_t expected_bits = static_cast<uint64_t>(entry_bits) * num_entries;
|
|
62
64
|
const size_t expected_size_bytes = data_offset_bytes + whole_bytes_to_hold_bits(expected_bits);
|
|
63
65
|
check_memory_size(ptr, size, expected_size_bytes, dump_on_error);
|
|
64
66
|
return {false, true, seed_hash, num_entries, theta,
|
|
@@ -80,8 +82,9 @@ auto compact_theta_sketch_parser<dummy>::parse(const void* ptr, size_t size, uin
|
|
|
80
82
|
check_memory_size(ptr, size, 16, dump_on_error);
|
|
81
83
|
return {false, true, seed_hash, 1, theta, reinterpret_cast<const uint64_t*>(ptr) + COMPACT_SKETCH_SINGLE_ENTRY_U64, 64};
|
|
82
84
|
}
|
|
83
|
-
const uint32_t num_entries = reinterpret_cast<const uint32_t*>(ptr)[COMPACT_SKETCH_NUM_ENTRIES_U32];
|
|
84
85
|
const size_t entries_start_u64 = has_theta ? COMPACT_SKETCH_ENTRIES_ESTIMATION_U64 : COMPACT_SKETCH_ENTRIES_EXACT_U64;
|
|
86
|
+
check_memory_size(ptr, size, entries_start_u64 * sizeof(uint64_t), dump_on_error);
|
|
87
|
+
const uint32_t num_entries = reinterpret_cast<const uint32_t*>(ptr)[COMPACT_SKETCH_NUM_ENTRIES_U32];
|
|
85
88
|
const uint64_t* entries = reinterpret_cast<const uint64_t*>(ptr) + entries_start_u64;
|
|
86
89
|
const size_t expected_size_bytes = (entries_start_u64 + num_entries) * sizeof(uint64_t);
|
|
87
90
|
check_memory_size(ptr, size, expected_size_bytes, dump_on_error);
|
|
@@ -90,6 +93,7 @@ auto compact_theta_sketch_parser<dummy>::parse(const void* ptr, size_t size, uin
|
|
|
90
93
|
}
|
|
91
94
|
case 1: {
|
|
92
95
|
uint16_t seed_hash = compute_seed_hash(seed);
|
|
96
|
+
check_memory_size(ptr, size, COMPACT_SKETCH_ENTRIES_ESTIMATION_U64 * sizeof(uint64_t), dump_on_error);
|
|
93
97
|
const uint32_t num_entries = reinterpret_cast<const uint32_t*>(ptr)[COMPACT_SKETCH_NUM_ENTRIES_U32];
|
|
94
98
|
uint64_t theta = reinterpret_cast<const uint64_t*>(ptr)[COMPACT_SKETCH_THETA_U64];
|
|
95
99
|
bool is_empty = (num_entries == 0) && (theta == theta_constants::MAX_THETA);
|
|
@@ -106,16 +110,18 @@ auto compact_theta_sketch_parser<dummy>::parse(const void* ptr, size_t size, uin
|
|
|
106
110
|
if (preamble_size == 1) {
|
|
107
111
|
return {true, true, seed_hash, 0, theta_constants::MAX_THETA, nullptr, 64};
|
|
108
112
|
} else if (preamble_size == 2) {
|
|
113
|
+
check_memory_size(ptr, size, COMPACT_SKETCH_ENTRIES_EXACT_U64 * sizeof(uint64_t), dump_on_error);
|
|
109
114
|
const uint32_t num_entries = reinterpret_cast<const uint32_t*>(ptr)[COMPACT_SKETCH_NUM_ENTRIES_U32];
|
|
110
115
|
if (num_entries == 0) {
|
|
111
116
|
return {true, true, seed_hash, 0, theta_constants::MAX_THETA, nullptr, 64};
|
|
112
117
|
} else {
|
|
113
|
-
const size_t expected_size_bytes = (preamble_size + num_entries) << 3;
|
|
118
|
+
const size_t expected_size_bytes = (preamble_size + static_cast<size_t>(num_entries)) << 3;
|
|
114
119
|
check_memory_size(ptr, size, expected_size_bytes, dump_on_error);
|
|
115
120
|
const uint64_t* entries = reinterpret_cast<const uint64_t*>(ptr) + COMPACT_SKETCH_ENTRIES_EXACT_U64;
|
|
116
121
|
return {false, true, seed_hash, num_entries, theta_constants::MAX_THETA, entries, 64};
|
|
117
122
|
}
|
|
118
123
|
} else if (preamble_size == 3) {
|
|
124
|
+
check_memory_size(ptr, size, COMPACT_SKETCH_ENTRIES_ESTIMATION_U64 * sizeof(uint64_t), dump_on_error);
|
|
119
125
|
const uint32_t num_entries = reinterpret_cast<const uint32_t*>(ptr)[COMPACT_SKETCH_NUM_ENTRIES_U32];
|
|
120
126
|
uint64_t theta = reinterpret_cast<const uint64_t*>(ptr)[COMPACT_SKETCH_THETA_U64];
|
|
121
127
|
bool is_empty = (num_entries == 0) && (theta == theta_constants::MAX_THETA);
|
|
@@ -140,6 +146,21 @@ void compact_theta_sketch_parser<dummy>::check_memory_size(const void* ptr, size
|
|
|
140
146
|
+ (dump_on_error ? (", sketch dump: " + hex_dump(reinterpret_cast<const uint8_t*>(ptr), actual_bytes)) : ""));
|
|
141
147
|
}
|
|
142
148
|
|
|
149
|
+
template<bool dummy>
|
|
150
|
+
void compact_theta_sketch_parser<dummy>::check_v4_entry_bits(uint8_t entry_bits) {
|
|
151
|
+
// deltas between ordered hashes below 2^63 need 1 to 63 bits
|
|
152
|
+
if (entry_bits == 0 || entry_bits > 63) {
|
|
153
|
+
throw std::invalid_argument("entry bits must be in [1, 63], actual " + std::to_string(entry_bits));
|
|
154
|
+
}
|
|
155
|
+
}
|
|
156
|
+
|
|
157
|
+
template<bool dummy>
|
|
158
|
+
void compact_theta_sketch_parser<dummy>::check_v4_num_entries_bytes(uint8_t num_entries_bytes) {
|
|
159
|
+
if (num_entries_bytes == 0 || num_entries_bytes > sizeof(uint32_t)) {
|
|
160
|
+
throw std::invalid_argument("num entries bytes must be in [1, 4], actual " + std::to_string(num_entries_bytes));
|
|
161
|
+
}
|
|
162
|
+
}
|
|
163
|
+
|
|
143
164
|
template<bool dummy>
|
|
144
165
|
std::string compact_theta_sketch_parser<dummy>::hex_dump(const uint8_t* ptr, size_t size) {
|
|
145
166
|
std::stringstream s;
|
|
@@ -34,10 +34,12 @@ namespace theta_constants {
|
|
|
34
34
|
|
|
35
35
|
/// max theta - signed max for compatibility with Java
|
|
36
36
|
const uint64_t MAX_THETA = LLONG_MAX;
|
|
37
|
-
/// min log2 of K
|
|
38
|
-
const uint8_t MIN_LG_K =
|
|
37
|
+
/// min log2 of nominal entries (K)
|
|
38
|
+
const uint8_t MIN_LG_K = 4;
|
|
39
39
|
/// max log2 of K
|
|
40
40
|
const uint8_t MAX_LG_K = 26;
|
|
41
|
+
/// min log2 of cache size
|
|
42
|
+
const uint8_t MIN_LG_ARR = 5;
|
|
41
43
|
/// default log2 of K
|
|
42
44
|
const uint8_t DEFAULT_LG_K = 12;
|
|
43
45
|
}
|
|
@@ -20,13 +20,45 @@
|
|
|
20
20
|
#ifndef THETA_HELPERS_HPP_
|
|
21
21
|
#define THETA_HELPERS_HPP_
|
|
22
22
|
|
|
23
|
+
#include <algorithm>
|
|
24
|
+
#include <cstdint>
|
|
23
25
|
#include <stdexcept>
|
|
24
26
|
#include <string>
|
|
27
|
+
#include <vector>
|
|
25
28
|
|
|
26
29
|
#include "theta_constants.hpp"
|
|
30
|
+
#include "theta_comparators.hpp"
|
|
27
31
|
|
|
28
32
|
namespace datasketches {
|
|
29
33
|
|
|
34
|
+
/**
|
|
35
|
+
* Trims a vector of theta entries down to at most nominal_size and returns the theta the
|
|
36
|
+
* result must carry.
|
|
37
|
+
*
|
|
38
|
+
* The entry at index nominal_size becomes the new theta and is itself discarded, so every
|
|
39
|
+
* entry kept is strictly below the returned value. That is what keeps the estimator
|
|
40
|
+
* unbiased: theta must be an exclusive upper bound on the retained hashes. Getting this
|
|
41
|
+
* off by one does not fail loudly, it quietly biases every estimate the sketch produces.
|
|
42
|
+
*
|
|
43
|
+
* Capacity is released as well as size. Callers reserve an upper bound before filling, so
|
|
44
|
+
* without the shrink a trimmed result keeps the untrimmed allocation, which for an update
|
|
45
|
+
* sketch just under the rebuild threshold is nearly twice what it reports.
|
|
46
|
+
*
|
|
47
|
+
* @param entries entries to trim in place; reordered even when nothing is removed
|
|
48
|
+
* @param nominal_size the most entries to keep
|
|
49
|
+
* @param theta returned unchanged when there is nothing to trim
|
|
50
|
+
* @return the theta of the trimmed result
|
|
51
|
+
*/
|
|
52
|
+
template<typename ExtractKey, typename Entry, typename Allocator>
|
|
53
|
+
static uint64_t trim_to_nominal(std::vector<Entry, Allocator>& entries, uint32_t nominal_size, uint64_t theta) {
|
|
54
|
+
if (entries.size() <= nominal_size) return theta;
|
|
55
|
+
std::nth_element(entries.begin(), entries.begin() + nominal_size, entries.end(), compare_by_key<ExtractKey>());
|
|
56
|
+
const uint64_t new_theta = ExtractKey()(entries[nominal_size]);
|
|
57
|
+
entries.erase(entries.begin() + nominal_size, entries.end());
|
|
58
|
+
entries.shrink_to_fit();
|
|
59
|
+
return new_theta;
|
|
60
|
+
}
|
|
61
|
+
|
|
30
62
|
template<typename T>
|
|
31
63
|
static void check_value(T actual, T expected, const char* description) {
|
|
32
64
|
if (actual != expected) {
|
|
@@ -39,7 +39,9 @@ template<typename FwdSketch, typename Sketch>
|
|
|
39
39
|
CS theta_set_difference_base<EN, EK, CS, A>::compute(FwdSketch&& a, const Sketch& b, bool ordered) const {
|
|
40
40
|
if (a.is_empty() || (a.get_num_retained() > 0 && b.is_empty())) return CS(a, ordered);
|
|
41
41
|
if (a.get_seed_hash() != seed_hash_) throw std::invalid_argument("A seed hash mismatch");
|
|
42
|
-
|
|
42
|
+
// an empty sketch has no hashes, so its seed hash is meaningless and must be ignored,
|
|
43
|
+
// consistent with deserialization, theta_union::update() and theta_intersection::update()
|
|
44
|
+
if (!b.is_empty() && b.get_seed_hash() != seed_hash_) throw std::invalid_argument("B seed hash mismatch");
|
|
43
45
|
|
|
44
46
|
const uint64_t theta = std::min(a.get_theta64(), b.get_theta64());
|
|
45
47
|
std::vector<EN, A> entries(allocator_);
|
|
@@ -69,7 +71,7 @@ CS theta_set_difference_base<EN, EK, CS, A>::compute(FwdSketch&& a, const Sketch
|
|
|
69
71
|
const uint64_t hash = EK()(entry);
|
|
70
72
|
if (hash < theta) {
|
|
71
73
|
auto result = table.find(hash);
|
|
72
|
-
if (!result.second) entries.
|
|
74
|
+
if (!result.second) entries.emplace_back(conditional_forward<FwdSketch>(entry));
|
|
73
75
|
} else if (a.is_ordered()) {
|
|
74
76
|
break; // early stop
|
|
75
77
|
}
|
|
@@ -330,11 +330,25 @@ public:
|
|
|
330
330
|
void reset();
|
|
331
331
|
|
|
332
332
|
/**
|
|
333
|
-
* Converts this sketch to a compact sketch (ordered or unordered).
|
|
333
|
+
* Converts this sketch to a compact sketch (ordered or unordered, trimmed or not trimmed).
|
|
334
|
+
* This does not modify the source update sketch.
|
|
334
335
|
* @param ordered optional flag to specify if an ordered sketch should be produced
|
|
336
|
+
* @param trim optional flag to reduce the size of the returned sketch to at most
|
|
337
|
+
* the nominal size k, if required. An update sketch retains more than k entries
|
|
338
|
+
* between rebuilds; those extra entries below theta improve the estimate, so the
|
|
339
|
+
* default is to keep them.
|
|
340
|
+
* Trimming is lossy and is never required for correctness. It discards retained
|
|
341
|
+
* entries, and since the relative error scales with 1 / sqrt(retained), it always
|
|
342
|
+
* degrades accuracy and widens the confidence bounds, whatever mode the source is
|
|
343
|
+
* in. Worst case, a sketch grown to just under the rebuild threshold of 15/16 * 2k
|
|
344
|
+
* loses nearly half its entries, widening the bounds by about sqrt(15/8), or
|
|
345
|
+
* roughly 37%. A sketch in exact mode that retains more than k entries loses
|
|
346
|
+
* exactness as well: it is returned in estimation mode, so get_estimate() carries
|
|
347
|
+
* error where it would otherwise have returned an exact count.
|
|
348
|
+
* Only pass true if a bounded result size matters more than that accuracy.
|
|
335
349
|
* @return compact sketch
|
|
336
350
|
*/
|
|
337
|
-
compact_theta_sketch_alloc<Allocator> compact(bool ordered = true) const;
|
|
351
|
+
compact_theta_sketch_alloc<Allocator> compact(bool ordered = true, bool trim = false) const;
|
|
338
352
|
|
|
339
353
|
virtual iterator begin();
|
|
340
354
|
virtual iterator end();
|
|
@@ -492,7 +506,7 @@ public:
|
|
|
492
506
|
uint64_t seed = DEFAULT_SEED, const Allocator& allocator = Allocator());
|
|
493
507
|
|
|
494
508
|
private:
|
|
495
|
-
enum flags { IS_BIG_ENDIAN, IS_READ_ONLY, IS_EMPTY, IS_COMPACT, IS_ORDERED };
|
|
509
|
+
enum flags { IS_BIG_ENDIAN, IS_READ_ONLY, IS_EMPTY, IS_COMPACT, IS_ORDERED, IS_SINGLE_ITEM };
|
|
496
510
|
|
|
497
511
|
bool is_empty_;
|
|
498
512
|
bool is_ordered_;
|
|
@@ -501,6 +515,7 @@ private:
|
|
|
501
515
|
std::vector<uint64_t, Allocator> entries_;
|
|
502
516
|
|
|
503
517
|
uint8_t get_preamble_longs(bool compressed) const;
|
|
518
|
+
bool is_single_item() const;
|
|
504
519
|
bool is_suitable_for_compression() const;
|
|
505
520
|
uint8_t compute_entry_bits() const;
|
|
506
521
|
uint8_t get_num_entries_bytes() const;
|
|
@@ -518,6 +533,7 @@ private:
|
|
|
518
533
|
template<typename E, typename EK, typename P, typename S, typename CS, typename A> friend class theta_union_base;
|
|
519
534
|
template<typename E, typename EK, typename P, typename S, typename CS, typename A> friend class theta_intersection_base;
|
|
520
535
|
template<typename E, typename EK, typename CS, typename A> friend class theta_set_difference_base;
|
|
536
|
+
template<typename A> friend class update_theta_sketch_alloc;
|
|
521
537
|
compact_theta_sketch_alloc(bool is_empty, bool is_ordered, uint16_t seed_hash, uint64_t theta, std::vector<uint64_t, Allocator>&& entries);
|
|
522
538
|
};
|
|
523
539
|
|
|
@@ -609,9 +625,11 @@ private:
|
|
|
609
625
|
uint32_t index_;
|
|
610
626
|
uint64_t previous_;
|
|
611
627
|
bool is_block_mode_;
|
|
612
|
-
uint8_t buf_i_;
|
|
613
628
|
uint8_t offset_;
|
|
614
629
|
uint64_t buffer_[8];
|
|
630
|
+
|
|
631
|
+
inline void unpack1();
|
|
632
|
+
inline void unpack8();
|
|
615
633
|
};
|
|
616
634
|
|
|
617
635
|
} /* namespace datasketches */
|