faiss 0.2.0 → 0.2.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +16 -0
- data/LICENSE.txt +1 -1
- data/README.md +7 -7
- data/ext/faiss/extconf.rb +6 -3
- data/ext/faiss/numo.hpp +4 -4
- data/ext/faiss/utils.cpp +1 -1
- data/ext/faiss/utils.h +1 -1
- data/lib/faiss/version.rb +1 -1
- data/vendor/faiss/faiss/AutoTune.cpp +292 -291
- data/vendor/faiss/faiss/AutoTune.h +55 -56
- data/vendor/faiss/faiss/Clustering.cpp +365 -194
- data/vendor/faiss/faiss/Clustering.h +102 -35
- data/vendor/faiss/faiss/IVFlib.cpp +171 -195
- data/vendor/faiss/faiss/IVFlib.h +48 -51
- data/vendor/faiss/faiss/Index.cpp +85 -103
- data/vendor/faiss/faiss/Index.h +54 -48
- data/vendor/faiss/faiss/Index2Layer.cpp +126 -224
- data/vendor/faiss/faiss/Index2Layer.h +22 -36
- data/vendor/faiss/faiss/IndexAdditiveQuantizer.cpp +407 -0
- data/vendor/faiss/faiss/IndexAdditiveQuantizer.h +195 -0
- data/vendor/faiss/faiss/IndexBinary.cpp +45 -37
- data/vendor/faiss/faiss/IndexBinary.h +140 -132
- data/vendor/faiss/faiss/IndexBinaryFlat.cpp +73 -53
- data/vendor/faiss/faiss/IndexBinaryFlat.h +29 -24
- data/vendor/faiss/faiss/IndexBinaryFromFloat.cpp +46 -43
- data/vendor/faiss/faiss/IndexBinaryFromFloat.h +16 -15
- data/vendor/faiss/faiss/IndexBinaryHNSW.cpp +215 -232
- data/vendor/faiss/faiss/IndexBinaryHNSW.h +25 -24
- data/vendor/faiss/faiss/IndexBinaryHash.cpp +182 -177
- data/vendor/faiss/faiss/IndexBinaryHash.h +41 -34
- data/vendor/faiss/faiss/IndexBinaryIVF.cpp +489 -461
- data/vendor/faiss/faiss/IndexBinaryIVF.h +97 -68
- data/vendor/faiss/faiss/IndexFlat.cpp +115 -176
- data/vendor/faiss/faiss/IndexFlat.h +42 -59
- data/vendor/faiss/faiss/IndexFlatCodes.cpp +67 -0
- data/vendor/faiss/faiss/IndexFlatCodes.h +47 -0
- data/vendor/faiss/faiss/IndexHNSW.cpp +372 -348
- data/vendor/faiss/faiss/IndexHNSW.h +57 -41
- data/vendor/faiss/faiss/IndexIVF.cpp +545 -453
- data/vendor/faiss/faiss/IndexIVF.h +169 -118
- data/vendor/faiss/faiss/IndexIVFAdditiveQuantizer.cpp +316 -0
- data/vendor/faiss/faiss/IndexIVFAdditiveQuantizer.h +121 -0
- data/vendor/faiss/faiss/IndexIVFFlat.cpp +247 -252
- data/vendor/faiss/faiss/IndexIVFFlat.h +48 -51
- data/vendor/faiss/faiss/IndexIVFPQ.cpp +459 -517
- data/vendor/faiss/faiss/IndexIVFPQ.h +75 -67
- data/vendor/faiss/faiss/IndexIVFPQFastScan.cpp +406 -372
- data/vendor/faiss/faiss/IndexIVFPQFastScan.h +82 -57
- data/vendor/faiss/faiss/IndexIVFPQR.cpp +104 -102
- data/vendor/faiss/faiss/IndexIVFPQR.h +33 -28
- data/vendor/faiss/faiss/IndexIVFSpectralHash.cpp +163 -150
- data/vendor/faiss/faiss/IndexIVFSpectralHash.h +38 -25
- data/vendor/faiss/faiss/IndexLSH.cpp +66 -113
- data/vendor/faiss/faiss/IndexLSH.h +20 -38
- data/vendor/faiss/faiss/IndexLattice.cpp +42 -56
- data/vendor/faiss/faiss/IndexLattice.h +11 -16
- data/vendor/faiss/faiss/IndexNNDescent.cpp +229 -0
- data/vendor/faiss/faiss/IndexNNDescent.h +72 -0
- data/vendor/faiss/faiss/IndexNSG.cpp +301 -0
- data/vendor/faiss/faiss/IndexNSG.h +85 -0
- data/vendor/faiss/faiss/IndexPQ.cpp +387 -495
- data/vendor/faiss/faiss/IndexPQ.h +64 -82
- data/vendor/faiss/faiss/IndexPQFastScan.cpp +143 -170
- data/vendor/faiss/faiss/IndexPQFastScan.h +46 -32
- data/vendor/faiss/faiss/IndexPreTransform.cpp +120 -150
- data/vendor/faiss/faiss/IndexPreTransform.h +33 -36
- data/vendor/faiss/faiss/IndexRefine.cpp +139 -127
- data/vendor/faiss/faiss/IndexRefine.h +32 -23
- data/vendor/faiss/faiss/IndexReplicas.cpp +147 -153
- data/vendor/faiss/faiss/IndexReplicas.h +62 -56
- data/vendor/faiss/faiss/IndexScalarQuantizer.cpp +111 -172
- data/vendor/faiss/faiss/IndexScalarQuantizer.h +41 -59
- data/vendor/faiss/faiss/IndexShards.cpp +256 -240
- data/vendor/faiss/faiss/IndexShards.h +85 -73
- data/vendor/faiss/faiss/MatrixStats.cpp +112 -97
- data/vendor/faiss/faiss/MatrixStats.h +7 -10
- data/vendor/faiss/faiss/MetaIndexes.cpp +135 -157
- data/vendor/faiss/faiss/MetaIndexes.h +40 -34
- data/vendor/faiss/faiss/MetricType.h +7 -7
- data/vendor/faiss/faiss/VectorTransform.cpp +654 -475
- data/vendor/faiss/faiss/VectorTransform.h +64 -89
- data/vendor/faiss/faiss/clone_index.cpp +78 -73
- data/vendor/faiss/faiss/clone_index.h +4 -9
- data/vendor/faiss/faiss/gpu/GpuAutoTune.cpp +33 -38
- data/vendor/faiss/faiss/gpu/GpuAutoTune.h +11 -9
- data/vendor/faiss/faiss/gpu/GpuCloner.cpp +198 -171
- data/vendor/faiss/faiss/gpu/GpuCloner.h +53 -35
- data/vendor/faiss/faiss/gpu/GpuClonerOptions.cpp +12 -14
- data/vendor/faiss/faiss/gpu/GpuClonerOptions.h +27 -25
- data/vendor/faiss/faiss/gpu/GpuDistance.h +116 -112
- data/vendor/faiss/faiss/gpu/GpuFaissAssert.h +1 -2
- data/vendor/faiss/faiss/gpu/GpuIcmEncoder.h +60 -0
- data/vendor/faiss/faiss/gpu/GpuIndex.h +134 -137
- data/vendor/faiss/faiss/gpu/GpuIndexBinaryFlat.h +76 -73
- data/vendor/faiss/faiss/gpu/GpuIndexFlat.h +173 -162
- data/vendor/faiss/faiss/gpu/GpuIndexIVF.h +67 -64
- data/vendor/faiss/faiss/gpu/GpuIndexIVFFlat.h +89 -86
- data/vendor/faiss/faiss/gpu/GpuIndexIVFPQ.h +150 -141
- data/vendor/faiss/faiss/gpu/GpuIndexIVFScalarQuantizer.h +101 -103
- data/vendor/faiss/faiss/gpu/GpuIndicesOptions.h +17 -16
- data/vendor/faiss/faiss/gpu/GpuResources.cpp +116 -128
- data/vendor/faiss/faiss/gpu/GpuResources.h +182 -186
- data/vendor/faiss/faiss/gpu/StandardGpuResources.cpp +433 -422
- data/vendor/faiss/faiss/gpu/StandardGpuResources.h +131 -130
- data/vendor/faiss/faiss/gpu/impl/InterleavedCodes.cpp +468 -456
- data/vendor/faiss/faiss/gpu/impl/InterleavedCodes.h +25 -19
- data/vendor/faiss/faiss/gpu/impl/RemapIndices.cpp +22 -20
- data/vendor/faiss/faiss/gpu/impl/RemapIndices.h +9 -8
- data/vendor/faiss/faiss/gpu/perf/IndexWrapper-inl.h +39 -44
- data/vendor/faiss/faiss/gpu/perf/IndexWrapper.h +16 -14
- data/vendor/faiss/faiss/gpu/perf/PerfClustering.cpp +77 -71
- data/vendor/faiss/faiss/gpu/perf/PerfIVFPQAdd.cpp +109 -88
- data/vendor/faiss/faiss/gpu/perf/WriteIndex.cpp +75 -64
- data/vendor/faiss/faiss/gpu/test/TestCodePacking.cpp +230 -215
- data/vendor/faiss/faiss/gpu/test/TestGpuIndexBinaryFlat.cpp +80 -86
- data/vendor/faiss/faiss/gpu/test/TestGpuIndexFlat.cpp +284 -277
- data/vendor/faiss/faiss/gpu/test/TestGpuIndexIVFFlat.cpp +416 -416
- data/vendor/faiss/faiss/gpu/test/TestGpuIndexIVFPQ.cpp +611 -517
- data/vendor/faiss/faiss/gpu/test/TestGpuIndexIVFScalarQuantizer.cpp +166 -164
- data/vendor/faiss/faiss/gpu/test/TestGpuMemoryException.cpp +61 -53
- data/vendor/faiss/faiss/gpu/test/TestUtils.cpp +274 -238
- data/vendor/faiss/faiss/gpu/test/TestUtils.h +73 -57
- data/vendor/faiss/faiss/gpu/test/demo_ivfpq_indexing_gpu.cpp +47 -50
- data/vendor/faiss/faiss/gpu/utils/DeviceUtils.h +79 -72
- data/vendor/faiss/faiss/gpu/utils/StackDeviceMemory.cpp +140 -146
- data/vendor/faiss/faiss/gpu/utils/StackDeviceMemory.h +69 -71
- data/vendor/faiss/faiss/gpu/utils/StaticUtils.h +21 -16
- data/vendor/faiss/faiss/gpu/utils/Timer.cpp +25 -29
- data/vendor/faiss/faiss/gpu/utils/Timer.h +30 -29
- data/vendor/faiss/faiss/impl/AdditiveQuantizer.cpp +503 -0
- data/vendor/faiss/faiss/impl/AdditiveQuantizer.h +175 -0
- data/vendor/faiss/faiss/impl/AuxIndexStructures.cpp +90 -120
- data/vendor/faiss/faiss/impl/AuxIndexStructures.h +81 -65
- data/vendor/faiss/faiss/impl/FaissAssert.h +73 -58
- data/vendor/faiss/faiss/impl/FaissException.cpp +56 -48
- data/vendor/faiss/faiss/impl/FaissException.h +41 -29
- data/vendor/faiss/faiss/impl/HNSW.cpp +606 -617
- data/vendor/faiss/faiss/impl/HNSW.h +179 -200
- data/vendor/faiss/faiss/impl/LocalSearchQuantizer.cpp +855 -0
- data/vendor/faiss/faiss/impl/LocalSearchQuantizer.h +244 -0
- data/vendor/faiss/faiss/impl/NNDescent.cpp +487 -0
- data/vendor/faiss/faiss/impl/NNDescent.h +154 -0
- data/vendor/faiss/faiss/impl/NSG.cpp +679 -0
- data/vendor/faiss/faiss/impl/NSG.h +199 -0
- data/vendor/faiss/faiss/impl/PolysemousTraining.cpp +484 -454
- data/vendor/faiss/faiss/impl/PolysemousTraining.h +52 -55
- data/vendor/faiss/faiss/impl/ProductQuantizer-inl.h +26 -47
- data/vendor/faiss/faiss/impl/ProductQuantizer.cpp +469 -459
- data/vendor/faiss/faiss/impl/ProductQuantizer.h +76 -87
- data/vendor/faiss/faiss/impl/ResidualQuantizer.cpp +758 -0
- data/vendor/faiss/faiss/impl/ResidualQuantizer.h +188 -0
- data/vendor/faiss/faiss/impl/ResultHandler.h +96 -132
- data/vendor/faiss/faiss/impl/ScalarQuantizer.cpp +647 -707
- data/vendor/faiss/faiss/impl/ScalarQuantizer.h +48 -46
- data/vendor/faiss/faiss/impl/ThreadedIndex-inl.h +129 -131
- data/vendor/faiss/faiss/impl/ThreadedIndex.h +61 -55
- data/vendor/faiss/faiss/impl/index_read.cpp +631 -480
- data/vendor/faiss/faiss/impl/index_write.cpp +547 -407
- data/vendor/faiss/faiss/impl/io.cpp +76 -95
- data/vendor/faiss/faiss/impl/io.h +31 -41
- data/vendor/faiss/faiss/impl/io_macros.h +60 -29
- data/vendor/faiss/faiss/impl/kmeans1d.cpp +301 -0
- data/vendor/faiss/faiss/impl/kmeans1d.h +48 -0
- data/vendor/faiss/faiss/impl/lattice_Zn.cpp +137 -186
- data/vendor/faiss/faiss/impl/lattice_Zn.h +40 -51
- data/vendor/faiss/faiss/impl/platform_macros.h +29 -8
- data/vendor/faiss/faiss/impl/pq4_fast_scan.cpp +77 -124
- data/vendor/faiss/faiss/impl/pq4_fast_scan.h +39 -48
- data/vendor/faiss/faiss/impl/pq4_fast_scan_search_1.cpp +41 -52
- data/vendor/faiss/faiss/impl/pq4_fast_scan_search_qbs.cpp +80 -117
- data/vendor/faiss/faiss/impl/simd_result_handlers.h +109 -137
- data/vendor/faiss/faiss/index_factory.cpp +619 -397
- data/vendor/faiss/faiss/index_factory.h +8 -6
- data/vendor/faiss/faiss/index_io.h +23 -26
- data/vendor/faiss/faiss/invlists/BlockInvertedLists.cpp +67 -75
- data/vendor/faiss/faiss/invlists/BlockInvertedLists.h +22 -24
- data/vendor/faiss/faiss/invlists/DirectMap.cpp +96 -112
- data/vendor/faiss/faiss/invlists/DirectMap.h +29 -33
- data/vendor/faiss/faiss/invlists/InvertedLists.cpp +307 -364
- data/vendor/faiss/faiss/invlists/InvertedLists.h +151 -151
- data/vendor/faiss/faiss/invlists/InvertedListsIOHook.cpp +29 -34
- data/vendor/faiss/faiss/invlists/InvertedListsIOHook.h +17 -18
- data/vendor/faiss/faiss/invlists/OnDiskInvertedLists.cpp +257 -293
- data/vendor/faiss/faiss/invlists/OnDiskInvertedLists.h +50 -45
- data/vendor/faiss/faiss/python/python_callbacks.cpp +23 -26
- data/vendor/faiss/faiss/python/python_callbacks.h +9 -16
- data/vendor/faiss/faiss/utils/AlignedTable.h +79 -44
- data/vendor/faiss/faiss/utils/Heap.cpp +40 -48
- data/vendor/faiss/faiss/utils/Heap.h +186 -209
- data/vendor/faiss/faiss/utils/WorkerThread.cpp +67 -76
- data/vendor/faiss/faiss/utils/WorkerThread.h +32 -33
- data/vendor/faiss/faiss/utils/distances.cpp +305 -312
- data/vendor/faiss/faiss/utils/distances.h +170 -122
- data/vendor/faiss/faiss/utils/distances_simd.cpp +498 -508
- data/vendor/faiss/faiss/utils/extra_distances-inl.h +117 -0
- data/vendor/faiss/faiss/utils/extra_distances.cpp +113 -232
- data/vendor/faiss/faiss/utils/extra_distances.h +30 -29
- data/vendor/faiss/faiss/utils/hamming-inl.h +260 -209
- data/vendor/faiss/faiss/utils/hamming.cpp +375 -469
- data/vendor/faiss/faiss/utils/hamming.h +62 -85
- data/vendor/faiss/faiss/utils/ordered_key_value.h +16 -18
- data/vendor/faiss/faiss/utils/partitioning.cpp +393 -318
- data/vendor/faiss/faiss/utils/partitioning.h +26 -21
- data/vendor/faiss/faiss/utils/quantize_lut.cpp +78 -66
- data/vendor/faiss/faiss/utils/quantize_lut.h +22 -20
- data/vendor/faiss/faiss/utils/random.cpp +39 -63
- data/vendor/faiss/faiss/utils/random.h +13 -16
- data/vendor/faiss/faiss/utils/simdlib.h +4 -2
- data/vendor/faiss/faiss/utils/simdlib_avx2.h +88 -85
- data/vendor/faiss/faiss/utils/simdlib_emulated.h +226 -165
- data/vendor/faiss/faiss/utils/simdlib_neon.h +832 -0
- data/vendor/faiss/faiss/utils/utils.cpp +304 -287
- data/vendor/faiss/faiss/utils/utils.h +54 -49
- metadata +29 -4
|
@@ -15,15 +15,15 @@
|
|
|
15
15
|
|
|
16
16
|
#include <stdint.h>
|
|
17
17
|
|
|
18
|
-
#include <
|
|
19
|
-
#include <unordered_set>
|
|
18
|
+
#include <cstring>
|
|
20
19
|
#include <memory>
|
|
21
20
|
#include <mutex>
|
|
21
|
+
#include <unordered_set>
|
|
22
|
+
#include <vector>
|
|
22
23
|
|
|
23
24
|
#include <faiss/Index.h>
|
|
24
25
|
#include <faiss/impl/platform_macros.h>
|
|
25
26
|
|
|
26
|
-
|
|
27
27
|
namespace faiss {
|
|
28
28
|
|
|
29
29
|
/** The objective is to have a simple result structure while
|
|
@@ -31,42 +31,39 @@ namespace faiss {
|
|
|
31
31
|
* do_allocation can be overloaded to allocate the result tables in
|
|
32
32
|
* the matrix type of a scripting language like Lua or Python. */
|
|
33
33
|
struct RangeSearchResult {
|
|
34
|
-
size_t nq;
|
|
35
|
-
size_t
|
|
34
|
+
size_t nq; ///< nb of queries
|
|
35
|
+
size_t* lims; ///< size (nq + 1)
|
|
36
36
|
|
|
37
37
|
typedef Index::idx_t idx_t;
|
|
38
38
|
|
|
39
|
-
idx_t
|
|
40
|
-
float
|
|
39
|
+
idx_t* labels; ///< result for query i is labels[lims[i]:lims[i+1]]
|
|
40
|
+
float* distances; ///< corresponding distances (not sorted)
|
|
41
41
|
|
|
42
42
|
size_t buffer_size; ///< size of the result buffers used
|
|
43
43
|
|
|
44
44
|
/// lims must be allocated on input to range_search.
|
|
45
|
-
explicit RangeSearchResult
|
|
45
|
+
explicit RangeSearchResult(idx_t nq, bool alloc_lims = true);
|
|
46
46
|
|
|
47
47
|
/// called when lims contains the nb of elements result entries
|
|
48
48
|
/// for each query
|
|
49
49
|
|
|
50
|
-
virtual void do_allocation
|
|
50
|
+
virtual void do_allocation();
|
|
51
51
|
|
|
52
|
-
virtual ~RangeSearchResult
|
|
52
|
+
virtual ~RangeSearchResult();
|
|
53
53
|
};
|
|
54
54
|
|
|
55
|
-
|
|
56
55
|
/** Encapsulates a set of ids to remove. */
|
|
57
56
|
struct IDSelector {
|
|
58
57
|
typedef Index::idx_t idx_t;
|
|
59
|
-
virtual bool is_member
|
|
58
|
+
virtual bool is_member(idx_t id) const = 0;
|
|
60
59
|
virtual ~IDSelector() {}
|
|
61
60
|
};
|
|
62
61
|
|
|
63
|
-
|
|
64
|
-
|
|
65
62
|
/** remove ids between [imni, imax) */
|
|
66
|
-
struct IDSelectorRange: IDSelector {
|
|
63
|
+
struct IDSelectorRange : IDSelector {
|
|
67
64
|
idx_t imin, imax;
|
|
68
65
|
|
|
69
|
-
IDSelectorRange
|
|
66
|
+
IDSelectorRange(idx_t imin, idx_t imax);
|
|
70
67
|
bool is_member(idx_t id) const override;
|
|
71
68
|
~IDSelectorRange() override {}
|
|
72
69
|
};
|
|
@@ -76,11 +73,11 @@ struct IDSelectorRange: IDSelector {
|
|
|
76
73
|
* this is inefficient in most cases, except for IndexIVF with
|
|
77
74
|
* maintain_direct_map
|
|
78
75
|
*/
|
|
79
|
-
struct IDSelectorArray: IDSelector {
|
|
76
|
+
struct IDSelectorArray : IDSelector {
|
|
80
77
|
size_t n;
|
|
81
|
-
const idx_t
|
|
78
|
+
const idx_t* ids;
|
|
82
79
|
|
|
83
|
-
IDSelectorArray
|
|
80
|
+
IDSelectorArray(size_t n, const idx_t* ids);
|
|
84
81
|
bool is_member(idx_t id) const override;
|
|
85
82
|
~IDSelectorArray() override {}
|
|
86
83
|
};
|
|
@@ -91,8 +88,7 @@ struct IDSelectorArray: IDSelector {
|
|
|
91
88
|
* unordered_set are just the least significant bits of the id. This
|
|
92
89
|
* works fine for random ids or ids in sequences but will produce many
|
|
93
90
|
* hash collisions if lsb's are always the same */
|
|
94
|
-
struct IDSelectorBatch: IDSelector {
|
|
95
|
-
|
|
91
|
+
struct IDSelectorBatch : IDSelector {
|
|
96
92
|
std::unordered_set<idx_t> set;
|
|
97
93
|
|
|
98
94
|
typedef unsigned char uint8_t;
|
|
@@ -100,7 +96,7 @@ struct IDSelectorBatch: IDSelector {
|
|
|
100
96
|
int nbits;
|
|
101
97
|
idx_t mask;
|
|
102
98
|
|
|
103
|
-
IDSelectorBatch
|
|
99
|
+
IDSelectorBatch(size_t n, const idx_t* indices);
|
|
104
100
|
bool is_member(idx_t id) const override;
|
|
105
101
|
~IDSelectorBatch() override {}
|
|
106
102
|
};
|
|
@@ -124,28 +120,26 @@ struct BufferList {
|
|
|
124
120
|
size_t buffer_size;
|
|
125
121
|
|
|
126
122
|
struct Buffer {
|
|
127
|
-
idx_t
|
|
128
|
-
float
|
|
123
|
+
idx_t* ids;
|
|
124
|
+
float* dis;
|
|
129
125
|
};
|
|
130
126
|
|
|
131
127
|
std::vector<Buffer> buffers;
|
|
132
128
|
size_t wp; ///< write pointer in the last buffer.
|
|
133
129
|
|
|
134
|
-
explicit BufferList
|
|
130
|
+
explicit BufferList(size_t buffer_size);
|
|
135
131
|
|
|
136
|
-
~BufferList
|
|
132
|
+
~BufferList();
|
|
137
133
|
|
|
138
134
|
/// create a new buffer
|
|
139
|
-
void append_buffer
|
|
135
|
+
void append_buffer();
|
|
140
136
|
|
|
141
137
|
/// add one result, possibly appending a new buffer if needed
|
|
142
|
-
void add
|
|
138
|
+
void add(idx_t id, float dis);
|
|
143
139
|
|
|
144
140
|
/// copy elemnts ofs:ofs+n-1 seen as linear data in the buffers to
|
|
145
141
|
/// tables dest_ids, dest_dis
|
|
146
|
-
void copy_range
|
|
147
|
-
idx_t * dest_ids, float *dest_dis);
|
|
148
|
-
|
|
142
|
+
void copy_range(size_t ofs, size_t n, idx_t* dest_ids, float* dest_dis);
|
|
149
143
|
};
|
|
150
144
|
|
|
151
145
|
struct RangeSearchPartialResult;
|
|
@@ -153,46 +147,45 @@ struct RangeSearchPartialResult;
|
|
|
153
147
|
/// result structure for a single query
|
|
154
148
|
struct RangeQueryResult {
|
|
155
149
|
using idx_t = Index::idx_t;
|
|
156
|
-
idx_t qno;
|
|
157
|
-
size_t nres;
|
|
158
|
-
RangeSearchPartialResult
|
|
150
|
+
idx_t qno; //< id of the query
|
|
151
|
+
size_t nres; //< nb of results for this query
|
|
152
|
+
RangeSearchPartialResult* pres;
|
|
159
153
|
|
|
160
154
|
/// called by search function to report a new result
|
|
161
|
-
void add
|
|
155
|
+
void add(float dis, idx_t id);
|
|
162
156
|
};
|
|
163
157
|
|
|
164
158
|
/// the entries in the buffers are split per query
|
|
165
|
-
struct RangeSearchPartialResult: BufferList {
|
|
166
|
-
RangeSearchResult
|
|
159
|
+
struct RangeSearchPartialResult : BufferList {
|
|
160
|
+
RangeSearchResult* res;
|
|
167
161
|
|
|
168
162
|
/// eventually the result will be stored in res_in
|
|
169
|
-
explicit RangeSearchPartialResult
|
|
163
|
+
explicit RangeSearchPartialResult(RangeSearchResult* res_in);
|
|
170
164
|
|
|
171
165
|
/// query ids + nb of results per query.
|
|
172
166
|
std::vector<RangeQueryResult> queries;
|
|
173
167
|
|
|
174
168
|
/// begin a new result
|
|
175
|
-
RangeQueryResult
|
|
169
|
+
RangeQueryResult& new_result(idx_t qno);
|
|
176
170
|
|
|
177
171
|
/*****************************************
|
|
178
172
|
* functions used at the end of the search to merge the result
|
|
179
173
|
* lists */
|
|
180
|
-
void finalize
|
|
174
|
+
void finalize();
|
|
181
175
|
|
|
182
176
|
/// called by range_search before do_allocation
|
|
183
|
-
void set_lims
|
|
177
|
+
void set_lims();
|
|
184
178
|
|
|
185
179
|
/// called by range_search after do_allocation
|
|
186
|
-
void copy_result
|
|
180
|
+
void copy_result(bool incremental = false);
|
|
187
181
|
|
|
188
182
|
/// merge a set of PartialResult's into one RangeSearchResult
|
|
189
183
|
/// on ouptut the partialresults are empty!
|
|
190
|
-
static void merge
|
|
191
|
-
|
|
192
|
-
|
|
184
|
+
static void merge(
|
|
185
|
+
std::vector<RangeSearchPartialResult*>& partial_results,
|
|
186
|
+
bool do_delete = true);
|
|
193
187
|
};
|
|
194
188
|
|
|
195
|
-
|
|
196
189
|
/***********************************************************
|
|
197
190
|
* The distance computer maintains a current query and computes
|
|
198
191
|
* distances to elements in an index that supports random access.
|
|
@@ -202,19 +195,19 @@ struct RangeSearchPartialResult: BufferList {
|
|
|
202
195
|
* instantiate one from each thread if needed.
|
|
203
196
|
***********************************************************/
|
|
204
197
|
struct DistanceComputer {
|
|
205
|
-
|
|
198
|
+
using idx_t = Index::idx_t;
|
|
206
199
|
|
|
207
|
-
|
|
208
|
-
|
|
209
|
-
|
|
200
|
+
/// called before computing distances. Pointer x should remain valid
|
|
201
|
+
/// while operator () is called
|
|
202
|
+
virtual void set_query(const float* x) = 0;
|
|
210
203
|
|
|
211
|
-
|
|
212
|
-
|
|
204
|
+
/// compute distance of vector i to current query
|
|
205
|
+
virtual float operator()(idx_t i) = 0;
|
|
213
206
|
|
|
214
|
-
|
|
215
|
-
|
|
207
|
+
/// compute distance between two stored vectors
|
|
208
|
+
virtual float symmetric_dis(idx_t i, idx_t j) = 0;
|
|
216
209
|
|
|
217
|
-
|
|
210
|
+
virtual ~DistanceComputer() {}
|
|
218
211
|
};
|
|
219
212
|
|
|
220
213
|
/***********************************************************
|
|
@@ -222,7 +215,7 @@ struct DistanceComputer {
|
|
|
222
215
|
***********************************************************/
|
|
223
216
|
|
|
224
217
|
struct FAISS_API InterruptCallback {
|
|
225
|
-
virtual bool want_interrupt
|
|
218
|
+
virtual bool want_interrupt() = 0;
|
|
226
219
|
virtual ~InterruptCallback() {}
|
|
227
220
|
|
|
228
221
|
// lock that protects concurrent calls to is_interrupted
|
|
@@ -230,7 +223,7 @@ struct FAISS_API InterruptCallback {
|
|
|
230
223
|
|
|
231
224
|
static std::unique_ptr<InterruptCallback> instance;
|
|
232
225
|
|
|
233
|
-
static void clear_instance
|
|
226
|
+
static void clear_instance();
|
|
234
227
|
|
|
235
228
|
/** check if:
|
|
236
229
|
* - an interrupt callback is set
|
|
@@ -238,23 +231,46 @@ struct FAISS_API InterruptCallback {
|
|
|
238
231
|
* if this is the case, then throw an exception. Should not be called
|
|
239
232
|
* from multiple threads.
|
|
240
233
|
*/
|
|
241
|
-
static void check
|
|
234
|
+
static void check();
|
|
242
235
|
|
|
243
236
|
/// same as check() but return true if is interrupted instead of
|
|
244
237
|
/// throwing. Can be called from multiple threads.
|
|
245
|
-
static bool is_interrupted
|
|
238
|
+
static bool is_interrupted();
|
|
246
239
|
|
|
247
240
|
/** assuming each iteration takes a certain number of flops, what
|
|
248
241
|
* is a reasonable interval to check for interrupts?
|
|
249
242
|
*/
|
|
250
|
-
static size_t get_period_hint
|
|
251
|
-
|
|
243
|
+
static size_t get_period_hint(size_t flops);
|
|
252
244
|
};
|
|
253
245
|
|
|
246
|
+
/// set implementation optimized for fast access.
|
|
247
|
+
struct VisitedTable {
|
|
248
|
+
std::vector<uint8_t> visited;
|
|
249
|
+
int visno;
|
|
250
|
+
|
|
251
|
+
explicit VisitedTable(int size) : visited(size), visno(1) {}
|
|
252
|
+
|
|
253
|
+
/// set flag #no to true
|
|
254
|
+
void set(int no) {
|
|
255
|
+
visited[no] = visno;
|
|
256
|
+
}
|
|
257
|
+
|
|
258
|
+
/// get flag #no
|
|
259
|
+
bool get(int no) const {
|
|
260
|
+
return visited[no] == visno;
|
|
261
|
+
}
|
|
262
|
+
|
|
263
|
+
/// reset all flags to false
|
|
264
|
+
void advance() {
|
|
265
|
+
visno++;
|
|
266
|
+
if (visno == 250) {
|
|
267
|
+
// 250 rather than 255 because sometimes we use visno and visno+1
|
|
268
|
+
memset(visited.data(), 0, sizeof(visited[0]) * visited.size());
|
|
269
|
+
visno = 1;
|
|
270
|
+
}
|
|
271
|
+
}
|
|
272
|
+
};
|
|
254
273
|
|
|
255
|
-
|
|
256
|
-
}; // namespace faiss
|
|
257
|
-
|
|
258
|
-
|
|
274
|
+
} // namespace faiss
|
|
259
275
|
|
|
260
276
|
#endif
|
|
@@ -12,85 +12,100 @@
|
|
|
12
12
|
|
|
13
13
|
#include <faiss/impl/FaissException.h>
|
|
14
14
|
#include <faiss/impl/platform_macros.h>
|
|
15
|
-
#include <cstdlib>
|
|
16
15
|
#include <cstdio>
|
|
16
|
+
#include <cstdlib>
|
|
17
17
|
#include <string>
|
|
18
18
|
|
|
19
19
|
///
|
|
20
20
|
/// Assertions
|
|
21
21
|
///
|
|
22
22
|
|
|
23
|
-
#define FAISS_ASSERT(X)
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
23
|
+
#define FAISS_ASSERT(X) \
|
|
24
|
+
do { \
|
|
25
|
+
if (!(X)) { \
|
|
26
|
+
fprintf(stderr, \
|
|
27
|
+
"Faiss assertion '%s' failed in %s " \
|
|
28
|
+
"at %s:%d\n", \
|
|
29
|
+
#X, \
|
|
30
|
+
__PRETTY_FUNCTION__, \
|
|
31
|
+
__FILE__, \
|
|
32
|
+
__LINE__); \
|
|
33
|
+
abort(); \
|
|
34
|
+
} \
|
|
35
|
+
} while (false)
|
|
32
36
|
|
|
33
|
-
#define FAISS_ASSERT_MSG(X, MSG)
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
37
|
+
#define FAISS_ASSERT_MSG(X, MSG) \
|
|
38
|
+
do { \
|
|
39
|
+
if (!(X)) { \
|
|
40
|
+
fprintf(stderr, \
|
|
41
|
+
"Faiss assertion '%s' failed in %s " \
|
|
42
|
+
"at %s:%d; details: " MSG "\n", \
|
|
43
|
+
#X, \
|
|
44
|
+
__PRETTY_FUNCTION__, \
|
|
45
|
+
__FILE__, \
|
|
46
|
+
__LINE__); \
|
|
47
|
+
abort(); \
|
|
48
|
+
} \
|
|
49
|
+
} while (false)
|
|
42
50
|
|
|
43
|
-
#define FAISS_ASSERT_FMT(X, FMT, ...)
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
51
|
+
#define FAISS_ASSERT_FMT(X, FMT, ...) \
|
|
52
|
+
do { \
|
|
53
|
+
if (!(X)) { \
|
|
54
|
+
fprintf(stderr, \
|
|
55
|
+
"Faiss assertion '%s' failed in %s " \
|
|
56
|
+
"at %s:%d; details: " FMT "\n", \
|
|
57
|
+
#X, \
|
|
58
|
+
__PRETTY_FUNCTION__, \
|
|
59
|
+
__FILE__, \
|
|
60
|
+
__LINE__, \
|
|
61
|
+
__VA_ARGS__); \
|
|
62
|
+
abort(); \
|
|
63
|
+
} \
|
|
64
|
+
} while (false)
|
|
52
65
|
|
|
53
66
|
///
|
|
54
67
|
/// Exceptions for returning user errors
|
|
55
68
|
///
|
|
56
69
|
|
|
57
|
-
#define FAISS_THROW_MSG(MSG)
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
70
|
+
#define FAISS_THROW_MSG(MSG) \
|
|
71
|
+
do { \
|
|
72
|
+
throw faiss::FaissException( \
|
|
73
|
+
MSG, __PRETTY_FUNCTION__, __FILE__, __LINE__); \
|
|
74
|
+
} while (false)
|
|
61
75
|
|
|
62
|
-
#define FAISS_THROW_FMT(FMT, ...)
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
76
|
+
#define FAISS_THROW_FMT(FMT, ...) \
|
|
77
|
+
do { \
|
|
78
|
+
std::string __s; \
|
|
79
|
+
int __size = snprintf(nullptr, 0, FMT, __VA_ARGS__); \
|
|
80
|
+
__s.resize(__size + 1); \
|
|
81
|
+
snprintf(&__s[0], __s.size(), FMT, __VA_ARGS__); \
|
|
82
|
+
throw faiss::FaissException( \
|
|
83
|
+
__s, __PRETTY_FUNCTION__, __FILE__, __LINE__); \
|
|
84
|
+
} while (false)
|
|
70
85
|
|
|
71
86
|
///
|
|
72
87
|
/// Exceptions thrown upon a conditional failure
|
|
73
88
|
///
|
|
74
89
|
|
|
75
|
-
#define FAISS_THROW_IF_NOT(X)
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
90
|
+
#define FAISS_THROW_IF_NOT(X) \
|
|
91
|
+
do { \
|
|
92
|
+
if (!(X)) { \
|
|
93
|
+
FAISS_THROW_FMT("Error: '%s' failed", #X); \
|
|
94
|
+
} \
|
|
95
|
+
} while (false)
|
|
81
96
|
|
|
82
|
-
#define FAISS_THROW_IF_NOT_MSG(X, MSG)
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
97
|
+
#define FAISS_THROW_IF_NOT_MSG(X, MSG) \
|
|
98
|
+
do { \
|
|
99
|
+
if (!(X)) { \
|
|
100
|
+
FAISS_THROW_FMT("Error: '%s' failed: " MSG, #X); \
|
|
101
|
+
} \
|
|
102
|
+
} while (false)
|
|
88
103
|
|
|
89
|
-
#define FAISS_THROW_IF_NOT_FMT(X, FMT, ...)
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
104
|
+
#define FAISS_THROW_IF_NOT_FMT(X, FMT, ...) \
|
|
105
|
+
do { \
|
|
106
|
+
if (!(X)) { \
|
|
107
|
+
FAISS_THROW_FMT("Error: '%s' failed: " FMT, #X, __VA_ARGS__); \
|
|
108
|
+
} \
|
|
109
|
+
} while (false)
|
|
95
110
|
|
|
96
111
|
#endif
|
|
@@ -10,71 +10,81 @@
|
|
|
10
10
|
#include <faiss/impl/FaissException.h>
|
|
11
11
|
#include <sstream>
|
|
12
12
|
|
|
13
|
-
#ifdef
|
|
13
|
+
#ifdef __GNUG__
|
|
14
14
|
#include <cxxabi.h>
|
|
15
15
|
#endif
|
|
16
16
|
|
|
17
17
|
namespace faiss {
|
|
18
18
|
|
|
19
|
-
FaissException::FaissException(const std::string& m)
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
19
|
+
FaissException::FaissException(const std::string& m) : msg(m) {}
|
|
20
|
+
|
|
21
|
+
FaissException::FaissException(
|
|
22
|
+
const std::string& m,
|
|
23
|
+
const char* funcName,
|
|
24
|
+
const char* file,
|
|
25
|
+
int line) {
|
|
26
|
+
int size = snprintf(
|
|
27
|
+
nullptr,
|
|
28
|
+
0,
|
|
29
|
+
"Error in %s at %s:%d: %s",
|
|
30
|
+
funcName,
|
|
31
|
+
file,
|
|
32
|
+
line,
|
|
33
|
+
m.c_str());
|
|
34
|
+
msg.resize(size + 1);
|
|
35
|
+
snprintf(
|
|
36
|
+
&msg[0],
|
|
37
|
+
msg.size(),
|
|
38
|
+
"Error in %s at %s:%d: %s",
|
|
39
|
+
funcName,
|
|
40
|
+
file,
|
|
41
|
+
line,
|
|
42
|
+
m.c_str());
|
|
32
43
|
}
|
|
33
44
|
|
|
34
|
-
const char*
|
|
35
|
-
|
|
36
|
-
return msg.c_str();
|
|
45
|
+
const char* FaissException::what() const noexcept {
|
|
46
|
+
return msg.c_str();
|
|
37
47
|
}
|
|
38
48
|
|
|
39
49
|
void handleExceptions(
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
50
|
+
std::vector<std::pair<int, std::exception_ptr>>& exceptions) {
|
|
51
|
+
if (exceptions.size() == 1) {
|
|
52
|
+
// throw the single received exception directly
|
|
53
|
+
std::rethrow_exception(exceptions.front().second);
|
|
54
|
+
|
|
55
|
+
} else if (exceptions.size() > 1) {
|
|
56
|
+
// multiple exceptions; aggregate them and return a single exception
|
|
57
|
+
std::stringstream ss;
|
|
58
|
+
|
|
59
|
+
for (auto& p : exceptions) {
|
|
60
|
+
try {
|
|
61
|
+
std::rethrow_exception(p.second);
|
|
62
|
+
} catch (std::exception& ex) {
|
|
63
|
+
if (ex.what()) {
|
|
64
|
+
// exception message available
|
|
65
|
+
ss << "Exception thrown from index " << p.first << ": "
|
|
66
|
+
<< ex.what() << "\n";
|
|
67
|
+
} else {
|
|
68
|
+
// No message available
|
|
69
|
+
ss << "Unknown exception thrown from index " << p.first
|
|
70
|
+
<< "\n";
|
|
71
|
+
}
|
|
72
|
+
} catch (...) {
|
|
73
|
+
ss << "Unknown exception thrown from index " << p.first << "\n";
|
|
74
|
+
}
|
|
60
75
|
}
|
|
61
|
-
} catch (...) {
|
|
62
|
-
ss << "Unknown exception thrown from index " << p.first << "\n";
|
|
63
|
-
}
|
|
64
|
-
}
|
|
65
76
|
|
|
66
|
-
|
|
67
|
-
|
|
77
|
+
throw FaissException(ss.str());
|
|
78
|
+
}
|
|
68
79
|
}
|
|
69
80
|
|
|
70
|
-
|
|
71
81
|
// From
|
|
72
82
|
// https://stackoverflow.com/questions/281818/unmangling-the-result-of-stdtype-infoname
|
|
73
83
|
|
|
74
84
|
std::string demangle_cpp_symbol(const char* name) {
|
|
75
85
|
#ifdef __GNUG__
|
|
76
86
|
int status = -1;
|
|
77
|
-
const char
|
|
87
|
+
const char* res = abi::__cxa_demangle(name, nullptr, nullptr, &status);
|
|
78
88
|
std::string sres;
|
|
79
89
|
if (status == 0) {
|
|
80
90
|
sres = res;
|
|
@@ -87,6 +97,4 @@ std::string demangle_cpp_symbol(const char* name) {
|
|
|
87
97
|
#endif
|
|
88
98
|
}
|
|
89
99
|
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
} // namespace
|
|
100
|
+
} // namespace faiss
|