faiss 0.2.0 → 0.2.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (215) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +16 -0
  3. data/LICENSE.txt +1 -1
  4. data/README.md +7 -7
  5. data/ext/faiss/extconf.rb +6 -3
  6. data/ext/faiss/numo.hpp +4 -4
  7. data/ext/faiss/utils.cpp +1 -1
  8. data/ext/faiss/utils.h +1 -1
  9. data/lib/faiss/version.rb +1 -1
  10. data/vendor/faiss/faiss/AutoTune.cpp +292 -291
  11. data/vendor/faiss/faiss/AutoTune.h +55 -56
  12. data/vendor/faiss/faiss/Clustering.cpp +365 -194
  13. data/vendor/faiss/faiss/Clustering.h +102 -35
  14. data/vendor/faiss/faiss/IVFlib.cpp +171 -195
  15. data/vendor/faiss/faiss/IVFlib.h +48 -51
  16. data/vendor/faiss/faiss/Index.cpp +85 -103
  17. data/vendor/faiss/faiss/Index.h +54 -48
  18. data/vendor/faiss/faiss/Index2Layer.cpp +126 -224
  19. data/vendor/faiss/faiss/Index2Layer.h +22 -36
  20. data/vendor/faiss/faiss/IndexAdditiveQuantizer.cpp +407 -0
  21. data/vendor/faiss/faiss/IndexAdditiveQuantizer.h +195 -0
  22. data/vendor/faiss/faiss/IndexBinary.cpp +45 -37
  23. data/vendor/faiss/faiss/IndexBinary.h +140 -132
  24. data/vendor/faiss/faiss/IndexBinaryFlat.cpp +73 -53
  25. data/vendor/faiss/faiss/IndexBinaryFlat.h +29 -24
  26. data/vendor/faiss/faiss/IndexBinaryFromFloat.cpp +46 -43
  27. data/vendor/faiss/faiss/IndexBinaryFromFloat.h +16 -15
  28. data/vendor/faiss/faiss/IndexBinaryHNSW.cpp +215 -232
  29. data/vendor/faiss/faiss/IndexBinaryHNSW.h +25 -24
  30. data/vendor/faiss/faiss/IndexBinaryHash.cpp +182 -177
  31. data/vendor/faiss/faiss/IndexBinaryHash.h +41 -34
  32. data/vendor/faiss/faiss/IndexBinaryIVF.cpp +489 -461
  33. data/vendor/faiss/faiss/IndexBinaryIVF.h +97 -68
  34. data/vendor/faiss/faiss/IndexFlat.cpp +115 -176
  35. data/vendor/faiss/faiss/IndexFlat.h +42 -59
  36. data/vendor/faiss/faiss/IndexFlatCodes.cpp +67 -0
  37. data/vendor/faiss/faiss/IndexFlatCodes.h +47 -0
  38. data/vendor/faiss/faiss/IndexHNSW.cpp +372 -348
  39. data/vendor/faiss/faiss/IndexHNSW.h +57 -41
  40. data/vendor/faiss/faiss/IndexIVF.cpp +545 -453
  41. data/vendor/faiss/faiss/IndexIVF.h +169 -118
  42. data/vendor/faiss/faiss/IndexIVFAdditiveQuantizer.cpp +316 -0
  43. data/vendor/faiss/faiss/IndexIVFAdditiveQuantizer.h +121 -0
  44. data/vendor/faiss/faiss/IndexIVFFlat.cpp +247 -252
  45. data/vendor/faiss/faiss/IndexIVFFlat.h +48 -51
  46. data/vendor/faiss/faiss/IndexIVFPQ.cpp +459 -517
  47. data/vendor/faiss/faiss/IndexIVFPQ.h +75 -67
  48. data/vendor/faiss/faiss/IndexIVFPQFastScan.cpp +406 -372
  49. data/vendor/faiss/faiss/IndexIVFPQFastScan.h +82 -57
  50. data/vendor/faiss/faiss/IndexIVFPQR.cpp +104 -102
  51. data/vendor/faiss/faiss/IndexIVFPQR.h +33 -28
  52. data/vendor/faiss/faiss/IndexIVFSpectralHash.cpp +163 -150
  53. data/vendor/faiss/faiss/IndexIVFSpectralHash.h +38 -25
  54. data/vendor/faiss/faiss/IndexLSH.cpp +66 -113
  55. data/vendor/faiss/faiss/IndexLSH.h +20 -38
  56. data/vendor/faiss/faiss/IndexLattice.cpp +42 -56
  57. data/vendor/faiss/faiss/IndexLattice.h +11 -16
  58. data/vendor/faiss/faiss/IndexNNDescent.cpp +229 -0
  59. data/vendor/faiss/faiss/IndexNNDescent.h +72 -0
  60. data/vendor/faiss/faiss/IndexNSG.cpp +301 -0
  61. data/vendor/faiss/faiss/IndexNSG.h +85 -0
  62. data/vendor/faiss/faiss/IndexPQ.cpp +387 -495
  63. data/vendor/faiss/faiss/IndexPQ.h +64 -82
  64. data/vendor/faiss/faiss/IndexPQFastScan.cpp +143 -170
  65. data/vendor/faiss/faiss/IndexPQFastScan.h +46 -32
  66. data/vendor/faiss/faiss/IndexPreTransform.cpp +120 -150
  67. data/vendor/faiss/faiss/IndexPreTransform.h +33 -36
  68. data/vendor/faiss/faiss/IndexRefine.cpp +139 -127
  69. data/vendor/faiss/faiss/IndexRefine.h +32 -23
  70. data/vendor/faiss/faiss/IndexReplicas.cpp +147 -153
  71. data/vendor/faiss/faiss/IndexReplicas.h +62 -56
  72. data/vendor/faiss/faiss/IndexScalarQuantizer.cpp +111 -172
  73. data/vendor/faiss/faiss/IndexScalarQuantizer.h +41 -59
  74. data/vendor/faiss/faiss/IndexShards.cpp +256 -240
  75. data/vendor/faiss/faiss/IndexShards.h +85 -73
  76. data/vendor/faiss/faiss/MatrixStats.cpp +112 -97
  77. data/vendor/faiss/faiss/MatrixStats.h +7 -10
  78. data/vendor/faiss/faiss/MetaIndexes.cpp +135 -157
  79. data/vendor/faiss/faiss/MetaIndexes.h +40 -34
  80. data/vendor/faiss/faiss/MetricType.h +7 -7
  81. data/vendor/faiss/faiss/VectorTransform.cpp +654 -475
  82. data/vendor/faiss/faiss/VectorTransform.h +64 -89
  83. data/vendor/faiss/faiss/clone_index.cpp +78 -73
  84. data/vendor/faiss/faiss/clone_index.h +4 -9
  85. data/vendor/faiss/faiss/gpu/GpuAutoTune.cpp +33 -38
  86. data/vendor/faiss/faiss/gpu/GpuAutoTune.h +11 -9
  87. data/vendor/faiss/faiss/gpu/GpuCloner.cpp +198 -171
  88. data/vendor/faiss/faiss/gpu/GpuCloner.h +53 -35
  89. data/vendor/faiss/faiss/gpu/GpuClonerOptions.cpp +12 -14
  90. data/vendor/faiss/faiss/gpu/GpuClonerOptions.h +27 -25
  91. data/vendor/faiss/faiss/gpu/GpuDistance.h +116 -112
  92. data/vendor/faiss/faiss/gpu/GpuFaissAssert.h +1 -2
  93. data/vendor/faiss/faiss/gpu/GpuIcmEncoder.h +60 -0
  94. data/vendor/faiss/faiss/gpu/GpuIndex.h +134 -137
  95. data/vendor/faiss/faiss/gpu/GpuIndexBinaryFlat.h +76 -73
  96. data/vendor/faiss/faiss/gpu/GpuIndexFlat.h +173 -162
  97. data/vendor/faiss/faiss/gpu/GpuIndexIVF.h +67 -64
  98. data/vendor/faiss/faiss/gpu/GpuIndexIVFFlat.h +89 -86
  99. data/vendor/faiss/faiss/gpu/GpuIndexIVFPQ.h +150 -141
  100. data/vendor/faiss/faiss/gpu/GpuIndexIVFScalarQuantizer.h +101 -103
  101. data/vendor/faiss/faiss/gpu/GpuIndicesOptions.h +17 -16
  102. data/vendor/faiss/faiss/gpu/GpuResources.cpp +116 -128
  103. data/vendor/faiss/faiss/gpu/GpuResources.h +182 -186
  104. data/vendor/faiss/faiss/gpu/StandardGpuResources.cpp +433 -422
  105. data/vendor/faiss/faiss/gpu/StandardGpuResources.h +131 -130
  106. data/vendor/faiss/faiss/gpu/impl/InterleavedCodes.cpp +468 -456
  107. data/vendor/faiss/faiss/gpu/impl/InterleavedCodes.h +25 -19
  108. data/vendor/faiss/faiss/gpu/impl/RemapIndices.cpp +22 -20
  109. data/vendor/faiss/faiss/gpu/impl/RemapIndices.h +9 -8
  110. data/vendor/faiss/faiss/gpu/perf/IndexWrapper-inl.h +39 -44
  111. data/vendor/faiss/faiss/gpu/perf/IndexWrapper.h +16 -14
  112. data/vendor/faiss/faiss/gpu/perf/PerfClustering.cpp +77 -71
  113. data/vendor/faiss/faiss/gpu/perf/PerfIVFPQAdd.cpp +109 -88
  114. data/vendor/faiss/faiss/gpu/perf/WriteIndex.cpp +75 -64
  115. data/vendor/faiss/faiss/gpu/test/TestCodePacking.cpp +230 -215
  116. data/vendor/faiss/faiss/gpu/test/TestGpuIndexBinaryFlat.cpp +80 -86
  117. data/vendor/faiss/faiss/gpu/test/TestGpuIndexFlat.cpp +284 -277
  118. data/vendor/faiss/faiss/gpu/test/TestGpuIndexIVFFlat.cpp +416 -416
  119. data/vendor/faiss/faiss/gpu/test/TestGpuIndexIVFPQ.cpp +611 -517
  120. data/vendor/faiss/faiss/gpu/test/TestGpuIndexIVFScalarQuantizer.cpp +166 -164
  121. data/vendor/faiss/faiss/gpu/test/TestGpuMemoryException.cpp +61 -53
  122. data/vendor/faiss/faiss/gpu/test/TestUtils.cpp +274 -238
  123. data/vendor/faiss/faiss/gpu/test/TestUtils.h +73 -57
  124. data/vendor/faiss/faiss/gpu/test/demo_ivfpq_indexing_gpu.cpp +47 -50
  125. data/vendor/faiss/faiss/gpu/utils/DeviceUtils.h +79 -72
  126. data/vendor/faiss/faiss/gpu/utils/StackDeviceMemory.cpp +140 -146
  127. data/vendor/faiss/faiss/gpu/utils/StackDeviceMemory.h +69 -71
  128. data/vendor/faiss/faiss/gpu/utils/StaticUtils.h +21 -16
  129. data/vendor/faiss/faiss/gpu/utils/Timer.cpp +25 -29
  130. data/vendor/faiss/faiss/gpu/utils/Timer.h +30 -29
  131. data/vendor/faiss/faiss/impl/AdditiveQuantizer.cpp +503 -0
  132. data/vendor/faiss/faiss/impl/AdditiveQuantizer.h +175 -0
  133. data/vendor/faiss/faiss/impl/AuxIndexStructures.cpp +90 -120
  134. data/vendor/faiss/faiss/impl/AuxIndexStructures.h +81 -65
  135. data/vendor/faiss/faiss/impl/FaissAssert.h +73 -58
  136. data/vendor/faiss/faiss/impl/FaissException.cpp +56 -48
  137. data/vendor/faiss/faiss/impl/FaissException.h +41 -29
  138. data/vendor/faiss/faiss/impl/HNSW.cpp +606 -617
  139. data/vendor/faiss/faiss/impl/HNSW.h +179 -200
  140. data/vendor/faiss/faiss/impl/LocalSearchQuantizer.cpp +855 -0
  141. data/vendor/faiss/faiss/impl/LocalSearchQuantizer.h +244 -0
  142. data/vendor/faiss/faiss/impl/NNDescent.cpp +487 -0
  143. data/vendor/faiss/faiss/impl/NNDescent.h +154 -0
  144. data/vendor/faiss/faiss/impl/NSG.cpp +679 -0
  145. data/vendor/faiss/faiss/impl/NSG.h +199 -0
  146. data/vendor/faiss/faiss/impl/PolysemousTraining.cpp +484 -454
  147. data/vendor/faiss/faiss/impl/PolysemousTraining.h +52 -55
  148. data/vendor/faiss/faiss/impl/ProductQuantizer-inl.h +26 -47
  149. data/vendor/faiss/faiss/impl/ProductQuantizer.cpp +469 -459
  150. data/vendor/faiss/faiss/impl/ProductQuantizer.h +76 -87
  151. data/vendor/faiss/faiss/impl/ResidualQuantizer.cpp +758 -0
  152. data/vendor/faiss/faiss/impl/ResidualQuantizer.h +188 -0
  153. data/vendor/faiss/faiss/impl/ResultHandler.h +96 -132
  154. data/vendor/faiss/faiss/impl/ScalarQuantizer.cpp +647 -707
  155. data/vendor/faiss/faiss/impl/ScalarQuantizer.h +48 -46
  156. data/vendor/faiss/faiss/impl/ThreadedIndex-inl.h +129 -131
  157. data/vendor/faiss/faiss/impl/ThreadedIndex.h +61 -55
  158. data/vendor/faiss/faiss/impl/index_read.cpp +631 -480
  159. data/vendor/faiss/faiss/impl/index_write.cpp +547 -407
  160. data/vendor/faiss/faiss/impl/io.cpp +76 -95
  161. data/vendor/faiss/faiss/impl/io.h +31 -41
  162. data/vendor/faiss/faiss/impl/io_macros.h +60 -29
  163. data/vendor/faiss/faiss/impl/kmeans1d.cpp +301 -0
  164. data/vendor/faiss/faiss/impl/kmeans1d.h +48 -0
  165. data/vendor/faiss/faiss/impl/lattice_Zn.cpp +137 -186
  166. data/vendor/faiss/faiss/impl/lattice_Zn.h +40 -51
  167. data/vendor/faiss/faiss/impl/platform_macros.h +29 -8
  168. data/vendor/faiss/faiss/impl/pq4_fast_scan.cpp +77 -124
  169. data/vendor/faiss/faiss/impl/pq4_fast_scan.h +39 -48
  170. data/vendor/faiss/faiss/impl/pq4_fast_scan_search_1.cpp +41 -52
  171. data/vendor/faiss/faiss/impl/pq4_fast_scan_search_qbs.cpp +80 -117
  172. data/vendor/faiss/faiss/impl/simd_result_handlers.h +109 -137
  173. data/vendor/faiss/faiss/index_factory.cpp +619 -397
  174. data/vendor/faiss/faiss/index_factory.h +8 -6
  175. data/vendor/faiss/faiss/index_io.h +23 -26
  176. data/vendor/faiss/faiss/invlists/BlockInvertedLists.cpp +67 -75
  177. data/vendor/faiss/faiss/invlists/BlockInvertedLists.h +22 -24
  178. data/vendor/faiss/faiss/invlists/DirectMap.cpp +96 -112
  179. data/vendor/faiss/faiss/invlists/DirectMap.h +29 -33
  180. data/vendor/faiss/faiss/invlists/InvertedLists.cpp +307 -364
  181. data/vendor/faiss/faiss/invlists/InvertedLists.h +151 -151
  182. data/vendor/faiss/faiss/invlists/InvertedListsIOHook.cpp +29 -34
  183. data/vendor/faiss/faiss/invlists/InvertedListsIOHook.h +17 -18
  184. data/vendor/faiss/faiss/invlists/OnDiskInvertedLists.cpp +257 -293
  185. data/vendor/faiss/faiss/invlists/OnDiskInvertedLists.h +50 -45
  186. data/vendor/faiss/faiss/python/python_callbacks.cpp +23 -26
  187. data/vendor/faiss/faiss/python/python_callbacks.h +9 -16
  188. data/vendor/faiss/faiss/utils/AlignedTable.h +79 -44
  189. data/vendor/faiss/faiss/utils/Heap.cpp +40 -48
  190. data/vendor/faiss/faiss/utils/Heap.h +186 -209
  191. data/vendor/faiss/faiss/utils/WorkerThread.cpp +67 -76
  192. data/vendor/faiss/faiss/utils/WorkerThread.h +32 -33
  193. data/vendor/faiss/faiss/utils/distances.cpp +305 -312
  194. data/vendor/faiss/faiss/utils/distances.h +170 -122
  195. data/vendor/faiss/faiss/utils/distances_simd.cpp +498 -508
  196. data/vendor/faiss/faiss/utils/extra_distances-inl.h +117 -0
  197. data/vendor/faiss/faiss/utils/extra_distances.cpp +113 -232
  198. data/vendor/faiss/faiss/utils/extra_distances.h +30 -29
  199. data/vendor/faiss/faiss/utils/hamming-inl.h +260 -209
  200. data/vendor/faiss/faiss/utils/hamming.cpp +375 -469
  201. data/vendor/faiss/faiss/utils/hamming.h +62 -85
  202. data/vendor/faiss/faiss/utils/ordered_key_value.h +16 -18
  203. data/vendor/faiss/faiss/utils/partitioning.cpp +393 -318
  204. data/vendor/faiss/faiss/utils/partitioning.h +26 -21
  205. data/vendor/faiss/faiss/utils/quantize_lut.cpp +78 -66
  206. data/vendor/faiss/faiss/utils/quantize_lut.h +22 -20
  207. data/vendor/faiss/faiss/utils/random.cpp +39 -63
  208. data/vendor/faiss/faiss/utils/random.h +13 -16
  209. data/vendor/faiss/faiss/utils/simdlib.h +4 -2
  210. data/vendor/faiss/faiss/utils/simdlib_avx2.h +88 -85
  211. data/vendor/faiss/faiss/utils/simdlib_emulated.h +226 -165
  212. data/vendor/faiss/faiss/utils/simdlib_neon.h +832 -0
  213. data/vendor/faiss/faiss/utils/utils.cpp +304 -287
  214. data/vendor/faiss/faiss/utils/utils.h +54 -49
  215. metadata +29 -4
@@ -15,15 +15,15 @@
15
15
 
16
16
  #include <stdint.h>
17
17
 
18
- #include <vector>
19
- #include <unordered_set>
18
+ #include <cstring>
20
19
  #include <memory>
21
20
  #include <mutex>
21
+ #include <unordered_set>
22
+ #include <vector>
22
23
 
23
24
  #include <faiss/Index.h>
24
25
  #include <faiss/impl/platform_macros.h>
25
26
 
26
-
27
27
  namespace faiss {
28
28
 
29
29
  /** The objective is to have a simple result structure while
@@ -31,42 +31,39 @@ namespace faiss {
31
31
  * do_allocation can be overloaded to allocate the result tables in
32
32
  * the matrix type of a scripting language like Lua or Python. */
33
33
  struct RangeSearchResult {
34
- size_t nq; ///< nb of queries
35
- size_t *lims; ///< size (nq + 1)
34
+ size_t nq; ///< nb of queries
35
+ size_t* lims; ///< size (nq + 1)
36
36
 
37
37
  typedef Index::idx_t idx_t;
38
38
 
39
- idx_t *labels; ///< result for query i is labels[lims[i]:lims[i+1]]
40
- float *distances; ///< corresponding distances (not sorted)
39
+ idx_t* labels; ///< result for query i is labels[lims[i]:lims[i+1]]
40
+ float* distances; ///< corresponding distances (not sorted)
41
41
 
42
42
  size_t buffer_size; ///< size of the result buffers used
43
43
 
44
44
  /// lims must be allocated on input to range_search.
45
- explicit RangeSearchResult (idx_t nq, bool alloc_lims=true);
45
+ explicit RangeSearchResult(idx_t nq, bool alloc_lims = true);
46
46
 
47
47
  /// called when lims contains the nb of elements result entries
48
48
  /// for each query
49
49
 
50
- virtual void do_allocation ();
50
+ virtual void do_allocation();
51
51
 
52
- virtual ~RangeSearchResult ();
52
+ virtual ~RangeSearchResult();
53
53
  };
54
54
 
55
-
56
55
  /** Encapsulates a set of ids to remove. */
57
56
  struct IDSelector {
58
57
  typedef Index::idx_t idx_t;
59
- virtual bool is_member (idx_t id) const = 0;
58
+ virtual bool is_member(idx_t id) const = 0;
60
59
  virtual ~IDSelector() {}
61
60
  };
62
61
 
63
-
64
-
65
62
  /** remove ids between [imni, imax) */
66
- struct IDSelectorRange: IDSelector {
63
+ struct IDSelectorRange : IDSelector {
67
64
  idx_t imin, imax;
68
65
 
69
- IDSelectorRange (idx_t imin, idx_t imax);
66
+ IDSelectorRange(idx_t imin, idx_t imax);
70
67
  bool is_member(idx_t id) const override;
71
68
  ~IDSelectorRange() override {}
72
69
  };
@@ -76,11 +73,11 @@ struct IDSelectorRange: IDSelector {
76
73
  * this is inefficient in most cases, except for IndexIVF with
77
74
  * maintain_direct_map
78
75
  */
79
- struct IDSelectorArray: IDSelector {
76
+ struct IDSelectorArray : IDSelector {
80
77
  size_t n;
81
- const idx_t *ids;
78
+ const idx_t* ids;
82
79
 
83
- IDSelectorArray (size_t n, const idx_t *ids);
80
+ IDSelectorArray(size_t n, const idx_t* ids);
84
81
  bool is_member(idx_t id) const override;
85
82
  ~IDSelectorArray() override {}
86
83
  };
@@ -91,8 +88,7 @@ struct IDSelectorArray: IDSelector {
91
88
  * unordered_set are just the least significant bits of the id. This
92
89
  * works fine for random ids or ids in sequences but will produce many
93
90
  * hash collisions if lsb's are always the same */
94
- struct IDSelectorBatch: IDSelector {
95
-
91
+ struct IDSelectorBatch : IDSelector {
96
92
  std::unordered_set<idx_t> set;
97
93
 
98
94
  typedef unsigned char uint8_t;
@@ -100,7 +96,7 @@ struct IDSelectorBatch: IDSelector {
100
96
  int nbits;
101
97
  idx_t mask;
102
98
 
103
- IDSelectorBatch (size_t n, const idx_t *indices);
99
+ IDSelectorBatch(size_t n, const idx_t* indices);
104
100
  bool is_member(idx_t id) const override;
105
101
  ~IDSelectorBatch() override {}
106
102
  };
@@ -124,28 +120,26 @@ struct BufferList {
124
120
  size_t buffer_size;
125
121
 
126
122
  struct Buffer {
127
- idx_t *ids;
128
- float *dis;
123
+ idx_t* ids;
124
+ float* dis;
129
125
  };
130
126
 
131
127
  std::vector<Buffer> buffers;
132
128
  size_t wp; ///< write pointer in the last buffer.
133
129
 
134
- explicit BufferList (size_t buffer_size);
130
+ explicit BufferList(size_t buffer_size);
135
131
 
136
- ~BufferList ();
132
+ ~BufferList();
137
133
 
138
134
  /// create a new buffer
139
- void append_buffer ();
135
+ void append_buffer();
140
136
 
141
137
  /// add one result, possibly appending a new buffer if needed
142
- void add (idx_t id, float dis);
138
+ void add(idx_t id, float dis);
143
139
 
144
140
  /// copy elemnts ofs:ofs+n-1 seen as linear data in the buffers to
145
141
  /// tables dest_ids, dest_dis
146
- void copy_range (size_t ofs, size_t n,
147
- idx_t * dest_ids, float *dest_dis);
148
-
142
+ void copy_range(size_t ofs, size_t n, idx_t* dest_ids, float* dest_dis);
149
143
  };
150
144
 
151
145
  struct RangeSearchPartialResult;
@@ -153,46 +147,45 @@ struct RangeSearchPartialResult;
153
147
  /// result structure for a single query
154
148
  struct RangeQueryResult {
155
149
  using idx_t = Index::idx_t;
156
- idx_t qno; //< id of the query
157
- size_t nres; //< nb of results for this query
158
- RangeSearchPartialResult * pres;
150
+ idx_t qno; //< id of the query
151
+ size_t nres; //< nb of results for this query
152
+ RangeSearchPartialResult* pres;
159
153
 
160
154
  /// called by search function to report a new result
161
- void add (float dis, idx_t id);
155
+ void add(float dis, idx_t id);
162
156
  };
163
157
 
164
158
  /// the entries in the buffers are split per query
165
- struct RangeSearchPartialResult: BufferList {
166
- RangeSearchResult * res;
159
+ struct RangeSearchPartialResult : BufferList {
160
+ RangeSearchResult* res;
167
161
 
168
162
  /// eventually the result will be stored in res_in
169
- explicit RangeSearchPartialResult (RangeSearchResult * res_in);
163
+ explicit RangeSearchPartialResult(RangeSearchResult* res_in);
170
164
 
171
165
  /// query ids + nb of results per query.
172
166
  std::vector<RangeQueryResult> queries;
173
167
 
174
168
  /// begin a new result
175
- RangeQueryResult & new_result (idx_t qno);
169
+ RangeQueryResult& new_result(idx_t qno);
176
170
 
177
171
  /*****************************************
178
172
  * functions used at the end of the search to merge the result
179
173
  * lists */
180
- void finalize ();
174
+ void finalize();
181
175
 
182
176
  /// called by range_search before do_allocation
183
- void set_lims ();
177
+ void set_lims();
184
178
 
185
179
  /// called by range_search after do_allocation
186
- void copy_result (bool incremental = false);
180
+ void copy_result(bool incremental = false);
187
181
 
188
182
  /// merge a set of PartialResult's into one RangeSearchResult
189
183
  /// on ouptut the partialresults are empty!
190
- static void merge (std::vector <RangeSearchPartialResult *> &
191
- partial_results, bool do_delete=true);
192
-
184
+ static void merge(
185
+ std::vector<RangeSearchPartialResult*>& partial_results,
186
+ bool do_delete = true);
193
187
  };
194
188
 
195
-
196
189
  /***********************************************************
197
190
  * The distance computer maintains a current query and computes
198
191
  * distances to elements in an index that supports random access.
@@ -202,19 +195,19 @@ struct RangeSearchPartialResult: BufferList {
202
195
  * instantiate one from each thread if needed.
203
196
  ***********************************************************/
204
197
  struct DistanceComputer {
205
- using idx_t = Index::idx_t;
198
+ using idx_t = Index::idx_t;
206
199
 
207
- /// called before computing distances. Pointer x should remain valid
208
- /// while operator () is called
209
- virtual void set_query(const float *x) = 0;
200
+ /// called before computing distances. Pointer x should remain valid
201
+ /// while operator () is called
202
+ virtual void set_query(const float* x) = 0;
210
203
 
211
- /// compute distance of vector i to current query
212
- virtual float operator () (idx_t i) = 0;
204
+ /// compute distance of vector i to current query
205
+ virtual float operator()(idx_t i) = 0;
213
206
 
214
- /// compute distance between two stored vectors
215
- virtual float symmetric_dis (idx_t i, idx_t j) = 0;
207
+ /// compute distance between two stored vectors
208
+ virtual float symmetric_dis(idx_t i, idx_t j) = 0;
216
209
 
217
- virtual ~DistanceComputer() {}
210
+ virtual ~DistanceComputer() {}
218
211
  };
219
212
 
220
213
  /***********************************************************
@@ -222,7 +215,7 @@ struct DistanceComputer {
222
215
  ***********************************************************/
223
216
 
224
217
  struct FAISS_API InterruptCallback {
225
- virtual bool want_interrupt () = 0;
218
+ virtual bool want_interrupt() = 0;
226
219
  virtual ~InterruptCallback() {}
227
220
 
228
221
  // lock that protects concurrent calls to is_interrupted
@@ -230,7 +223,7 @@ struct FAISS_API InterruptCallback {
230
223
 
231
224
  static std::unique_ptr<InterruptCallback> instance;
232
225
 
233
- static void clear_instance ();
226
+ static void clear_instance();
234
227
 
235
228
  /** check if:
236
229
  * - an interrupt callback is set
@@ -238,23 +231,46 @@ struct FAISS_API InterruptCallback {
238
231
  * if this is the case, then throw an exception. Should not be called
239
232
  * from multiple threads.
240
233
  */
241
- static void check ();
234
+ static void check();
242
235
 
243
236
  /// same as check() but return true if is interrupted instead of
244
237
  /// throwing. Can be called from multiple threads.
245
- static bool is_interrupted ();
238
+ static bool is_interrupted();
246
239
 
247
240
  /** assuming each iteration takes a certain number of flops, what
248
241
  * is a reasonable interval to check for interrupts?
249
242
  */
250
- static size_t get_period_hint (size_t flops);
251
-
243
+ static size_t get_period_hint(size_t flops);
252
244
  };
253
245
 
246
+ /// set implementation optimized for fast access.
247
+ struct VisitedTable {
248
+ std::vector<uint8_t> visited;
249
+ int visno;
250
+
251
+ explicit VisitedTable(int size) : visited(size), visno(1) {}
252
+
253
+ /// set flag #no to true
254
+ void set(int no) {
255
+ visited[no] = visno;
256
+ }
257
+
258
+ /// get flag #no
259
+ bool get(int no) const {
260
+ return visited[no] == visno;
261
+ }
262
+
263
+ /// reset all flags to false
264
+ void advance() {
265
+ visno++;
266
+ if (visno == 250) {
267
+ // 250 rather than 255 because sometimes we use visno and visno+1
268
+ memset(visited.data(), 0, sizeof(visited[0]) * visited.size());
269
+ visno = 1;
270
+ }
271
+ }
272
+ };
254
273
 
255
-
256
- }; // namespace faiss
257
-
258
-
274
+ } // namespace faiss
259
275
 
260
276
  #endif
@@ -12,85 +12,100 @@
12
12
 
13
13
  #include <faiss/impl/FaissException.h>
14
14
  #include <faiss/impl/platform_macros.h>
15
- #include <cstdlib>
16
15
  #include <cstdio>
16
+ #include <cstdlib>
17
17
  #include <string>
18
18
 
19
19
  ///
20
20
  /// Assertions
21
21
  ///
22
22
 
23
- #define FAISS_ASSERT(X) \
24
- do { \
25
- if (! (X)) { \
26
- fprintf(stderr, "Faiss assertion '%s' failed in %s " \
27
- "at %s:%d\n", \
28
- #X, __PRETTY_FUNCTION__, __FILE__, __LINE__); \
29
- abort(); \
30
- } \
31
- } while (false)
23
+ #define FAISS_ASSERT(X) \
24
+ do { \
25
+ if (!(X)) { \
26
+ fprintf(stderr, \
27
+ "Faiss assertion '%s' failed in %s " \
28
+ "at %s:%d\n", \
29
+ #X, \
30
+ __PRETTY_FUNCTION__, \
31
+ __FILE__, \
32
+ __LINE__); \
33
+ abort(); \
34
+ } \
35
+ } while (false)
32
36
 
33
- #define FAISS_ASSERT_MSG(X, MSG) \
34
- do { \
35
- if (! (X)) { \
36
- fprintf(stderr, "Faiss assertion '%s' failed in %s " \
37
- "at %s:%d; details: " MSG "\n", \
38
- #X, __PRETTY_FUNCTION__, __FILE__, __LINE__); \
39
- abort(); \
40
- } \
41
- } while (false)
37
+ #define FAISS_ASSERT_MSG(X, MSG) \
38
+ do { \
39
+ if (!(X)) { \
40
+ fprintf(stderr, \
41
+ "Faiss assertion '%s' failed in %s " \
42
+ "at %s:%d; details: " MSG "\n", \
43
+ #X, \
44
+ __PRETTY_FUNCTION__, \
45
+ __FILE__, \
46
+ __LINE__); \
47
+ abort(); \
48
+ } \
49
+ } while (false)
42
50
 
43
- #define FAISS_ASSERT_FMT(X, FMT, ...) \
44
- do { \
45
- if (! (X)) { \
46
- fprintf(stderr, "Faiss assertion '%s' failed in %s " \
47
- "at %s:%d; details: " FMT "\n", \
48
- #X, __PRETTY_FUNCTION__, __FILE__, __LINE__, __VA_ARGS__); \
49
- abort(); \
50
- } \
51
- } while (false)
51
+ #define FAISS_ASSERT_FMT(X, FMT, ...) \
52
+ do { \
53
+ if (!(X)) { \
54
+ fprintf(stderr, \
55
+ "Faiss assertion '%s' failed in %s " \
56
+ "at %s:%d; details: " FMT "\n", \
57
+ #X, \
58
+ __PRETTY_FUNCTION__, \
59
+ __FILE__, \
60
+ __LINE__, \
61
+ __VA_ARGS__); \
62
+ abort(); \
63
+ } \
64
+ } while (false)
52
65
 
53
66
  ///
54
67
  /// Exceptions for returning user errors
55
68
  ///
56
69
 
57
- #define FAISS_THROW_MSG(MSG) \
58
- do { \
59
- throw faiss::FaissException(MSG, __PRETTY_FUNCTION__, __FILE__, __LINE__); \
60
- } while (false)
70
+ #define FAISS_THROW_MSG(MSG) \
71
+ do { \
72
+ throw faiss::FaissException( \
73
+ MSG, __PRETTY_FUNCTION__, __FILE__, __LINE__); \
74
+ } while (false)
61
75
 
62
- #define FAISS_THROW_FMT(FMT, ...) \
63
- do { \
64
- std::string __s; \
65
- int __size = snprintf(nullptr, 0, FMT, __VA_ARGS__); \
66
- __s.resize(__size + 1); \
67
- snprintf(&__s[0], __s.size(), FMT, __VA_ARGS__); \
68
- throw faiss::FaissException(__s, __PRETTY_FUNCTION__, __FILE__, __LINE__); \
69
- } while (false)
76
+ #define FAISS_THROW_FMT(FMT, ...) \
77
+ do { \
78
+ std::string __s; \
79
+ int __size = snprintf(nullptr, 0, FMT, __VA_ARGS__); \
80
+ __s.resize(__size + 1); \
81
+ snprintf(&__s[0], __s.size(), FMT, __VA_ARGS__); \
82
+ throw faiss::FaissException( \
83
+ __s, __PRETTY_FUNCTION__, __FILE__, __LINE__); \
84
+ } while (false)
70
85
 
71
86
  ///
72
87
  /// Exceptions thrown upon a conditional failure
73
88
  ///
74
89
 
75
- #define FAISS_THROW_IF_NOT(X) \
76
- do { \
77
- if (!(X)) { \
78
- FAISS_THROW_FMT("Error: '%s' failed", #X); \
79
- } \
80
- } while (false)
90
+ #define FAISS_THROW_IF_NOT(X) \
91
+ do { \
92
+ if (!(X)) { \
93
+ FAISS_THROW_FMT("Error: '%s' failed", #X); \
94
+ } \
95
+ } while (false)
81
96
 
82
- #define FAISS_THROW_IF_NOT_MSG(X, MSG) \
83
- do { \
84
- if (!(X)) { \
85
- FAISS_THROW_FMT("Error: '%s' failed: " MSG, #X); \
86
- } \
87
- } while (false)
97
+ #define FAISS_THROW_IF_NOT_MSG(X, MSG) \
98
+ do { \
99
+ if (!(X)) { \
100
+ FAISS_THROW_FMT("Error: '%s' failed: " MSG, #X); \
101
+ } \
102
+ } while (false)
88
103
 
89
- #define FAISS_THROW_IF_NOT_FMT(X, FMT, ...) \
90
- do { \
91
- if (!(X)) { \
92
- FAISS_THROW_FMT("Error: '%s' failed: " FMT, #X, __VA_ARGS__); \
93
- } \
94
- } while (false)
104
+ #define FAISS_THROW_IF_NOT_FMT(X, FMT, ...) \
105
+ do { \
106
+ if (!(X)) { \
107
+ FAISS_THROW_FMT("Error: '%s' failed: " FMT, #X, __VA_ARGS__); \
108
+ } \
109
+ } while (false)
95
110
 
96
111
  #endif
@@ -10,71 +10,81 @@
10
10
  #include <faiss/impl/FaissException.h>
11
11
  #include <sstream>
12
12
 
13
- #ifdef __GNUG__
13
+ #ifdef __GNUG__
14
14
  #include <cxxabi.h>
15
15
  #endif
16
16
 
17
17
  namespace faiss {
18
18
 
19
- FaissException::FaissException(const std::string& m)
20
- : msg(m) {
21
- }
22
-
23
- FaissException::FaissException(const std::string& m,
24
- const char* funcName,
25
- const char* file,
26
- int line) {
27
- int size = snprintf(nullptr, 0, "Error in %s at %s:%d: %s",
28
- funcName, file, line, m.c_str());
29
- msg.resize(size + 1);
30
- snprintf(&msg[0], msg.size(), "Error in %s at %s:%d: %s",
31
- funcName, file, line, m.c_str());
19
+ FaissException::FaissException(const std::string& m) : msg(m) {}
20
+
21
+ FaissException::FaissException(
22
+ const std::string& m,
23
+ const char* funcName,
24
+ const char* file,
25
+ int line) {
26
+ int size = snprintf(
27
+ nullptr,
28
+ 0,
29
+ "Error in %s at %s:%d: %s",
30
+ funcName,
31
+ file,
32
+ line,
33
+ m.c_str());
34
+ msg.resize(size + 1);
35
+ snprintf(
36
+ &msg[0],
37
+ msg.size(),
38
+ "Error in %s at %s:%d: %s",
39
+ funcName,
40
+ file,
41
+ line,
42
+ m.c_str());
32
43
  }
33
44
 
34
- const char*
35
- FaissException::what() const noexcept {
36
- return msg.c_str();
45
+ const char* FaissException::what() const noexcept {
46
+ return msg.c_str();
37
47
  }
38
48
 
39
49
  void handleExceptions(
40
- std::vector<std::pair<int, std::exception_ptr>>& exceptions) {
41
- if (exceptions.size() == 1) {
42
- // throw the single received exception directly
43
- std::rethrow_exception(exceptions.front().second);
44
-
45
- } else if (exceptions.size() > 1) {
46
- // multiple exceptions; aggregate them and return a single exception
47
- std::stringstream ss;
48
-
49
- for (auto& p : exceptions) {
50
- try {
51
- std::rethrow_exception(p.second);
52
- } catch (std::exception& ex) {
53
- if (ex.what()) {
54
- // exception message available
55
- ss << "Exception thrown from index " << p.first << ": "
56
- << ex.what() << "\n";
57
- } else {
58
- // No message available
59
- ss << "Unknown exception thrown from index " << p.first << "\n";
50
+ std::vector<std::pair<int, std::exception_ptr>>& exceptions) {
51
+ if (exceptions.size() == 1) {
52
+ // throw the single received exception directly
53
+ std::rethrow_exception(exceptions.front().second);
54
+
55
+ } else if (exceptions.size() > 1) {
56
+ // multiple exceptions; aggregate them and return a single exception
57
+ std::stringstream ss;
58
+
59
+ for (auto& p : exceptions) {
60
+ try {
61
+ std::rethrow_exception(p.second);
62
+ } catch (std::exception& ex) {
63
+ if (ex.what()) {
64
+ // exception message available
65
+ ss << "Exception thrown from index " << p.first << ": "
66
+ << ex.what() << "\n";
67
+ } else {
68
+ // No message available
69
+ ss << "Unknown exception thrown from index " << p.first
70
+ << "\n";
71
+ }
72
+ } catch (...) {
73
+ ss << "Unknown exception thrown from index " << p.first << "\n";
74
+ }
60
75
  }
61
- } catch (...) {
62
- ss << "Unknown exception thrown from index " << p.first << "\n";
63
- }
64
- }
65
76
 
66
- throw FaissException(ss.str());
67
- }
77
+ throw FaissException(ss.str());
78
+ }
68
79
  }
69
80
 
70
-
71
81
  // From
72
82
  // https://stackoverflow.com/questions/281818/unmangling-the-result-of-stdtype-infoname
73
83
 
74
84
  std::string demangle_cpp_symbol(const char* name) {
75
85
  #ifdef __GNUG__
76
86
  int status = -1;
77
- const char * res = abi::__cxa_demangle(name, nullptr, nullptr, &status);
87
+ const char* res = abi::__cxa_demangle(name, nullptr, nullptr, &status);
78
88
  std::string sres;
79
89
  if (status == 0) {
80
90
  sres = res;
@@ -87,6 +97,4 @@ std::string demangle_cpp_symbol(const char* name) {
87
97
  #endif
88
98
  }
89
99
 
90
-
91
-
92
- } // namespace
100
+ } // namespace faiss