infinity-sdk 0.7.3__tar.gz → 0.7.4__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (39) hide show
  1. {infinity_sdk-0.7.3/python/infinity_sdk/infinity_sdk.egg-info → infinity_sdk-0.7.4}/PKG-INFO +2 -2
  2. {infinity_sdk-0.7.3 → infinity_sdk-0.7.4}/README.md +1 -1
  3. {infinity_sdk-0.7.3 → infinity_sdk-0.7.4}/pyproject.toml +1 -1
  4. {infinity_sdk-0.7.3 → infinity_sdk-0.7.4}/python/infinity_sdk/README.md +1 -1
  5. {infinity_sdk-0.7.3 → infinity_sdk-0.7.4}/python/infinity_sdk/infinity/__init__.py +4 -1
  6. infinity_sdk-0.7.4/python/infinity_sdk/infinity/filter_utils.py +50 -0
  7. {infinity_sdk-0.7.3 → infinity_sdk-0.7.4}/python/infinity_sdk/infinity/http_utils.py +3 -2
  8. {infinity_sdk-0.7.3 → infinity_sdk-0.7.4}/python/infinity_sdk/infinity/huqie.txt.trie +1 -1
  9. {infinity_sdk-0.7.3 → infinity_sdk-0.7.4}/python/infinity_sdk/infinity/index.py +63 -0
  10. {infinity_sdk-0.7.3 → infinity_sdk-0.7.4}/python/infinity_sdk/infinity/infinity_http.py +2 -0
  11. {infinity_sdk-0.7.3 → infinity_sdk-0.7.4}/python/infinity_sdk/infinity/rag_tokenizer.py +43 -1
  12. {infinity_sdk-0.7.3 → infinity_sdk-0.7.4}/python/infinity_sdk/infinity/remote_thrift/client.py +2 -2
  13. {infinity_sdk-0.7.3 → infinity_sdk-0.7.4}/python/infinity_sdk/infinity/remote_thrift/infinity_thrift_rpc/InfinityService.py +2745 -1961
  14. {infinity_sdk-0.7.3 → infinity_sdk-0.7.4}/python/infinity_sdk/infinity/remote_thrift/infinity_thrift_rpc/constants.py +1 -1
  15. infinity_sdk-0.7.4/python/infinity_sdk/infinity/remote_thrift/infinity_thrift_rpc/ttypes.py +12382 -0
  16. {infinity_sdk-0.7.3 → infinity_sdk-0.7.4}/python/infinity_sdk/infinity/remote_thrift/utils.py +17 -12
  17. {infinity_sdk-0.7.3 → infinity_sdk-0.7.4/python/infinity_sdk/infinity_sdk.egg-info}/PKG-INFO +2 -2
  18. {infinity_sdk-0.7.3 → infinity_sdk-0.7.4}/python/infinity_sdk/infinity_sdk.egg-info/SOURCES.txt +1 -0
  19. infinity_sdk-0.7.3/python/infinity_sdk/infinity/remote_thrift/infinity_thrift_rpc/ttypes.py +0 -11558
  20. {infinity_sdk-0.7.3 → infinity_sdk-0.7.4}/LICENSE +0 -0
  21. {infinity_sdk-0.7.3 → infinity_sdk-0.7.4}/python/infinity_sdk/infinity/common.py +0 -0
  22. {infinity_sdk-0.7.3 → infinity_sdk-0.7.4}/python/infinity_sdk/infinity/connection_pool.py +0 -0
  23. {infinity_sdk-0.7.3 → infinity_sdk-0.7.4}/python/infinity_sdk/infinity/db.py +0 -0
  24. {infinity_sdk-0.7.3 → infinity_sdk-0.7.4}/python/infinity_sdk/infinity/errors.py +0 -0
  25. {infinity_sdk-0.7.3 → infinity_sdk-0.7.4}/python/infinity_sdk/infinity/huqie.txt +0 -0
  26. {infinity_sdk-0.7.3 → infinity_sdk-0.7.4}/python/infinity_sdk/infinity/infinity.py +0 -0
  27. {infinity_sdk-0.7.3 → infinity_sdk-0.7.4}/python/infinity_sdk/infinity/remote_thrift/__init__.py +0 -0
  28. {infinity_sdk-0.7.3 → infinity_sdk-0.7.4}/python/infinity_sdk/infinity/remote_thrift/db.py +0 -0
  29. {infinity_sdk-0.7.3 → infinity_sdk-0.7.4}/python/infinity_sdk/infinity/remote_thrift/infinity.py +0 -0
  30. {infinity_sdk-0.7.3 → infinity_sdk-0.7.4}/python/infinity_sdk/infinity/remote_thrift/infinity_thrift_rpc/__init__.py +0 -0
  31. {infinity_sdk-0.7.3 → infinity_sdk-0.7.4}/python/infinity_sdk/infinity/remote_thrift/query_builder.py +0 -0
  32. {infinity_sdk-0.7.3 → infinity_sdk-0.7.4}/python/infinity_sdk/infinity/remote_thrift/table.py +0 -0
  33. {infinity_sdk-0.7.3 → infinity_sdk-0.7.4}/python/infinity_sdk/infinity/remote_thrift/types.py +0 -0
  34. {infinity_sdk-0.7.3 → infinity_sdk-0.7.4}/python/infinity_sdk/infinity/table.py +0 -0
  35. {infinity_sdk-0.7.3 → infinity_sdk-0.7.4}/python/infinity_sdk/infinity/utils.py +0 -0
  36. {infinity_sdk-0.7.3 → infinity_sdk-0.7.4}/python/infinity_sdk/infinity_sdk.egg-info/dependency_links.txt +0 -0
  37. {infinity_sdk-0.7.3 → infinity_sdk-0.7.4}/python/infinity_sdk/infinity_sdk.egg-info/requires.txt +0 -0
  38. {infinity_sdk-0.7.3 → infinity_sdk-0.7.4}/python/infinity_sdk/infinity_sdk.egg-info/top_level.txt +0 -0
  39. {infinity_sdk-0.7.3 → infinity_sdk-0.7.4}/setup.cfg +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: infinity-sdk
3
- Version: 0.7.3
3
+ Version: 0.7.4
4
4
  Summary: infinity
5
5
  License-Expression: Apache-2.0
6
6
  Requires-Python: <3.14,>=3.11
@@ -97,7 +97,7 @@ Infinity supports two working modes, embedded mode and client-server mode. The f
97
97
 
98
98
  2. Install the `infinity-sdk` package:
99
99
  ```bash
100
- pip install infinity-sdk==0.7.3
100
+ pip install infinity-sdk==0.7.4
101
101
  ```
102
102
 
103
103
  3. Use Infinity to conduct a dense vector search:
@@ -96,7 +96,7 @@ If you are on Windows 10+, you must enable WSL or WSL2 to deploy Infinity using
96
96
  ### Install Infinity client
97
97
 
98
98
  ```
99
- pip install infinity-sdk==0.7.3
99
+ pip install infinity-sdk==0.7.4
100
100
  ```
101
101
 
102
102
  ### Run a vector search
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "infinity-sdk"
3
- version = "0.7.3"
3
+ version = "0.7.4"
4
4
  description = "infinity"
5
5
  readme = "python/infinity_sdk/README.md"
6
6
  license = "Apache-2.0"
@@ -63,7 +63,7 @@ Infinity supports two working modes, embedded mode and client-server mode. The f
63
63
 
64
64
  2. Install the `infinity-sdk` package:
65
65
  ```bash
66
- pip install infinity-sdk==0.7.3
66
+ pip install infinity-sdk==0.7.4
67
67
  ```
68
68
 
69
69
  3. Use Infinity to conduct a dense vector search:
@@ -28,6 +28,7 @@ from infinity.common import (
28
28
  NetworkAddress,
29
29
  )
30
30
  from infinity.errors import ErrorCode
31
+ from infinity.filter_utils import quote_string_literal, regex_filter
31
32
  from infinity.infinity import InfinityConnection
32
33
  from infinity.remote_thrift.infinity import RemoteThriftInfinityConnection
33
34
 
@@ -40,7 +41,9 @@ __all__ = [
40
41
  "InfinityException",
41
42
  "NetworkAddress",
42
43
  "RemoteThriftInfinityConnection",
43
- "connect"
44
+ "connect",
45
+ "quote_string_literal",
46
+ "regex_filter"
44
47
  ]
45
48
 
46
49
 
@@ -0,0 +1,50 @@
1
+ # Copyright(C) 2026 InfiniFlow, Inc. All rights reserved.
2
+ #
3
+ # Licensed under the Apache License, Version 2.0 (the "License");
4
+ # you may not use this file except in compliance with the License.
5
+ # You may obtain a copy of the License at
6
+ #
7
+ # https://www.apache.org/licenses/LICENSE-2.0
8
+ #
9
+ # Unless required by applicable law or agreed to in writing, software
10
+ # distributed under the License is distributed on an "AS IS" BASIS,
11
+ # WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
12
+ # See the License for the specific language governing permissions and
13
+ # limitations under the License.
14
+
15
+ """Helpers that build filter expressions."""
16
+
17
+ from __future__ import annotations
18
+
19
+ __all__ = ["quote_string_literal", "regex_filter"]
20
+
21
+
22
+ def quote_string_literal(value: str) -> str:
23
+ """Quote a value as a string literal for a filter expression.
24
+
25
+ A backslash is passed through unchanged, which is what a regular expression
26
+ needs, so only the quote itself has to be doubled.
27
+ """
28
+ if not isinstance(value, str):
29
+ raise TypeError(f"Expected a string, but got {type(value).__name__}")
30
+ return "'" + value.replace("'", "''") + "'"
31
+
32
+
33
+ def regex_filter(column: str, pattern: str) -> str:
34
+ """Build a ``regex(column, pattern)`` filter.
35
+
36
+ The pattern is evaluated by RE2 on the rows the filter is applied to. When
37
+ ``column`` has a full-text index built with a sparse gram analyzer
38
+ (``sparsegram-3-12``, optionally ``-fold``), the server additionally
39
+ narrows those rows with the literals the pattern proves mandatory before
40
+ the regular expression runs, so the regular expression only sees
41
+ candidates. The narrowing never removes a row the pattern would have
42
+ matched, and a column without such an index keeps a plain scan.
43
+
44
+ Example::
45
+
46
+ table.filter(regex_filter("doc", r"colou?r of the (sky|sea)"))
47
+ """
48
+ if not isinstance(column, str) or not column:
49
+ raise ValueError("column must be a non-empty string")
50
+ return f"regex({column}, {quote_string_literal(pattern)})"
@@ -66,9 +66,10 @@ functions = [
66
66
  "ltrim",
67
67
  "rtrim",
68
68
  "reverse",
69
+ "regex",
69
70
  ]
70
71
 
71
- bool_functions = ["filter_text", "filter_fulltext", "or", "and", "not"]
72
+ bool_functions = ["filter_text", "filter_fulltext", "or", "and", "not", "regex"]
72
73
 
73
74
 
74
75
  def function_return_type(function_name, param_type):
@@ -79,7 +80,7 @@ def function_return_type(function_name, param_type):
79
80
  return param_type
80
81
  else:
81
82
  return 'Float64'
82
- elif function_name in ["filter_text", "filter_fulltext", "or", "and", "not"]:
83
+ elif function_name in ["filter_text", "filter_fulltext", "or", "and", "not", "regex"]:
83
84
  return 'boolean'
84
85
  elif function_name == "trunc":
85
86
  return 'string'
@@ -1,4 +1,4 @@
1
1
  [diffend] Oversized file quarantined before diffing.
2
- name: infinity_sdk-0.7.3/python/infinity_sdk/infinity/huqie.txt.trie
2
+ name: infinity_sdk-0.7.4/python/infinity_sdk/infinity/huqie.txt.trie
3
3
  size: 54775939 bytes
4
4
  sha256: 32ab213044af4aee58c46b4bfa4d2b61ba740ce8c894bb96e80fad00f145bcc4
@@ -75,6 +75,51 @@ class InitParameter:
75
75
  return ttypes.InitParameter(self.param_name, self.param_value)
76
76
 
77
77
 
78
+ SPARSEGRAM_DEFAULT_MIN_GRAM = 3
79
+ SPARSEGRAM_DEFAULT_MAX_GRAM = 12
80
+
81
+
82
+ def sparsegram_analyzer(min_gram: int = SPARSEGRAM_DEFAULT_MIN_GRAM,
83
+ max_gram: int = SPARSEGRAM_DEFAULT_MAX_GRAM,
84
+ fold_case: bool = False) -> str:
85
+ """Build the analyzer name of a sparse gram full-text index.
86
+
87
+ The analyzer emits content-defined n-grams of the whole value instead of
88
+ tokens, which is what lets ``regex(column, pattern)`` filters use the index:
89
+ the literals the pattern proves mandatory are turned into gram lookups
90
+ before the regular expression runs, so the regular expression only verifies
91
+ candidates. Chinese, Japanese and Korean values additionally get one and two
92
+ character grams, so a single character or a two character word can narrow as
93
+ well.
94
+
95
+ Args:
96
+ min_gram: Shortest window considered, in characters. Defaults to 3.
97
+ max_gram: Longest window considered, in characters. Defaults to 12.
98
+ Raising it adds rare, high information grams for little index space.
99
+ fold_case: Emit the ASCII-lowercased form of every gram too. Required for
100
+ a case insensitive pattern such as ``(?i)nobel prize`` to use the
101
+ index at all; without it such a pattern falls back to a scan.
102
+
103
+ Returns:
104
+ An analyzer name such as ``"sparsegram-3-12"`` or
105
+ ``"sparsegram-3-12-fold"``.
106
+
107
+ Note:
108
+ The name is stored in the index definition, so build a new index to
109
+ change it.
110
+ """
111
+ for name, value in (("min_gram", min_gram), ("max_gram", max_gram)):
112
+ if isinstance(value, bool) or not isinstance(value, int):
113
+ raise InfinityException(ErrorCode.INVALID_INDEX_PARAM, f"{name} should be an integer, but got {value!r}")
114
+ if min_gram < 1 or max_gram < min_gram:
115
+ raise InfinityException(ErrorCode.INVALID_INDEX_PARAM,
116
+ f"Expected 1 <= min_gram <= max_gram, but got min_gram={min_gram} and max_gram={max_gram}")
117
+ if not isinstance(fold_case, bool):
118
+ raise InfinityException(ErrorCode.INVALID_INDEX_PARAM, f"fold_case should be a boolean, but got {fold_case!r}")
119
+ name = f"sparsegram-{min_gram}-{max_gram}"
120
+ return f"{name}-fold" if fold_case else name
121
+
122
+
78
123
  class IndexInfo:
79
124
  def __init__(self, target_name: str, index_type: IndexType, params: dict = None):
80
125
  self.target_name = target_name
@@ -87,6 +132,24 @@ class IndexInfo:
87
132
  else:
88
133
  self.params = None
89
134
 
135
+ @staticmethod
136
+ def sparsegram(target_name: str,
137
+ min_gram: int = SPARSEGRAM_DEFAULT_MIN_GRAM,
138
+ max_gram: int = SPARSEGRAM_DEFAULT_MAX_GRAM,
139
+ fold_case: bool = False) -> "IndexInfo":
140
+ """Build a full-text index that makes `regex()` filters use grams.
141
+
142
+ Shorthand for ``IndexInfo(target_name, IndexType.FullText,
143
+ {"analyzer": sparsegram_analyzer(...)})``.
144
+
145
+ Example::
146
+
147
+ table.create_index("idx", IndexInfo.sparsegram("doc", fold_case=True))
148
+ table.filter(regex_filter("doc", r"(?i)colou?r of the (sky|sea)"))
149
+ """
150
+ return IndexInfo(target_name, IndexType.FullText,
151
+ {"analyzer": sparsegram_analyzer(min_gram, max_gram, fold_case)})
152
+
90
153
  def __str__(self):
91
154
  return f"IndexInfo({self.target_name}, {self.index_type}, {self.params})"
92
155
 
@@ -794,6 +794,8 @@ class table_http:
794
794
  for idx in range(len(value[key])):
795
795
  if isinstance(value[key][idx], np.ndarray):
796
796
  value[key][idx] = value[key][idx].tolist()
797
+ elif isinstance(value[key][idx], (np.integer, np.floating, np.longdouble)):
798
+ value[key][idx] = value[key][idx].item()
797
799
  elif isinstance(value[key], SparseVector):
798
800
  value[key] = value[key].to_dict()
799
801
 
@@ -37,6 +37,7 @@ import math
37
37
  import os
38
38
  import re
39
39
  import string
40
+ import unicodedata
40
41
 
41
42
  import datrie
42
43
  from hanziconv import HanziConv
@@ -65,6 +66,33 @@ _SNOWBALL_LANGUAGE_MAP = {
65
66
  "turkish": "turkish",
66
67
  }
67
68
 
69
+ # Languages tokenized with diacritics folded to ASCII and no stemming.
70
+ # SPLIT_CHAR only keeps ASCII letter runs whole, so accented words would
71
+ # otherwise be fragmented before indexing ('škola' -> 'š kola'), and
72
+ # neither language has a Snowball stemmer.
73
+ _DIACRITIC_FOLDING_LANGUAGES = {"slovak", "czech"}
74
+
75
+
76
+ def _fold_char(char: str) -> str:
77
+ # Latin-1 Supplement and Latin Extended-A letters whose NFD decomposition
78
+ # is one ASCII letter plus combining marks fold to that letter (Š -> S,
79
+ # ď -> d); everything else (æ, ø, ł, ß, non-Latin scripts) is kept. Same
80
+ # rule as RAGAnalyzer::FoldDiacritics in the C++ analyzer.
81
+ if not (0xC0 <= ord(char) < 0x180):
82
+ return char
83
+ decomposed = unicodedata.normalize("NFD", char)
84
+ base = decomposed[0]
85
+ if base.isascii() and base.isalpha() and all(unicodedata.combining(c) for c in decomposed[1:]):
86
+ return base
87
+ return char
88
+
89
+
90
+ def fold_diacritics(text: str) -> str:
91
+ """Fold Latin diacritics to ASCII: 'škola' -> 'skola'."""
92
+ if text.isascii():
93
+ return text
94
+ return "".join(_fold_char(c) for c in text)
95
+
68
96
 
69
97
  class RagTokenizer:
70
98
  def key_(self, line):
@@ -122,6 +150,7 @@ class RagTokenizer:
122
150
  self.stemmer = SnowballStemmer("english")
123
151
  self.lemmatizer = WordNetLemmatizer()
124
152
  self._use_lemmatizer = True # WordNet only supports English
153
+ self._fold_diacritics = False
125
154
 
126
155
  self.SPLIT_CHAR = r"([ ,\.<>/?;:'\[\]\\`!@#$%^&*\(\)\{\}\|_+=《》,。?、;‘’:“”【】~!¥%……()——-]+|[a-zA-Z0-9,\.-]+)"
127
156
 
@@ -163,6 +192,14 @@ class RagTokenizer:
163
192
  Case-insensitive.
164
193
  """
165
194
  lang_key = language.strip().lower()
195
+
196
+ self._fold_diacritics = lang_key in _DIACRITIC_FOLDING_LANGUAGES
197
+ if self._fold_diacritics:
198
+ # Folded to ASCII in tokenize() and left unstemmed: there is no
199
+ # Snowball stemmer for these languages (see _normalize_token).
200
+ logging.debug("Tokenizer language set to '%s' (diacritics folding, no stemming)", language)
201
+ return
202
+
166
203
  snowball_lang = _SNOWBALL_LANGUAGE_MAP.get(lang_key)
167
204
 
168
205
  if snowball_lang is not None:
@@ -389,8 +426,11 @@ class RagTokenizer:
389
426
 
390
427
  When the lemmatizer is enabled (English), applies lemmatization
391
428
  before stemming. For other Snowball-supported languages, only
392
- stemming is applied. Non-alphabetic tokens are returned as-is.
429
+ stemming is applied. Non-alphabetic tokens are returned as-is,
430
+ and so is every token of a diacritic-folding language.
393
431
  """
432
+ if self._fold_diacritics:
433
+ return t
394
434
  if re.match(r"[a-zA-Z_-]+$", t):
395
435
  if self._use_lemmatizer:
396
436
  return self.stemmer.stem(self.lemmatizer.lemmatize(t))
@@ -421,6 +461,8 @@ class RagTokenizer:
421
461
  return txt_lang_pairs
422
462
 
423
463
  def tokenize(self, line: str) -> str:
464
+ if self._fold_diacritics:
465
+ line = fold_diacritics(line)
424
466
  line = re.sub(r"\W+", " ", line)
425
467
  line = self._strQ2B(line).lower()
426
468
  line = self._tradi2simp(line)
@@ -165,8 +165,8 @@ class ThriftInfinityClient:
165
165
  # version: 0.6.13, client_version: 34
166
166
  # version: 0.6.15, client_version: 35
167
167
  # version: 0.7.0, 0.7.1, 0.7.2, client_version: 36
168
- # version: 0.7.3, client_version: 37
169
- res = self.client.Connect(ConnectRequest(client_version=37)) # 0.7.3
168
+ # version: 0.7.3, 0.7.4, client_version: 37
169
+ res = self.client.Connect(ConnectRequest(client_version=37)) # 0.7.4
170
170
  if res.error_code != 0:
171
171
  raise InfinityException(res.error_code, res.error_msg)
172
172
  self.session_id = res.session_id