infinity-sdk 0.7.2__tar.gz → 0.7.4__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {infinity_sdk-0.7.2/python/infinity_sdk/infinity_sdk.egg-info → infinity_sdk-0.7.4}/PKG-INFO +2 -2
- {infinity_sdk-0.7.2 → infinity_sdk-0.7.4}/README.md +1 -1
- {infinity_sdk-0.7.2 → infinity_sdk-0.7.4}/pyproject.toml +50 -1
- {infinity_sdk-0.7.2 → infinity_sdk-0.7.4}/python/infinity_sdk/README.md +1 -1
- {infinity_sdk-0.7.2 → infinity_sdk-0.7.4}/python/infinity_sdk/infinity/__init__.py +21 -5
- {infinity_sdk-0.7.2 → infinity_sdk-0.7.4}/python/infinity_sdk/infinity/common.py +4 -3
- {infinity_sdk-0.7.2 → infinity_sdk-0.7.4}/python/infinity_sdk/infinity/connection_pool.py +3 -2
- infinity_sdk-0.7.4/python/infinity_sdk/infinity/filter_utils.py +50 -0
- {infinity_sdk-0.7.2 → infinity_sdk-0.7.4}/python/infinity_sdk/infinity/http_utils.py +4 -2
- {infinity_sdk-0.7.2 → infinity_sdk-0.7.4}/python/infinity_sdk/infinity/huqie.txt.trie +1 -1
- {infinity_sdk-0.7.2 → infinity_sdk-0.7.4}/python/infinity_sdk/infinity/index.py +66 -2
- {infinity_sdk-0.7.2 → infinity_sdk-0.7.4}/python/infinity_sdk/infinity/infinity.py +0 -1
- {infinity_sdk-0.7.2 → infinity_sdk-0.7.4}/python/infinity_sdk/infinity/infinity_http.py +46 -34
- {infinity_sdk-0.7.2 → infinity_sdk-0.7.4}/python/infinity_sdk/infinity/rag_tokenizer.py +51 -9
- {infinity_sdk-0.7.2 → infinity_sdk-0.7.4}/python/infinity_sdk/infinity/remote_thrift/client.py +61 -19
- {infinity_sdk-0.7.2 → infinity_sdk-0.7.4}/python/infinity_sdk/infinity/remote_thrift/db.py +3 -4
- {infinity_sdk-0.7.2 → infinity_sdk-0.7.4}/python/infinity_sdk/infinity/remote_thrift/infinity.py +5 -5
- {infinity_sdk-0.7.2 → infinity_sdk-0.7.4}/python/infinity_sdk/infinity/remote_thrift/infinity_thrift_rpc/InfinityService.py +2745 -1961
- {infinity_sdk-0.7.2 → infinity_sdk-0.7.4}/python/infinity_sdk/infinity/remote_thrift/infinity_thrift_rpc/constants.py +1 -1
- infinity_sdk-0.7.4/python/infinity_sdk/infinity/remote_thrift/infinity_thrift_rpc/ttypes.py +12382 -0
- {infinity_sdk-0.7.2 → infinity_sdk-0.7.4}/python/infinity_sdk/infinity/remote_thrift/query_builder.py +45 -43
- {infinity_sdk-0.7.2 → infinity_sdk-0.7.4}/python/infinity_sdk/infinity/remote_thrift/table.py +36 -25
- {infinity_sdk-0.7.2 → infinity_sdk-0.7.4}/python/infinity_sdk/infinity/remote_thrift/types.py +63 -63
- {infinity_sdk-0.7.2 → infinity_sdk-0.7.4}/python/infinity_sdk/infinity/remote_thrift/utils.py +72 -19
- {infinity_sdk-0.7.2 → infinity_sdk-0.7.4}/python/infinity_sdk/infinity/table.py +1 -1
- {infinity_sdk-0.7.2 → infinity_sdk-0.7.4}/python/infinity_sdk/infinity/utils.py +1 -0
- {infinity_sdk-0.7.2 → infinity_sdk-0.7.4/python/infinity_sdk/infinity_sdk.egg-info}/PKG-INFO +2 -2
- {infinity_sdk-0.7.2 → infinity_sdk-0.7.4}/python/infinity_sdk/infinity_sdk.egg-info/SOURCES.txt +1 -0
- infinity_sdk-0.7.2/python/infinity_sdk/infinity/remote_thrift/infinity_thrift_rpc/ttypes.py +0 -11558
- {infinity_sdk-0.7.2 → infinity_sdk-0.7.4}/LICENSE +0 -0
- {infinity_sdk-0.7.2 → infinity_sdk-0.7.4}/python/infinity_sdk/infinity/db.py +0 -0
- {infinity_sdk-0.7.2 → infinity_sdk-0.7.4}/python/infinity_sdk/infinity/errors.py +0 -0
- {infinity_sdk-0.7.2 → infinity_sdk-0.7.4}/python/infinity_sdk/infinity/huqie.txt +0 -0
- {infinity_sdk-0.7.2 → infinity_sdk-0.7.4}/python/infinity_sdk/infinity/remote_thrift/__init__.py +0 -0
- {infinity_sdk-0.7.2 → infinity_sdk-0.7.4}/python/infinity_sdk/infinity/remote_thrift/infinity_thrift_rpc/__init__.py +0 -0
- {infinity_sdk-0.7.2 → infinity_sdk-0.7.4}/python/infinity_sdk/infinity_sdk.egg-info/dependency_links.txt +0 -0
- {infinity_sdk-0.7.2 → infinity_sdk-0.7.4}/python/infinity_sdk/infinity_sdk.egg-info/requires.txt +0 -0
- {infinity_sdk-0.7.2 → infinity_sdk-0.7.4}/python/infinity_sdk/infinity_sdk.egg-info/top_level.txt +0 -0
- {infinity_sdk-0.7.2 → infinity_sdk-0.7.4}/setup.cfg +0 -0
{infinity_sdk-0.7.2/python/infinity_sdk/infinity_sdk.egg-info → infinity_sdk-0.7.4}/PKG-INFO
RENAMED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: infinity-sdk
|
|
3
|
-
Version: 0.7.
|
|
3
|
+
Version: 0.7.4
|
|
4
4
|
Summary: infinity
|
|
5
5
|
License-Expression: Apache-2.0
|
|
6
6
|
Requires-Python: <3.14,>=3.11
|
|
@@ -97,7 +97,7 @@ Infinity supports two working modes, embedded mode and client-server mode. The f
|
|
|
97
97
|
|
|
98
98
|
2. Install the `infinity-sdk` package:
|
|
99
99
|
```bash
|
|
100
|
-
pip install infinity-sdk==0.7.
|
|
100
|
+
pip install infinity-sdk==0.7.4
|
|
101
101
|
```
|
|
102
102
|
|
|
103
103
|
3. Use Infinity to conduct a dense vector search:
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "infinity-sdk"
|
|
3
|
-
version = "0.7.
|
|
3
|
+
version = "0.7.4"
|
|
4
4
|
description = "infinity"
|
|
5
5
|
readme = "python/infinity_sdk/README.md"
|
|
6
6
|
license = "Apache-2.0"
|
|
@@ -71,6 +71,55 @@ filterwarnings = [
|
|
|
71
71
|
'ignore:function ham\(\) is deprecated:DeprecationWarning',
|
|
72
72
|
]
|
|
73
73
|
|
|
74
|
+
[tool.ruff.lint]
|
|
75
|
+
ignore = [
|
|
76
|
+
"RUF059",
|
|
77
|
+
"UP031",
|
|
78
|
+
"LOG015",
|
|
79
|
+
"BLE001",
|
|
80
|
+
"TRY002",
|
|
81
|
+
"B006",
|
|
82
|
+
"B023",
|
|
83
|
+
"S110",
|
|
84
|
+
"RUF012",
|
|
85
|
+
"RUF013",
|
|
86
|
+
"B017",
|
|
87
|
+
"C408",
|
|
88
|
+
"TRY201",
|
|
89
|
+
"SIM103",
|
|
90
|
+
"TRY203",
|
|
91
|
+
"SIM115",
|
|
92
|
+
"SIM118",
|
|
93
|
+
"EXE001",
|
|
94
|
+
"PT014",
|
|
95
|
+
"PLW0602",
|
|
96
|
+
"PERF402",
|
|
97
|
+
"PLR1722",
|
|
98
|
+
"PLW1510",
|
|
99
|
+
"SIM102",
|
|
100
|
+
"UP007",
|
|
101
|
+
"FURB132",
|
|
102
|
+
"SIM113",
|
|
103
|
+
"TRY004",
|
|
104
|
+
"B015",
|
|
105
|
+
"DTZ001",
|
|
106
|
+
"PERF102",
|
|
107
|
+
"PLW0127",
|
|
108
|
+
"SIM101",
|
|
109
|
+
"SIM210",
|
|
110
|
+
"UP022",
|
|
111
|
+
"B008",
|
|
112
|
+
"C417",
|
|
113
|
+
"C418",
|
|
114
|
+
"DTZ005",
|
|
115
|
+
"ISC004",
|
|
116
|
+
"N999",
|
|
117
|
+
"PLR0133",
|
|
118
|
+
"RUF015",
|
|
119
|
+
"SIM117",
|
|
120
|
+
"SIM211",
|
|
121
|
+
]
|
|
122
|
+
|
|
74
123
|
[tool.ruff.lint.per-file-ignores]
|
|
75
124
|
"python/infinity_embedded/local_infinity/client.py" = ["F403", "F405"]
|
|
76
125
|
"python/infinity_embedded/local_infinity/query_builder.py" = ["F403", "F405"]
|
|
@@ -63,7 +63,7 @@ Infinity supports two working modes, embedded mode and client-server mode. The f
|
|
|
63
63
|
|
|
64
64
|
2. Install the `infinity-sdk` package:
|
|
65
65
|
```bash
|
|
66
|
-
pip install infinity-sdk==0.7.
|
|
66
|
+
pip install infinity-sdk==0.7.4
|
|
67
67
|
```
|
|
68
68
|
|
|
69
69
|
3. Use Infinity to conduct a dense vector search:
|
|
@@ -17,17 +17,33 @@
|
|
|
17
17
|
# __version__ = importlib.metadata.version("infinity_sdk")
|
|
18
18
|
|
|
19
19
|
import logging
|
|
20
|
+
|
|
20
21
|
# import pkg_resources
|
|
21
22
|
# __version__ = pkg_resources.get_distribution("infinity_sdk").version
|
|
22
|
-
|
|
23
|
-
|
|
23
|
+
from infinity.common import (
|
|
24
|
+
LOCAL_HOST,
|
|
25
|
+
LOCAL_INFINITY_PATH,
|
|
26
|
+
URI,
|
|
27
|
+
InfinityException,
|
|
28
|
+
NetworkAddress,
|
|
29
|
+
)
|
|
30
|
+
from infinity.errors import ErrorCode
|
|
31
|
+
from infinity.filter_utils import quote_string_literal, regex_filter
|
|
24
32
|
from infinity.infinity import InfinityConnection
|
|
25
33
|
from infinity.remote_thrift.infinity import RemoteThriftInfinityConnection
|
|
26
|
-
from infinity.errors import ErrorCode
|
|
27
34
|
|
|
28
35
|
__all__ = [
|
|
29
|
-
"
|
|
30
|
-
"
|
|
36
|
+
"LOCAL_HOST",
|
|
37
|
+
"LOCAL_INFINITY_PATH",
|
|
38
|
+
"URI",
|
|
39
|
+
"ErrorCode",
|
|
40
|
+
"InfinityConnection",
|
|
41
|
+
"InfinityException",
|
|
42
|
+
"NetworkAddress",
|
|
43
|
+
"RemoteThriftInfinityConnection",
|
|
44
|
+
"connect",
|
|
45
|
+
"quote_string_literal",
|
|
46
|
+
"regex_filter"
|
|
31
47
|
]
|
|
32
48
|
|
|
33
49
|
|
|
@@ -11,9 +11,10 @@
|
|
|
11
11
|
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
12
12
|
# See the License for the specific language governing permissions and
|
|
13
13
|
# limitations under the License.
|
|
14
|
+
from dataclasses import dataclass, field
|
|
14
15
|
from pathlib import Path
|
|
15
16
|
from typing import Union
|
|
16
|
-
|
|
17
|
+
|
|
17
18
|
import numpy as np
|
|
18
19
|
|
|
19
20
|
|
|
@@ -102,13 +103,13 @@ LOCAL_HOST = NetworkAddress("127.0.0.1", 23817)
|
|
|
102
103
|
LOCAL_INFINITY_PATH = "/var/infinity"
|
|
103
104
|
|
|
104
105
|
|
|
105
|
-
class ConflictType
|
|
106
|
+
class ConflictType:
|
|
106
107
|
Ignore = 0
|
|
107
108
|
Error = 1
|
|
108
109
|
Replace = 2
|
|
109
110
|
|
|
110
111
|
|
|
111
|
-
class SortType
|
|
112
|
+
class SortType:
|
|
112
113
|
Asc = 0
|
|
113
114
|
Desc = 1
|
|
114
115
|
|
|
@@ -12,13 +12,14 @@
|
|
|
12
12
|
# See the License for the specific language governing permissions and
|
|
13
13
|
# limitations under the License.
|
|
14
14
|
|
|
15
|
+
import logging
|
|
15
16
|
from threading import Lock
|
|
17
|
+
|
|
16
18
|
import infinity
|
|
17
19
|
from infinity.common import NetworkAddress
|
|
18
|
-
import logging
|
|
19
20
|
|
|
20
21
|
|
|
21
|
-
class ConnectionPool
|
|
22
|
+
class ConnectionPool:
|
|
22
23
|
def __init__(self, uri=NetworkAddress("127.0.0.1", 23817), max_size=16):
|
|
23
24
|
self.uri_ = uri
|
|
24
25
|
self.max_size_ = max_size
|
|
@@ -0,0 +1,50 @@
|
|
|
1
|
+
# Copyright(C) 2026 InfiniFlow, Inc. All rights reserved.
|
|
2
|
+
#
|
|
3
|
+
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
4
|
+
# you may not use this file except in compliance with the License.
|
|
5
|
+
# You may obtain a copy of the License at
|
|
6
|
+
#
|
|
7
|
+
# https://www.apache.org/licenses/LICENSE-2.0
|
|
8
|
+
#
|
|
9
|
+
# Unless required by applicable law or agreed to in writing, software
|
|
10
|
+
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
11
|
+
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
12
|
+
# See the License for the specific language governing permissions and
|
|
13
|
+
# limitations under the License.
|
|
14
|
+
|
|
15
|
+
"""Helpers that build filter expressions."""
|
|
16
|
+
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
__all__ = ["quote_string_literal", "regex_filter"]
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def quote_string_literal(value: str) -> str:
|
|
23
|
+
"""Quote a value as a string literal for a filter expression.
|
|
24
|
+
|
|
25
|
+
A backslash is passed through unchanged, which is what a regular expression
|
|
26
|
+
needs, so only the quote itself has to be doubled.
|
|
27
|
+
"""
|
|
28
|
+
if not isinstance(value, str):
|
|
29
|
+
raise TypeError(f"Expected a string, but got {type(value).__name__}")
|
|
30
|
+
return "'" + value.replace("'", "''") + "'"
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def regex_filter(column: str, pattern: str) -> str:
|
|
34
|
+
"""Build a ``regex(column, pattern)`` filter.
|
|
35
|
+
|
|
36
|
+
The pattern is evaluated by RE2 on the rows the filter is applied to. When
|
|
37
|
+
``column`` has a full-text index built with a sparse gram analyzer
|
|
38
|
+
(``sparsegram-3-12``, optionally ``-fold``), the server additionally
|
|
39
|
+
narrows those rows with the literals the pattern proves mandatory before
|
|
40
|
+
the regular expression runs, so the regular expression only sees
|
|
41
|
+
candidates. The narrowing never removes a row the pattern would have
|
|
42
|
+
matched, and a column without such an index keeps a plain scan.
|
|
43
|
+
|
|
44
|
+
Example::
|
|
45
|
+
|
|
46
|
+
table.filter(regex_filter("doc", r"colou?r of the (sky|sea)"))
|
|
47
|
+
"""
|
|
48
|
+
if not isinstance(column, str) or not column:
|
|
49
|
+
raise ValueError("column must be a non-empty string")
|
|
50
|
+
return f"regex({column}, {quote_string_literal(pattern)})"
|
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
import ast
|
|
2
2
|
import re
|
|
3
3
|
from enum import Enum
|
|
4
|
+
|
|
4
5
|
from infinity.common import ConflictType
|
|
5
6
|
from infinity.table import ExplainType
|
|
6
7
|
|
|
@@ -65,9 +66,10 @@ functions = [
|
|
|
65
66
|
"ltrim",
|
|
66
67
|
"rtrim",
|
|
67
68
|
"reverse",
|
|
69
|
+
"regex",
|
|
68
70
|
]
|
|
69
71
|
|
|
70
|
-
bool_functions = ["filter_text", "filter_fulltext", "or", "and", "not"]
|
|
72
|
+
bool_functions = ["filter_text", "filter_fulltext", "or", "and", "not", "regex"]
|
|
71
73
|
|
|
72
74
|
|
|
73
75
|
def function_return_type(function_name, param_type):
|
|
@@ -78,7 +80,7 @@ def function_return_type(function_name, param_type):
|
|
|
78
80
|
return param_type
|
|
79
81
|
else:
|
|
80
82
|
return 'Float64'
|
|
81
|
-
elif function_name in ["filter_text", "filter_fulltext", "or", "and", "not"]:
|
|
83
|
+
elif function_name in ["filter_text", "filter_fulltext", "or", "and", "not", "regex"]:
|
|
82
84
|
return 'boolean'
|
|
83
85
|
elif function_name == "trunc":
|
|
84
86
|
return 'string'
|
|
@@ -1,4 +1,4 @@
|
|
|
1
1
|
[diffend] Oversized file quarantined before diffing.
|
|
2
|
-
name: infinity_sdk-0.7.
|
|
2
|
+
name: infinity_sdk-0.7.4/python/infinity_sdk/infinity/huqie.txt.trie
|
|
3
3
|
size: 54775939 bytes
|
|
4
4
|
sha256: 32ab213044af4aee58c46b4bfa4d2b61ba740ce8c894bb96e80fad00f145bcc4
|
|
@@ -14,11 +14,12 @@
|
|
|
14
14
|
|
|
15
15
|
from enum import Enum
|
|
16
16
|
|
|
17
|
-
|
|
17
|
+
from sqlglot import maybe_parse
|
|
18
|
+
|
|
18
19
|
from infinity.common import InfinityException
|
|
19
20
|
from infinity.errors import ErrorCode
|
|
21
|
+
from infinity.remote_thrift.infinity_thrift_rpc import ttypes
|
|
20
22
|
from infinity.remote_thrift.utils import parse_expr
|
|
21
|
-
from sqlglot import maybe_parse
|
|
22
23
|
|
|
23
24
|
|
|
24
25
|
class IndexType(Enum):
|
|
@@ -74,6 +75,51 @@ class InitParameter:
|
|
|
74
75
|
return ttypes.InitParameter(self.param_name, self.param_value)
|
|
75
76
|
|
|
76
77
|
|
|
78
|
+
SPARSEGRAM_DEFAULT_MIN_GRAM = 3
|
|
79
|
+
SPARSEGRAM_DEFAULT_MAX_GRAM = 12
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def sparsegram_analyzer(min_gram: int = SPARSEGRAM_DEFAULT_MIN_GRAM,
|
|
83
|
+
max_gram: int = SPARSEGRAM_DEFAULT_MAX_GRAM,
|
|
84
|
+
fold_case: bool = False) -> str:
|
|
85
|
+
"""Build the analyzer name of a sparse gram full-text index.
|
|
86
|
+
|
|
87
|
+
The analyzer emits content-defined n-grams of the whole value instead of
|
|
88
|
+
tokens, which is what lets ``regex(column, pattern)`` filters use the index:
|
|
89
|
+
the literals the pattern proves mandatory are turned into gram lookups
|
|
90
|
+
before the regular expression runs, so the regular expression only verifies
|
|
91
|
+
candidates. Chinese, Japanese and Korean values additionally get one and two
|
|
92
|
+
character grams, so a single character or a two character word can narrow as
|
|
93
|
+
well.
|
|
94
|
+
|
|
95
|
+
Args:
|
|
96
|
+
min_gram: Shortest window considered, in characters. Defaults to 3.
|
|
97
|
+
max_gram: Longest window considered, in characters. Defaults to 12.
|
|
98
|
+
Raising it adds rare, high information grams for little index space.
|
|
99
|
+
fold_case: Emit the ASCII-lowercased form of every gram too. Required for
|
|
100
|
+
a case insensitive pattern such as ``(?i)nobel prize`` to use the
|
|
101
|
+
index at all; without it such a pattern falls back to a scan.
|
|
102
|
+
|
|
103
|
+
Returns:
|
|
104
|
+
An analyzer name such as ``"sparsegram-3-12"`` or
|
|
105
|
+
``"sparsegram-3-12-fold"``.
|
|
106
|
+
|
|
107
|
+
Note:
|
|
108
|
+
The name is stored in the index definition, so build a new index to
|
|
109
|
+
change it.
|
|
110
|
+
"""
|
|
111
|
+
for name, value in (("min_gram", min_gram), ("max_gram", max_gram)):
|
|
112
|
+
if isinstance(value, bool) or not isinstance(value, int):
|
|
113
|
+
raise InfinityException(ErrorCode.INVALID_INDEX_PARAM, f"{name} should be an integer, but got {value!r}")
|
|
114
|
+
if min_gram < 1 or max_gram < min_gram:
|
|
115
|
+
raise InfinityException(ErrorCode.INVALID_INDEX_PARAM,
|
|
116
|
+
f"Expected 1 <= min_gram <= max_gram, but got min_gram={min_gram} and max_gram={max_gram}")
|
|
117
|
+
if not isinstance(fold_case, bool):
|
|
118
|
+
raise InfinityException(ErrorCode.INVALID_INDEX_PARAM, f"fold_case should be a boolean, but got {fold_case!r}")
|
|
119
|
+
name = f"sparsegram-{min_gram}-{max_gram}"
|
|
120
|
+
return f"{name}-fold" if fold_case else name
|
|
121
|
+
|
|
122
|
+
|
|
77
123
|
class IndexInfo:
|
|
78
124
|
def __init__(self, target_name: str, index_type: IndexType, params: dict = None):
|
|
79
125
|
self.target_name = target_name
|
|
@@ -86,6 +132,24 @@ class IndexInfo:
|
|
|
86
132
|
else:
|
|
87
133
|
self.params = None
|
|
88
134
|
|
|
135
|
+
@staticmethod
|
|
136
|
+
def sparsegram(target_name: str,
|
|
137
|
+
min_gram: int = SPARSEGRAM_DEFAULT_MIN_GRAM,
|
|
138
|
+
max_gram: int = SPARSEGRAM_DEFAULT_MAX_GRAM,
|
|
139
|
+
fold_case: bool = False) -> "IndexInfo":
|
|
140
|
+
"""Build a full-text index that makes `regex()` filters use grams.
|
|
141
|
+
|
|
142
|
+
Shorthand for ``IndexInfo(target_name, IndexType.FullText,
|
|
143
|
+
{"analyzer": sparsegram_analyzer(...)})``.
|
|
144
|
+
|
|
145
|
+
Example::
|
|
146
|
+
|
|
147
|
+
table.create_index("idx", IndexInfo.sparsegram("doc", fold_case=True))
|
|
148
|
+
table.filter(regex_filter("doc", r"(?i)colou?r of the (sky|sea)"))
|
|
149
|
+
"""
|
|
150
|
+
return IndexInfo(target_name, IndexType.FullText,
|
|
151
|
+
{"analyzer": sparsegram_analyzer(min_gram, max_gram, fold_case)})
|
|
152
|
+
|
|
89
153
|
def __str__(self):
|
|
90
154
|
return f"IndexInfo({self.target_name}, {self.index_type}, {self.params})"
|
|
91
155
|
|
|
@@ -1,28 +1,45 @@
|
|
|
1
|
+
import ast
|
|
2
|
+
import logging
|
|
1
3
|
import re
|
|
2
4
|
import time
|
|
5
|
+
from collections import namedtuple
|
|
6
|
+
from dataclasses import dataclass
|
|
7
|
+
from typing import Any
|
|
3
8
|
|
|
4
|
-
import requests
|
|
5
|
-
import logging
|
|
6
|
-
import ast
|
|
7
|
-
from numpy import dtype
|
|
8
|
-
from .http_utils import (
|
|
9
|
-
baseHeader, baseResponse, baseData, default_url, baseCreateOptions, baseDropOptions,
|
|
10
|
-
index_type_transfrom, ExplainType_transfrom, is_list, is_date, is_time, is_datetime,
|
|
11
|
-
is_sparse, str2sparse, type_to_dtype, function_return_type, is_float, functions, bool_functions
|
|
12
|
-
)
|
|
13
|
-
|
|
14
|
-
from infinity.common import ConflictType, InfinityException, SparseVector, SortType, FDE
|
|
15
|
-
from typing import Optional, Any
|
|
16
|
-
from infinity.errors import ErrorCode
|
|
17
|
-
from infinity.utils import deprecated_api
|
|
18
9
|
import numpy as np
|
|
19
10
|
import pandas as pd
|
|
20
11
|
import polars as pl
|
|
21
12
|
import pyarrow as pa
|
|
13
|
+
import requests
|
|
14
|
+
from numpy import dtype
|
|
15
|
+
|
|
16
|
+
from infinity.common import FDE, ConflictType, InfinityException, SortType, SparseVector
|
|
17
|
+
from infinity.errors import ErrorCode
|
|
22
18
|
from infinity.table import ExplainType
|
|
23
|
-
from
|
|
24
|
-
|
|
25
|
-
from
|
|
19
|
+
from infinity.utils import deprecated_api
|
|
20
|
+
|
|
21
|
+
from .http_utils import (
|
|
22
|
+
ExplainType_transfrom,
|
|
23
|
+
baseCreateOptions,
|
|
24
|
+
baseData,
|
|
25
|
+
baseDropOptions,
|
|
26
|
+
baseHeader,
|
|
27
|
+
baseResponse,
|
|
28
|
+
bool_functions,
|
|
29
|
+
default_url,
|
|
30
|
+
function_return_type,
|
|
31
|
+
functions,
|
|
32
|
+
index_type_transfrom,
|
|
33
|
+
is_date,
|
|
34
|
+
is_datetime,
|
|
35
|
+
is_float,
|
|
36
|
+
is_list,
|
|
37
|
+
is_sparse,
|
|
38
|
+
is_time,
|
|
39
|
+
str2sparse,
|
|
40
|
+
type_to_dtype,
|
|
41
|
+
)
|
|
42
|
+
|
|
26
43
|
|
|
27
44
|
def is_json_function(col_name: str) -> bool:
|
|
28
45
|
"""Check if column name is a JSON extraction function"""
|
|
@@ -40,7 +57,7 @@ def is_json_function(col_name: str) -> bool:
|
|
|
40
57
|
return any(col_name_lower.startswith(func) for func in json_functions)
|
|
41
58
|
|
|
42
59
|
|
|
43
|
-
def _parse_cast_target_type(cast_expr: str) ->
|
|
60
|
+
def _parse_cast_target_type(cast_expr: str) -> str | None:
|
|
44
61
|
if not cast_expr.lower().startswith("cast("):
|
|
45
62
|
return None
|
|
46
63
|
|
|
@@ -180,7 +197,6 @@ class http_network_util:
|
|
|
180
197
|
pass
|
|
181
198
|
|
|
182
199
|
logging.debug("----------------------------------------------")
|
|
183
|
-
return
|
|
184
200
|
|
|
185
201
|
def get_database_result(self, resp, expect={}):
|
|
186
202
|
try:
|
|
@@ -778,6 +794,8 @@ class table_http:
|
|
|
778
794
|
for idx in range(len(value[key])):
|
|
779
795
|
if isinstance(value[key][idx], np.ndarray):
|
|
780
796
|
value[key][idx] = value[key][idx].tolist()
|
|
797
|
+
elif isinstance(value[key][idx], (np.integer, np.floating, np.longdouble)):
|
|
798
|
+
value[key][idx] = value[key][idx].item()
|
|
781
799
|
elif isinstance(value[key], SparseVector):
|
|
782
800
|
value[key] = value[key].to_dict()
|
|
783
801
|
|
|
@@ -1044,7 +1062,7 @@ class table_http_result:
|
|
|
1044
1062
|
self._highlight = highlight
|
|
1045
1063
|
return self
|
|
1046
1064
|
|
|
1047
|
-
def sort(self, order_by_expr_list:
|
|
1065
|
+
def sort(self, order_by_expr_list: list[list[str, SortType]] | None):
|
|
1048
1066
|
for order_by_expr in order_by_expr_list:
|
|
1049
1067
|
tmp = {}
|
|
1050
1068
|
if len(order_by_expr) != 2:
|
|
@@ -1084,7 +1102,7 @@ class table_http_result:
|
|
|
1084
1102
|
self._option = option
|
|
1085
1103
|
return self
|
|
1086
1104
|
|
|
1087
|
-
def match_text(self, fields: str, query: str, topn: int, opt_params:
|
|
1105
|
+
def match_text(self, fields: str, query: str, topn: int, opt_params: dict | None = None):
|
|
1088
1106
|
tmp_match_expr = {"match_method": "text", "fields": fields, "matching_text": query, "topn": topn}
|
|
1089
1107
|
if opt_params is not None:
|
|
1090
1108
|
tmp_match_expr["params"] = opt_params
|
|
@@ -1096,7 +1114,7 @@ class table_http_result:
|
|
|
1096
1114
|
return self.match_text(*args, **kwargs)
|
|
1097
1115
|
|
|
1098
1116
|
def match_tensor(self, column_name: str, query_data, query_data_type: str, topn: int,
|
|
1099
|
-
extra_option:
|
|
1117
|
+
extra_option: dict | None = None):
|
|
1100
1118
|
tmp_match_tensor = {"match_method": "tensor", "field": column_name, "query_tensor": query_data,
|
|
1101
1119
|
"element_type": query_data_type, "topn": topn}
|
|
1102
1120
|
if extra_option is not None:
|
|
@@ -1105,7 +1123,7 @@ class table_http_result:
|
|
|
1105
1123
|
return self
|
|
1106
1124
|
|
|
1107
1125
|
def match_sparse(self, vector_column_name: str, sparse_data: SparseVector | dict, distance_type: str, topn: int,
|
|
1108
|
-
opt_params:
|
|
1126
|
+
opt_params: dict | None = None):
|
|
1109
1127
|
tmp_match_sparse = {"match_method": "sparse", "fields": vector_column_name,
|
|
1110
1128
|
"query_vector": sparse_data.to_dict(), "metric_type": distance_type, "topn": topn}
|
|
1111
1129
|
if opt_params is not None:
|
|
@@ -1118,7 +1136,7 @@ class table_http_result:
|
|
|
1118
1136
|
return self
|
|
1119
1137
|
|
|
1120
1138
|
def match_dense(self, fields: str, query_vector: list, element_type: str, metric_type: str, top_k: int,
|
|
1121
|
-
opt_params:
|
|
1139
|
+
opt_params: dict | None = None):
|
|
1122
1140
|
tmp_match_dense = {"match_method": "dense", "fields": fields, "query_vector": query_vector,
|
|
1123
1141
|
"element_type": element_type, "metric_type": metric_type, "topn": top_k}
|
|
1124
1142
|
if opt_params is not None:
|
|
@@ -1130,7 +1148,7 @@ class table_http_result:
|
|
|
1130
1148
|
deprecated_api("knn is deprecated, please use match_dense instead")
|
|
1131
1149
|
return self.match_dense(*args, **kwargs)
|
|
1132
1150
|
|
|
1133
|
-
def fusion(self, method: str, topn: int, fusion_params:
|
|
1151
|
+
def fusion(self, method: str, topn: int, fusion_params: dict | None = None):
|
|
1134
1152
|
tmp_fusion_expr = {"fusion_method": method, "topn": topn}
|
|
1135
1153
|
if method == "match_tensor":
|
|
1136
1154
|
tmp_new_params = {"field": fusion_params["field"], "query_tensor": fusion_params["query_tensor"],
|
|
@@ -1184,18 +1202,12 @@ class table_http_result:
|
|
|
1184
1202
|
if len(tup) == line_i + 1:
|
|
1185
1203
|
continue
|
|
1186
1204
|
|
|
1187
|
-
if v is None:
|
|
1188
|
-
new_tup = tup + (v,)
|
|
1189
|
-
elif isinstance(v, (int, float)):
|
|
1205
|
+
if v is None or isinstance(v, (int, float)):
|
|
1190
1206
|
new_tup = tup + (v,)
|
|
1191
1207
|
elif is_list(v) and not is_json_function(col_name):
|
|
1192
1208
|
# Don't parse lists for JSON extraction functions - keep them as strings
|
|
1193
1209
|
new_tup = tup + (ast.literal_eval(v),)
|
|
1194
|
-
elif is_date(v):
|
|
1195
|
-
new_tup = tup + (v,)
|
|
1196
|
-
elif is_time(v):
|
|
1197
|
-
new_tup = tup + (v,)
|
|
1198
|
-
elif is_datetime(v):
|
|
1210
|
+
elif is_date(v) or is_time(v) or is_datetime(v):
|
|
1199
1211
|
new_tup = tup + (v,)
|
|
1200
1212
|
elif is_sparse(v): # sparse vector
|
|
1201
1213
|
sparse_vec = str2sparse(v)
|
|
@@ -1312,7 +1324,7 @@ class table_http_result:
|
|
|
1312
1324
|
|
|
1313
1325
|
|
|
1314
1326
|
@dataclass
|
|
1315
|
-
class database_result
|
|
1327
|
+
class database_result:
|
|
1316
1328
|
def __init__(self, list=[], database_name: str = "", error_code=ErrorCode.OK, columns=[], index_names=[],
|
|
1317
1329
|
node_name="", node_role="", node_status="", index_type=None, index_comment=None, deleted_rows=0,
|
|
1318
1330
|
data={}, nodes=[], error_msg="", snapshots=[], config_value: Any = None):
|