fenecdb 0.1.5__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- fenecdb/__init__.py +82 -0
- fenecdb/langchain.py +306 -0
- fenecdb/llama_index.py +332 -0
- fenecdb-0.1.5.dist-info/METADATA +76 -0
- fenecdb-0.1.5.dist-info/RECORD +7 -0
- fenecdb-0.1.5.dist-info/WHEEL +5 -0
- fenecdb-0.1.5.dist-info/top_level.txt +1 -0
fenecdb/__init__.py
ADDED
|
@@ -0,0 +1,82 @@
|
|
|
1
|
+
"""fenecdb over HTTP.
|
|
2
|
+
|
|
3
|
+
A client for fenec-pg's HTTP endpoint (`fenec-pg --http <address>`), and on
|
|
4
|
+
top of it vector stores for LangChain (`fenecdb.langchain`) and LlamaIndex
|
|
5
|
+
(`fenecdb.llama_index`).
|
|
6
|
+
|
|
7
|
+
from fenecdb import Client
|
|
8
|
+
db = Client("http://127.0.0.1:8080", token="...")
|
|
9
|
+
db.query("get articles near embed $1 limit 5", [[0.1, 0.2, 0.3]])
|
|
10
|
+
|
|
11
|
+
The client is the standard library alone, as the server is: a statement is
|
|
12
|
+
one `POST /query`, several are one `POST /batch` under one lock.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
import json
|
|
18
|
+
import urllib.error
|
|
19
|
+
import urllib.request
|
|
20
|
+
from typing import Any, Iterable, Sequence
|
|
21
|
+
|
|
22
|
+
__all__ = ["Client", "FenecError", "placeholders"]
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
class FenecError(Exception):
|
|
26
|
+
"""A statement the server refused: its message, and the HTTP status."""
|
|
27
|
+
|
|
28
|
+
def __init__(self, message: str, status: int):
|
|
29
|
+
super().__init__(message)
|
|
30
|
+
self.status = status
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
class Client:
|
|
34
|
+
"""fenec-pg's HTTP endpoint, one statement a request."""
|
|
35
|
+
|
|
36
|
+
def __init__(
|
|
37
|
+
self,
|
|
38
|
+
url: str = "http://127.0.0.1:8080",
|
|
39
|
+
token: str | None = None,
|
|
40
|
+
timeout: float = 30.0,
|
|
41
|
+
):
|
|
42
|
+
self.url = url.rstrip("/")
|
|
43
|
+
self.token = token
|
|
44
|
+
self.timeout = timeout
|
|
45
|
+
|
|
46
|
+
def query(self, fenecql: str, params: Sequence[Any] | None = None) -> Any:
|
|
47
|
+
"""Runs one FenecQL statement. Values go in as `$1`, `$2`, ... and
|
|
48
|
+
never into the text."""
|
|
49
|
+
body = {"query": fenecql, "params": list(params or [])}
|
|
50
|
+
return self._post("/query", json.dumps(body).encode(), "application/json")
|
|
51
|
+
|
|
52
|
+
def batch(self, statements: Iterable[tuple[str, Sequence[Any]]]) -> Any:
|
|
53
|
+
"""Runs statements in order under one write lock, as one block:
|
|
54
|
+
their writes -- a create, a drop or a `create index` among them --
|
|
55
|
+
all land or, at the first error, none of them do. A batch holding a
|
|
56
|
+
`compact` runs each statement on its own, and what ran before an
|
|
57
|
+
error stays."""
|
|
58
|
+
lines = [json.dumps({"query": q, "params": list(p)}) for q, p in statements]
|
|
59
|
+
return self._post("/batch", "\n".join(lines).encode(), "application/x-ndjson")
|
|
60
|
+
|
|
61
|
+
def _post(self, path: str, body: bytes, content_type: str) -> Any:
|
|
62
|
+
req = urllib.request.Request(self.url + path, data=body, method="POST")
|
|
63
|
+
req.add_header("Content-Type", content_type)
|
|
64
|
+
if self.token:
|
|
65
|
+
req.add_header("Authorization", f"Bearer {self.token}")
|
|
66
|
+
try:
|
|
67
|
+
with urllib.request.urlopen(req, timeout=self.timeout) as resp:
|
|
68
|
+
return json.load(resp)
|
|
69
|
+
except urllib.error.HTTPError as e:
|
|
70
|
+
raw = e.read()
|
|
71
|
+
try:
|
|
72
|
+
message = json.loads(raw).get("error", raw.decode(errors="replace"))
|
|
73
|
+
except (ValueError, AttributeError):
|
|
74
|
+
message = raw.decode(errors="replace")
|
|
75
|
+
raise FenecError(message, e.code) from None
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def placeholders(start: int, n: int) -> str:
|
|
79
|
+
"""`$start, ..., $(start+n-1)`: a list literal's worth of parameters.
|
|
80
|
+
FenecQL binds a parameter to a value, not to a list, so `in` takes one
|
|
81
|
+
per element -- `in [$2, $3, $4]`."""
|
|
82
|
+
return ", ".join(f"${i}" for i in range(start, start + n))
|
fenecdb/langchain.py
ADDED
|
@@ -0,0 +1,306 @@
|
|
|
1
|
+
"""A LangChain vector store over fenecdb's HTTP endpoint.
|
|
2
|
+
|
|
3
|
+
from fenecdb.langchain import FenecVectorStore
|
|
4
|
+
|
|
5
|
+
store = FenecVectorStore(embeddings, "docs", url="http://127.0.0.1:8080", token="...")
|
|
6
|
+
store.add_documents(documents)
|
|
7
|
+
store.similarity_search("how do I compact", k=4)
|
|
8
|
+
|
|
9
|
+
A document is a row: its id, its text, its metadata as JSON, and its vector
|
|
10
|
+
under `@hnsw` -- over int8 or bit codes with `quant="int8"` or `"bit"`. The
|
|
11
|
+
collection is created on the first write, when the embedding's size is known.
|
|
12
|
+
A metadata field named in `metadata_fields` also gets a column of its own,
|
|
13
|
+
under `@hash`, and `filter` takes equality over those -- `filter={"source":
|
|
14
|
+
"handbook"}` is `where source = $2`, answered by the index, before the search
|
|
15
|
+
ranks what passed.
|
|
16
|
+
|
|
17
|
+
With `full_text=True` the text is indexed for BM25 as well (`@text`), and a
|
|
18
|
+
search takes `mode`: `"text"` ranks by the words alone (`match`), with no
|
|
19
|
+
embedding of the query, and `"hybrid"` ranks the words and the vector each to
|
|
20
|
+
their own depth and fuses the two rankings (`match ... near ... fuse`). A
|
|
21
|
+
retriever passes it through `search_kwargs={"mode": "hybrid"}`. The scores
|
|
22
|
+
are then BM25's, or the fusion's, not similarities.
|
|
23
|
+
|
|
24
|
+
Writing an id that is already there replaces it: the old row out and the new
|
|
25
|
+
one in, sent as one batch and run under one write lock.
|
|
26
|
+
"""
|
|
27
|
+
|
|
28
|
+
from __future__ import annotations
|
|
29
|
+
|
|
30
|
+
import json
|
|
31
|
+
import re
|
|
32
|
+
import uuid
|
|
33
|
+
from typing import Any, Callable, Iterable, Sequence
|
|
34
|
+
|
|
35
|
+
from langchain_core.documents import Document
|
|
36
|
+
from langchain_core.embeddings import Embeddings
|
|
37
|
+
from langchain_core.vectorstores import VectorStore
|
|
38
|
+
|
|
39
|
+
from fenecdb import Client, FenecError, placeholders
|
|
40
|
+
|
|
41
|
+
__all__ = ["FenecVectorStore"]
|
|
42
|
+
|
|
43
|
+
# Names go into the statement's text, so they are held to FenecQL's.
|
|
44
|
+
_NAME = re.compile(r"^[A-Za-z_][A-Za-z0-9_]*$")
|
|
45
|
+
_TYPES = {"text", "int", "float", "bool", "timestamp"}
|
|
46
|
+
_RESERVED = {"id", "lc_id", "content", "metadata", "embedding"}
|
|
47
|
+
_METRICS = {"cosine", "l2", "dot"}
|
|
48
|
+
_QUANT = {None, "int8", "bit"}
|
|
49
|
+
_MODES = {"similarity", "text", "hybrid"}
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
class FenecVectorStore(VectorStore):
|
|
53
|
+
"""A collection in fenecdb, holding LangChain documents."""
|
|
54
|
+
|
|
55
|
+
def __init__(
|
|
56
|
+
self,
|
|
57
|
+
embedding: Embeddings,
|
|
58
|
+
collection: str = "langchain",
|
|
59
|
+
*,
|
|
60
|
+
url: str = "http://127.0.0.1:8080",
|
|
61
|
+
token: str | None = None,
|
|
62
|
+
client: Client | None = None,
|
|
63
|
+
metric: str = "cosine",
|
|
64
|
+
metadata_fields: dict[str, str] | None = None,
|
|
65
|
+
quant: str | None = None,
|
|
66
|
+
full_text: bool = False,
|
|
67
|
+
):
|
|
68
|
+
if not _NAME.match(collection):
|
|
69
|
+
raise ValueError(f"not a collection name: {collection!r}")
|
|
70
|
+
if metric not in _METRICS:
|
|
71
|
+
raise ValueError(f"metric is one of {sorted(_METRICS)}, not {metric!r}")
|
|
72
|
+
if quant not in _QUANT:
|
|
73
|
+
raise ValueError(f"quant is None, 'int8' or 'bit', not {quant!r}")
|
|
74
|
+
if quant == "bit" and metric != "cosine":
|
|
75
|
+
raise ValueError("quant='bit' keeps signs, which only cosine can order by")
|
|
76
|
+
fields = dict(metadata_fields or {})
|
|
77
|
+
for name, ty in fields.items():
|
|
78
|
+
if not _NAME.match(name) or name in _RESERVED:
|
|
79
|
+
raise ValueError(f"not a metadata field this store can index: {name!r}")
|
|
80
|
+
if ty not in _TYPES:
|
|
81
|
+
raise ValueError(f"{name}: a field is one of {sorted(_TYPES)}, not {ty!r}")
|
|
82
|
+
self._embedding = embedding
|
|
83
|
+
self.collection = collection
|
|
84
|
+
self._client = client or Client(url, token)
|
|
85
|
+
self._metric = metric
|
|
86
|
+
self._fields = fields
|
|
87
|
+
self._quant = quant
|
|
88
|
+
self._full_text = full_text
|
|
89
|
+
self._dimension: int | None = None
|
|
90
|
+
|
|
91
|
+
@property
|
|
92
|
+
def embeddings(self) -> Embeddings:
|
|
93
|
+
return self._embedding
|
|
94
|
+
|
|
95
|
+
# ------------------------------------------------------------ writing
|
|
96
|
+
|
|
97
|
+
def add_texts(
|
|
98
|
+
self,
|
|
99
|
+
texts: Iterable[str],
|
|
100
|
+
metadatas: list[dict] | None = None,
|
|
101
|
+
*,
|
|
102
|
+
ids: list[str | None] | None = None,
|
|
103
|
+
**kwargs: Any,
|
|
104
|
+
) -> list[str]:
|
|
105
|
+
texts = list(texts)
|
|
106
|
+
if not texts:
|
|
107
|
+
return []
|
|
108
|
+
metadatas = list(metadatas) if metadatas else [{} for _ in texts]
|
|
109
|
+
# A document that came without an id gets one, and so does each of
|
|
110
|
+
# the documents without one among some that have theirs.
|
|
111
|
+
ids = [i or str(uuid.uuid4()) for i in (ids or [None] * len(texts))]
|
|
112
|
+
# numpy's floats are not JSON's.
|
|
113
|
+
vectors = [[float(x) for x in v] for v in self._embedding.embed_documents(texts)]
|
|
114
|
+
self._ensure(len(vectors[0]))
|
|
115
|
+
|
|
116
|
+
docs, params = [], []
|
|
117
|
+
for doc_id, text, meta, vector in zip(ids, texts, metadatas, vectors):
|
|
118
|
+
values = [doc_id, text, json.dumps(meta, default=str), vector]
|
|
119
|
+
values += [meta.get(f) for f in self._fields]
|
|
120
|
+
names = ["lc_id", "content", "metadata", "embedding", *self._fields]
|
|
121
|
+
first = len(params) + 1
|
|
122
|
+
docs.append(
|
|
123
|
+
"{" + ", ".join(f"{n}: ${first + i}" for i, n in enumerate(names)) + "}"
|
|
124
|
+
)
|
|
125
|
+
params += values
|
|
126
|
+
self._client.batch(
|
|
127
|
+
[
|
|
128
|
+
(f"del {self.collection} where lc_id in [{placeholders(1, len(ids))}]", ids),
|
|
129
|
+
(f"put {self.collection} [{', '.join(docs)}]", params),
|
|
130
|
+
]
|
|
131
|
+
)
|
|
132
|
+
return ids
|
|
133
|
+
|
|
134
|
+
def delete(self, ids: list[str] | None = None, **kwargs: Any) -> bool | None:
|
|
135
|
+
"""Deletes these ids, or every document when `ids` is None."""
|
|
136
|
+
if ids is None:
|
|
137
|
+
statement, params = f"del {self.collection}", []
|
|
138
|
+
else:
|
|
139
|
+
params = list(ids)
|
|
140
|
+
if not params:
|
|
141
|
+
return True
|
|
142
|
+
statement = (
|
|
143
|
+
f"del {self.collection} where lc_id in [{placeholders(1, len(params))}]"
|
|
144
|
+
)
|
|
145
|
+
self._run(statement, params)
|
|
146
|
+
return True
|
|
147
|
+
|
|
148
|
+
def drop_collection(self) -> None:
|
|
149
|
+
"""Drops the collection, documents and index together."""
|
|
150
|
+
self._client.query(f"drop collection if exists {self.collection}")
|
|
151
|
+
self._dimension = None
|
|
152
|
+
|
|
153
|
+
# ------------------------------------------------------------ reading
|
|
154
|
+
|
|
155
|
+
def get_by_ids(self, ids: Sequence[str], /) -> list[Document]:
|
|
156
|
+
ids = list(ids)
|
|
157
|
+
if not ids:
|
|
158
|
+
return []
|
|
159
|
+
rows = self._run(
|
|
160
|
+
f"get {self.collection} select lc_id, content, metadata "
|
|
161
|
+
f"where lc_id in [{placeholders(1, len(ids))}]",
|
|
162
|
+
ids,
|
|
163
|
+
)
|
|
164
|
+
return [self._document(r) for r in rows]
|
|
165
|
+
|
|
166
|
+
def similarity_search(
|
|
167
|
+
self, query: str, k: int = 4, filter: dict | None = None, **kwargs: Any
|
|
168
|
+
) -> list[Document]:
|
|
169
|
+
"""`mode` as `similarity_search_with_score` takes it."""
|
|
170
|
+
return [d for d, _ in self.similarity_search_with_score(query, k, filter, **kwargs)]
|
|
171
|
+
|
|
172
|
+
def similarity_search_with_score(
|
|
173
|
+
self,
|
|
174
|
+
query: str,
|
|
175
|
+
k: int = 4,
|
|
176
|
+
filter: dict | None = None,
|
|
177
|
+
*,
|
|
178
|
+
mode: str = "similarity",
|
|
179
|
+
**kwargs: Any,
|
|
180
|
+
) -> list[tuple[Document, float]]:
|
|
181
|
+
"""Documents with fenecdb's score: the cosine similarity under
|
|
182
|
+
`cosine`, higher is nearer; the distance under `l2`; the product
|
|
183
|
+
under `dot`. With `mode="text"` or `"hybrid"` (`full_text=True`)
|
|
184
|
+
BM25's score, or the fusion's."""
|
|
185
|
+
if mode not in _MODES:
|
|
186
|
+
raise ValueError(f"mode is one of {sorted(_MODES)}, not {mode!r}")
|
|
187
|
+
if mode == "similarity":
|
|
188
|
+
vector = self._embedding.embed_query(query)
|
|
189
|
+
return self.similarity_search_with_score_by_vector(vector, k, filter, **kwargs)
|
|
190
|
+
if not self._full_text:
|
|
191
|
+
raise ValueError(f"mode {mode!r} needs the store made with full_text=True")
|
|
192
|
+
if mode == "text":
|
|
193
|
+
rank, ahead = "match content $1", [query]
|
|
194
|
+
else:
|
|
195
|
+
vector = [float(x) for x in self._embedding.embed_query(query)]
|
|
196
|
+
rank, ahead = "match content $1 near embedding $2 fuse", [query, vector]
|
|
197
|
+
where, params = self._where(filter, first=len(ahead) + 1)
|
|
198
|
+
rows = self._run(
|
|
199
|
+
f"get {self.collection} select lc_id, content, metadata{where} "
|
|
200
|
+
f"{rank} limit {int(k)}",
|
|
201
|
+
[*ahead, *params],
|
|
202
|
+
)
|
|
203
|
+
return [(self._document(r), r["_score"]) for r in rows]
|
|
204
|
+
|
|
205
|
+
def similarity_search_by_vector(
|
|
206
|
+
self, embedding: list[float], k: int = 4, filter: dict | None = None, **kwargs: Any
|
|
207
|
+
) -> list[Document]:
|
|
208
|
+
pairs = self.similarity_search_with_score_by_vector(embedding, k, filter, **kwargs)
|
|
209
|
+
return [d for d, _ in pairs]
|
|
210
|
+
|
|
211
|
+
def similarity_search_with_score_by_vector(
|
|
212
|
+
self, embedding: list[float], k: int = 4, filter: dict | None = None, **kwargs: Any
|
|
213
|
+
) -> list[tuple[Document, float]]:
|
|
214
|
+
where, params = self._where(filter, first=2)
|
|
215
|
+
rows = self._run(
|
|
216
|
+
f"get {self.collection} select lc_id, content, metadata{where} "
|
|
217
|
+
f"near embedding $1 limit {int(k)}",
|
|
218
|
+
[[float(x) for x in embedding], *params],
|
|
219
|
+
)
|
|
220
|
+
return [(self._document(r), r["_score"]) for r in rows]
|
|
221
|
+
|
|
222
|
+
def _select_relevance_score_fn(self) -> Callable[[float], float]:
|
|
223
|
+
# Under `cosine` the score already is 1 - distance, LangChain's
|
|
224
|
+
# relevance; under `dot` the product is, for normalised vectors.
|
|
225
|
+
if self._metric == "l2":
|
|
226
|
+
return self._euclidean_relevance_score_fn
|
|
227
|
+
return lambda score: score
|
|
228
|
+
|
|
229
|
+
@classmethod
|
|
230
|
+
def from_texts(
|
|
231
|
+
cls,
|
|
232
|
+
texts: list[str],
|
|
233
|
+
embedding: Embeddings,
|
|
234
|
+
metadatas: list[dict] | None = None,
|
|
235
|
+
*,
|
|
236
|
+
ids: list[str] | None = None,
|
|
237
|
+
**kwargs: Any,
|
|
238
|
+
) -> "FenecVectorStore":
|
|
239
|
+
store = cls(embedding, **kwargs)
|
|
240
|
+
store.add_texts(texts, metadatas, ids=ids)
|
|
241
|
+
return store
|
|
242
|
+
|
|
243
|
+
# ------------------------------------------------------------ inside
|
|
244
|
+
|
|
245
|
+
def _ensure(self, dimension: int) -> None:
|
|
246
|
+
if self._dimension == dimension:
|
|
247
|
+
return
|
|
248
|
+
extra = "".join(f", {f} {t} @hash" for f, t in self._fields.items())
|
|
249
|
+
quant = f", quant={self._quant}" if self._quant else ""
|
|
250
|
+
content = "content text @text" if self._full_text else "content text"
|
|
251
|
+
self._client.query(
|
|
252
|
+
f"create collection if not exists {self.collection} ("
|
|
253
|
+
f"lc_id text @hash, {content}, metadata text, "
|
|
254
|
+
f"embedding vector<{dimension}> @hnsw({self._metric}{quant}){extra})"
|
|
255
|
+
)
|
|
256
|
+
if self._full_text:
|
|
257
|
+
# A collection made before without it takes the index now.
|
|
258
|
+
self._client.query(
|
|
259
|
+
f"create index if not exists on {self.collection} (content) @text"
|
|
260
|
+
)
|
|
261
|
+
self._dimension = dimension
|
|
262
|
+
|
|
263
|
+
def _run(self, statement: str, params: list) -> Any:
|
|
264
|
+
"""A statement over a collection that may not be there yet: a store
|
|
265
|
+
nothing was written to is an empty one, not an error."""
|
|
266
|
+
try:
|
|
267
|
+
return self._client.query(statement, params)
|
|
268
|
+
except FenecError as e:
|
|
269
|
+
if e.status == 404 and str(e).startswith("collection `"):
|
|
270
|
+
return []
|
|
271
|
+
raise
|
|
272
|
+
|
|
273
|
+
def _where(self, filter: dict | None, first: int) -> tuple[str, list]:
|
|
274
|
+
"""` where f = $n and g in [$m, ...]` over the indexed fields."""
|
|
275
|
+
if not filter:
|
|
276
|
+
return "", []
|
|
277
|
+
terms, params = [], []
|
|
278
|
+
for name, want in filter.items():
|
|
279
|
+
if name not in self._fields:
|
|
280
|
+
raise ValueError(
|
|
281
|
+
f"`{name}` is not in metadata_fields: only those can be filtered on"
|
|
282
|
+
)
|
|
283
|
+
if isinstance(want, dict):
|
|
284
|
+
if set(want) == {"$eq"}:
|
|
285
|
+
want = want["$eq"]
|
|
286
|
+
elif set(want) == {"$in"}:
|
|
287
|
+
want = list(want["$in"])
|
|
288
|
+
else:
|
|
289
|
+
raise ValueError(f"{name}: only $eq and $in are supported, not {want}")
|
|
290
|
+
if isinstance(want, (list, tuple, set)):
|
|
291
|
+
values = list(want)
|
|
292
|
+
n = first + len(params)
|
|
293
|
+
terms.append(f"{name} in [{placeholders(n, len(values))}]")
|
|
294
|
+
params += values
|
|
295
|
+
else:
|
|
296
|
+
terms.append(f"{name} = ${first + len(params)}")
|
|
297
|
+
params.append(want)
|
|
298
|
+
return " where " + " and ".join(terms), params
|
|
299
|
+
|
|
300
|
+
@staticmethod
|
|
301
|
+
def _document(row: dict) -> Document:
|
|
302
|
+
return Document(
|
|
303
|
+
id=row["lc_id"],
|
|
304
|
+
page_content=row["content"],
|
|
305
|
+
metadata=json.loads(row["metadata"] or "{}"),
|
|
306
|
+
)
|
fenecdb/llama_index.py
ADDED
|
@@ -0,0 +1,332 @@
|
|
|
1
|
+
"""A LlamaIndex vector store over fenecdb's HTTP endpoint.
|
|
2
|
+
|
|
3
|
+
from fenecdb.llama_index import FenecVectorStore
|
|
4
|
+
from llama_index.core import StorageContext, VectorStoreIndex
|
|
5
|
+
|
|
6
|
+
store = FenecVectorStore("docs", url="http://127.0.0.1:8080", token="...")
|
|
7
|
+
index = VectorStoreIndex.from_documents(
|
|
8
|
+
documents, storage_context=StorageContext.from_defaults(vector_store=store)
|
|
9
|
+
)
|
|
10
|
+
|
|
11
|
+
A node is a row: its id, its document's id, its text, its metadata as
|
|
12
|
+
LlamaIndex serialises it, and its vector under `@hnsw` -- over int8 or bit
|
|
13
|
+
codes with `quant="int8"` or `"bit"`. The collection is created on the first
|
|
14
|
+
write, when the embedding's size is known. A metadata field named in
|
|
15
|
+
`metadata_fields` also gets a column of its own, under `@hash`, and a query's
|
|
16
|
+
filters are answered over those -- `==`, `!=`, `<`, `<=`, `>`, `>=`, `in` and
|
|
17
|
+
`nin`, joined by `and` or `or` -- before the search ranks what passed.
|
|
18
|
+
|
|
19
|
+
With `full_text=True` the text is indexed for BM25 as well (`@text`), and a
|
|
20
|
+
query takes the modes that need it: `TEXT_SEARCH` ranks by the words alone
|
|
21
|
+
(`match`), and `HYBRID` ranks the words and the vector each to their own depth
|
|
22
|
+
and fuses the two rankings (`match ... near ... fuse`). The fusion adds ranks,
|
|
23
|
+
not scores, so `alpha` has nothing to weigh and is not read; `hybrid_top_k`,
|
|
24
|
+
or `sparse_top_k`, is how deep each side goes.
|
|
25
|
+
"""
|
|
26
|
+
|
|
27
|
+
from __future__ import annotations
|
|
28
|
+
|
|
29
|
+
import json
|
|
30
|
+
import re
|
|
31
|
+
from typing import Any, List, Optional, Sequence
|
|
32
|
+
|
|
33
|
+
from llama_index.core.bridge.pydantic import PrivateAttr
|
|
34
|
+
from llama_index.core.schema import BaseNode, MetadataMode
|
|
35
|
+
from llama_index.core.vector_stores.types import (
|
|
36
|
+
BasePydanticVectorStore,
|
|
37
|
+
FilterCondition,
|
|
38
|
+
FilterOperator,
|
|
39
|
+
MetadataFilter,
|
|
40
|
+
MetadataFilters,
|
|
41
|
+
VectorStoreQuery,
|
|
42
|
+
VectorStoreQueryMode,
|
|
43
|
+
VectorStoreQueryResult,
|
|
44
|
+
)
|
|
45
|
+
from llama_index.core.vector_stores.utils import (
|
|
46
|
+
metadata_dict_to_node,
|
|
47
|
+
node_to_metadata_dict,
|
|
48
|
+
)
|
|
49
|
+
|
|
50
|
+
from fenecdb import Client, FenecError, placeholders
|
|
51
|
+
|
|
52
|
+
__all__ = ["FenecVectorStore"]
|
|
53
|
+
|
|
54
|
+
_NAME = re.compile(r"^[A-Za-z_][A-Za-z0-9_]*$")
|
|
55
|
+
_TYPES = {"text", "int", "float", "bool", "timestamp"}
|
|
56
|
+
_RESERVED = {"id", "node_id", "ref_doc_id", "content", "metadata", "embedding"}
|
|
57
|
+
_METRICS = {"cosine", "l2", "dot"}
|
|
58
|
+
_QUANT = {None, "int8", "bit"}
|
|
59
|
+
_WORDS = {VectorStoreQueryMode.TEXT_SEARCH, VectorStoreQueryMode.HYBRID}
|
|
60
|
+
_COMPARE = {
|
|
61
|
+
FilterOperator.EQ: "=",
|
|
62
|
+
FilterOperator.NE: "!=",
|
|
63
|
+
FilterOperator.GT: ">",
|
|
64
|
+
FilterOperator.GTE: ">=",
|
|
65
|
+
FilterOperator.LT: "<",
|
|
66
|
+
FilterOperator.LTE: "<=",
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
class FenecVectorStore(BasePydanticVectorStore):
|
|
71
|
+
"""A collection in fenecdb, holding LlamaIndex nodes."""
|
|
72
|
+
|
|
73
|
+
stores_text: bool = True
|
|
74
|
+
flat_metadata: bool = False
|
|
75
|
+
|
|
76
|
+
collection: str = "llama_index"
|
|
77
|
+
url: str = "http://127.0.0.1:8080"
|
|
78
|
+
token: Optional[str] = None
|
|
79
|
+
metric: str = "cosine"
|
|
80
|
+
metadata_fields: dict = {}
|
|
81
|
+
quant: Optional[str] = None
|
|
82
|
+
full_text: bool = False
|
|
83
|
+
|
|
84
|
+
_client: Client = PrivateAttr()
|
|
85
|
+
_dimension: Optional[int] = PrivateAttr(default=None)
|
|
86
|
+
|
|
87
|
+
def __init__(
|
|
88
|
+
self,
|
|
89
|
+
collection: str = "llama_index",
|
|
90
|
+
url: str = "http://127.0.0.1:8080",
|
|
91
|
+
token: Optional[str] = None,
|
|
92
|
+
metric: str = "cosine",
|
|
93
|
+
metadata_fields: Optional[dict] = None,
|
|
94
|
+
client: Optional[Client] = None,
|
|
95
|
+
quant: Optional[str] = None,
|
|
96
|
+
full_text: bool = False,
|
|
97
|
+
**kwargs: Any,
|
|
98
|
+
):
|
|
99
|
+
if not _NAME.match(collection):
|
|
100
|
+
raise ValueError(f"not a collection name: {collection!r}")
|
|
101
|
+
if metric not in _METRICS:
|
|
102
|
+
raise ValueError(f"metric is one of {sorted(_METRICS)}, not {metric!r}")
|
|
103
|
+
if quant not in _QUANT:
|
|
104
|
+
raise ValueError(f"quant is None, 'int8' or 'bit', not {quant!r}")
|
|
105
|
+
if quant == "bit" and metric != "cosine":
|
|
106
|
+
raise ValueError("quant='bit' keeps signs, which only cosine can order by")
|
|
107
|
+
fields = dict(metadata_fields or {})
|
|
108
|
+
for name, ty in fields.items():
|
|
109
|
+
if not _NAME.match(name) or name in _RESERVED:
|
|
110
|
+
raise ValueError(f"not a metadata field this store can index: {name!r}")
|
|
111
|
+
if ty not in _TYPES:
|
|
112
|
+
raise ValueError(f"{name}: a field is one of {sorted(_TYPES)}, not {ty!r}")
|
|
113
|
+
super().__init__(
|
|
114
|
+
collection=collection,
|
|
115
|
+
url=url,
|
|
116
|
+
token=token,
|
|
117
|
+
metric=metric,
|
|
118
|
+
metadata_fields=fields,
|
|
119
|
+
quant=quant,
|
|
120
|
+
full_text=full_text,
|
|
121
|
+
**kwargs,
|
|
122
|
+
)
|
|
123
|
+
self._client = client or Client(url, token)
|
|
124
|
+
|
|
125
|
+
@classmethod
|
|
126
|
+
def class_name(cls) -> str:
|
|
127
|
+
return "FenecVectorStore"
|
|
128
|
+
|
|
129
|
+
@property
|
|
130
|
+
def client(self) -> Client:
|
|
131
|
+
return self._client
|
|
132
|
+
|
|
133
|
+
# ------------------------------------------------------------ writing
|
|
134
|
+
|
|
135
|
+
def add(self, nodes: Sequence[BaseNode], **kwargs: Any) -> List[str]:
|
|
136
|
+
"""Writes the nodes; a node id already there is replaced, the old
|
|
137
|
+
row out and the new one in as one batch under one write lock."""
|
|
138
|
+
nodes = list(nodes)
|
|
139
|
+
if not nodes:
|
|
140
|
+
return []
|
|
141
|
+
ids = [n.node_id for n in nodes]
|
|
142
|
+
vectors = [[float(x) for x in n.get_embedding()] for n in nodes]
|
|
143
|
+
self._ensure(len(vectors[0]))
|
|
144
|
+
names = ["node_id", "ref_doc_id", "content", "metadata", "embedding"]
|
|
145
|
+
names += list(self.metadata_fields)
|
|
146
|
+
docs, params = [], []
|
|
147
|
+
for node, vector in zip(nodes, vectors):
|
|
148
|
+
meta = node_to_metadata_dict(
|
|
149
|
+
node, remove_text=True, flat_metadata=self.flat_metadata
|
|
150
|
+
)
|
|
151
|
+
values = [
|
|
152
|
+
node.node_id,
|
|
153
|
+
node.ref_doc_id,
|
|
154
|
+
node.get_content(metadata_mode=MetadataMode.NONE),
|
|
155
|
+
json.dumps(meta, default=str),
|
|
156
|
+
vector,
|
|
157
|
+
]
|
|
158
|
+
values += [node.metadata.get(f) for f in self.metadata_fields]
|
|
159
|
+
first = len(params) + 1
|
|
160
|
+
docs.append(
|
|
161
|
+
"{" + ", ".join(f"{n}: ${first + i}" for i, n in enumerate(names)) + "}"
|
|
162
|
+
)
|
|
163
|
+
params += values
|
|
164
|
+
c = self.collection
|
|
165
|
+
self._client.batch(
|
|
166
|
+
[
|
|
167
|
+
(f"del {c} where node_id in [{placeholders(1, len(ids))}]", ids),
|
|
168
|
+
(f"put {c} [{', '.join(docs)}]", params),
|
|
169
|
+
]
|
|
170
|
+
)
|
|
171
|
+
return ids
|
|
172
|
+
|
|
173
|
+
def delete(self, ref_doc_id: str, **delete_kwargs: Any) -> None:
|
|
174
|
+
"""Deletes every node of a document."""
|
|
175
|
+
self._run(f"del {self.collection} where ref_doc_id = $1", [ref_doc_id])
|
|
176
|
+
|
|
177
|
+
def delete_nodes(
|
|
178
|
+
self,
|
|
179
|
+
node_ids: Optional[List[str]] = None,
|
|
180
|
+
filters: Optional[MetadataFilters] = None,
|
|
181
|
+
**delete_kwargs: Any,
|
|
182
|
+
) -> None:
|
|
183
|
+
"""Deletes these nodes, or those the filters pass. Given neither it
|
|
184
|
+
deletes nothing: everything is `clear`, and asked for by name."""
|
|
185
|
+
if node_ids is None and filters is None:
|
|
186
|
+
return
|
|
187
|
+
where, params = self._where(node_ids, None, filters, first=1)
|
|
188
|
+
self._run(f"del {self.collection}{where}", params)
|
|
189
|
+
|
|
190
|
+
def clear(self) -> None:
|
|
191
|
+
self._run(f"del {self.collection}", [])
|
|
192
|
+
|
|
193
|
+
def drop_collection(self) -> None:
|
|
194
|
+
"""Drops the collection, nodes and index together."""
|
|
195
|
+
self._client.query(f"drop collection if exists {self.collection}")
|
|
196
|
+
self._dimension = None
|
|
197
|
+
|
|
198
|
+
# ------------------------------------------------------------ reading
|
|
199
|
+
|
|
200
|
+
def get_nodes(
|
|
201
|
+
self,
|
|
202
|
+
node_ids: Optional[List[str]] = None,
|
|
203
|
+
filters: Optional[MetadataFilters] = None,
|
|
204
|
+
) -> List[BaseNode]:
|
|
205
|
+
where, params = self._where(node_ids, None, filters, first=1)
|
|
206
|
+
rows = self._run(
|
|
207
|
+
f"get {self.collection} select node_id, content, metadata{where}", params
|
|
208
|
+
)
|
|
209
|
+
return [self._node(r) for r in rows]
|
|
210
|
+
|
|
211
|
+
def query(self, query: VectorStoreQuery, **kwargs: Any) -> VectorStoreQueryResult:
|
|
212
|
+
mode = query.mode
|
|
213
|
+
if mode not in (VectorStoreQueryMode.DEFAULT, *_WORDS):
|
|
214
|
+
raise NotImplementedError(f"query mode {mode} is not supported")
|
|
215
|
+
if mode in _WORDS and not self.full_text:
|
|
216
|
+
raise ValueError(f"query mode {mode} needs the store made with full_text=True")
|
|
217
|
+
if mode in _WORDS and not query.query_str:
|
|
218
|
+
raise ValueError(f"query mode {mode} needs the query's text")
|
|
219
|
+
if mode != VectorStoreQueryMode.TEXT_SEARCH and query.query_embedding is None:
|
|
220
|
+
raise ValueError("a query embedding is needed")
|
|
221
|
+
# What the query ranks by, and its parameters ahead of the filter's.
|
|
222
|
+
rank, ahead = self._rank(query)
|
|
223
|
+
# An empty list is no restriction here, as LlamaIndex means it: its
|
|
224
|
+
# retriever sends `node_ids=[]` for a store that keeps its own text.
|
|
225
|
+
where, params = self._where(
|
|
226
|
+
query.node_ids or None, query.doc_ids or None, query.filters, first=len(ahead) + 1
|
|
227
|
+
)
|
|
228
|
+
rows = self._run(
|
|
229
|
+
f"get {self.collection} select node_id, content, metadata{where} "
|
|
230
|
+
f"{rank} limit {int(query.similarity_top_k)}",
|
|
231
|
+
[*ahead, *params],
|
|
232
|
+
)
|
|
233
|
+
return VectorStoreQueryResult(
|
|
234
|
+
nodes=[self._node(r) for r in rows],
|
|
235
|
+
similarities=[r["_score"] for r in rows],
|
|
236
|
+
ids=[r["node_id"] for r in rows],
|
|
237
|
+
)
|
|
238
|
+
|
|
239
|
+
# ------------------------------------------------------------ inside
|
|
240
|
+
|
|
241
|
+
def _rank(self, query: VectorStoreQuery) -> tuple[str, list]:
|
|
242
|
+
"""`near`, `match`, or both fused, and their parameters, $1 on."""
|
|
243
|
+
vector = [float(x) for x in query.query_embedding or []]
|
|
244
|
+
if query.mode == VectorStoreQueryMode.TEXT_SEARCH:
|
|
245
|
+
return "match content $1", [query.query_str]
|
|
246
|
+
if query.mode == VectorStoreQueryMode.HYBRID:
|
|
247
|
+
depth = query.hybrid_top_k or query.sparse_top_k
|
|
248
|
+
deep = f" candidates {int(depth)}" if depth else ""
|
|
249
|
+
return f"match content $1 near embedding $2 fuse{deep}", [query.query_str, vector]
|
|
250
|
+
return "near embedding $1", [vector]
|
|
251
|
+
|
|
252
|
+
def _ensure(self, dimension: int) -> None:
|
|
253
|
+
if self._dimension == dimension:
|
|
254
|
+
return
|
|
255
|
+
extra = "".join(f", {f} {t} @hash" for f, t in self.metadata_fields.items())
|
|
256
|
+
quant = f", quant={self.quant}" if self.quant else ""
|
|
257
|
+
content = "content text @text" if self.full_text else "content text"
|
|
258
|
+
self._client.query(
|
|
259
|
+
f"create collection if not exists {self.collection} ("
|
|
260
|
+
f"node_id text @hash, ref_doc_id text @hash, {content}, metadata text, "
|
|
261
|
+
f"embedding vector<{dimension}> @hnsw({self.metric}{quant}){extra})"
|
|
262
|
+
)
|
|
263
|
+
if self.full_text:
|
|
264
|
+
# A collection made before without it takes the index now.
|
|
265
|
+
self._client.query(
|
|
266
|
+
f"create index if not exists on {self.collection} (content) @text"
|
|
267
|
+
)
|
|
268
|
+
self._dimension = dimension
|
|
269
|
+
|
|
270
|
+
def _run(self, statement: str, params: list) -> Any:
|
|
271
|
+
"""A collection nothing was written to yet is an empty one."""
|
|
272
|
+
try:
|
|
273
|
+
return self._client.query(statement, params)
|
|
274
|
+
except FenecError as e:
|
|
275
|
+
if e.status == 404 and str(e).startswith("collection `"):
|
|
276
|
+
return []
|
|
277
|
+
raise
|
|
278
|
+
|
|
279
|
+
def _where(
|
|
280
|
+
self,
|
|
281
|
+
node_ids: Optional[List[str]],
|
|
282
|
+
doc_ids: Optional[List[str]],
|
|
283
|
+
filters: Optional[MetadataFilters],
|
|
284
|
+
first: int,
|
|
285
|
+
) -> tuple[str, list]:
|
|
286
|
+
terms, params = [], []
|
|
287
|
+
for field, values in (("node_id", node_ids), ("ref_doc_id", doc_ids)):
|
|
288
|
+
if values is not None:
|
|
289
|
+
n = first + len(params)
|
|
290
|
+
terms.append(f"{field} in [{placeholders(n, len(values))}]")
|
|
291
|
+
params += list(values)
|
|
292
|
+
if filters is not None and filters.filters:
|
|
293
|
+
terms.append(self._filters(filters, params, first))
|
|
294
|
+
if not terms:
|
|
295
|
+
return "", []
|
|
296
|
+
return " where " + " and ".join(terms), params
|
|
297
|
+
|
|
298
|
+
def _filters(self, filters: MetadataFilters, params: list, first: int) -> str:
|
|
299
|
+
joiner = {FilterCondition.AND: " and ", FilterCondition.OR: " or "}.get(
|
|
300
|
+
filters.condition
|
|
301
|
+
)
|
|
302
|
+
if joiner is None:
|
|
303
|
+
raise NotImplementedError(f"filter condition {filters.condition} is not supported")
|
|
304
|
+
parts = []
|
|
305
|
+
for f in filters.filters:
|
|
306
|
+
if isinstance(f, MetadataFilters):
|
|
307
|
+
parts.append(self._filters(f, params, first))
|
|
308
|
+
else:
|
|
309
|
+
parts.append(self._filter(f, params, first))
|
|
310
|
+
return "(" + joiner.join(parts) + ")"
|
|
311
|
+
|
|
312
|
+
def _filter(self, f: MetadataFilter, params: list, first: int) -> str:
|
|
313
|
+
if f.key not in self.metadata_fields:
|
|
314
|
+
raise ValueError(
|
|
315
|
+
f"`{f.key}` is not in metadata_fields: only those can be filtered on"
|
|
316
|
+
)
|
|
317
|
+
if f.operator in _COMPARE:
|
|
318
|
+
params.append(f.value)
|
|
319
|
+
return f"{f.key} {_COMPARE[f.operator]} ${first + len(params) - 1}"
|
|
320
|
+
if f.operator in (FilterOperator.IN, FilterOperator.NIN):
|
|
321
|
+
values = list(f.value)
|
|
322
|
+
n = first + len(params)
|
|
323
|
+
params += values
|
|
324
|
+
term = f"{f.key} in [{placeholders(n, len(values))}]"
|
|
325
|
+
return term if f.operator == FilterOperator.IN else f"not ({term})"
|
|
326
|
+
raise NotImplementedError(f"filter operator {f.operator} is not supported")
|
|
327
|
+
|
|
328
|
+
@staticmethod
|
|
329
|
+
def _node(row: dict) -> BaseNode:
|
|
330
|
+
node = metadata_dict_to_node(json.loads(row["metadata"] or "{}"))
|
|
331
|
+
node.set_content(row["content"] or "")
|
|
332
|
+
return node
|
|
@@ -0,0 +1,76 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: fenecdb
|
|
3
|
+
Version: 0.1.5
|
|
4
|
+
Summary: fenecdb over HTTP: a client, and vector stores for LangChain and LlamaIndex
|
|
5
|
+
License-Expression: Apache-2.0
|
|
6
|
+
Project-URL: Homepage, https://fenecdb.com
|
|
7
|
+
Project-URL: Documentation, https://fenecdb.com/docs/integrations
|
|
8
|
+
Project-URL: Source, https://github.com/fenecdb/fenec
|
|
9
|
+
Keywords: fenecdb,vector,database,langchain,llama-index
|
|
10
|
+
Requires-Python: >=3.10
|
|
11
|
+
Description-Content-Type: text/markdown
|
|
12
|
+
Provides-Extra: langchain
|
|
13
|
+
Requires-Dist: langchain-core>=1.0; extra == "langchain"
|
|
14
|
+
Provides-Extra: llama-index
|
|
15
|
+
Requires-Dist: llama-index-core>=0.12; extra == "llama-index"
|
|
16
|
+
Provides-Extra: test
|
|
17
|
+
Requires-Dist: pytest>=8; extra == "test"
|
|
18
|
+
Requires-Dist: pytest-asyncio>=0.23; extra == "test"
|
|
19
|
+
Requires-Dist: langchain-core>=1.0; extra == "test"
|
|
20
|
+
Requires-Dist: langchain-tests>=1.0; extra == "test"
|
|
21
|
+
Requires-Dist: llama-index-core>=0.12; extra == "test"
|
|
22
|
+
|
|
23
|
+
# fenecdb for Python
|
|
24
|
+
|
|
25
|
+
A client for fenec-pg's HTTP endpoint (`fenec-pg --http <address>`), and
|
|
26
|
+
vector stores for LangChain and LlamaIndex on top of it.
|
|
27
|
+
|
|
28
|
+
```sh
|
|
29
|
+
pip install "fenecdb[langchain] @ git+https://github.com/fenecdb/fenec#subdirectory=integrations/python"
|
|
30
|
+
# or fenecdb[llama-index]
|
|
31
|
+
```
|
|
32
|
+
|
|
33
|
+
```python
|
|
34
|
+
from fenecdb import Client
|
|
35
|
+
|
|
36
|
+
db = Client("http://127.0.0.1:8080", token="...")
|
|
37
|
+
db.query("get articles select title near embed $1 limit 5", [[0.1, 0.2, 0.3]])
|
|
38
|
+
```
|
|
39
|
+
|
|
40
|
+
```python
|
|
41
|
+
from fenecdb.langchain import FenecVectorStore
|
|
42
|
+
|
|
43
|
+
store = FenecVectorStore(embeddings, "docs", url="http://127.0.0.1:8080", token="...",
|
|
44
|
+
metadata_fields={"source": "text"})
|
|
45
|
+
store.add_documents(documents)
|
|
46
|
+
store.similarity_search("how do I compact", k=4, filter={"source": "handbook"})
|
|
47
|
+
```
|
|
48
|
+
|
|
49
|
+
```python
|
|
50
|
+
from fenecdb.llama_index import FenecVectorStore
|
|
51
|
+
from llama_index.core import StorageContext, VectorStoreIndex
|
|
52
|
+
|
|
53
|
+
store = FenecVectorStore("docs", url="http://127.0.0.1:8080", token="...")
|
|
54
|
+
index = VectorStoreIndex.from_documents(
|
|
55
|
+
documents, storage_context=StorageContext.from_defaults(vector_store=store)
|
|
56
|
+
)
|
|
57
|
+
```
|
|
58
|
+
|
|
59
|
+
A document or node is a row -- its id, text, metadata as JSON and vector
|
|
60
|
+
under `@hnsw` -- in a collection created on the first write. Fields named in
|
|
61
|
+
`metadata_fields` get columns of their own under `@hash`, and filters over
|
|
62
|
+
them are answered by the index before the search runs.
|
|
63
|
+
|
|
64
|
+
With `full_text=True` the text is indexed for BM25 too, and both stores search
|
|
65
|
+
by the words alone or by the words and the vector fused -- LangChain's
|
|
66
|
+
`mode="text"` and `mode="hybrid"`, LlamaIndex's `TEXT_SEARCH` and `HYBRID`.
|
|
67
|
+
`quant="int8"` or `"bit"` keeps the graph over codes.
|
|
68
|
+
|
|
69
|
+
```python
|
|
70
|
+
store = FenecVectorStore(embeddings, "docs", url=..., token=..., full_text=True)
|
|
71
|
+
store.similarity_search("how do I compact", k=4, mode="hybrid")
|
|
72
|
+
```
|
|
73
|
+
|
|
74
|
+
`./run-tests.sh` runs LangChain's standard vector store suite and the tests
|
|
75
|
+
LlamaIndex's own integrations run against a fenec-pg it builds and starts.
|
|
76
|
+
Full reference: https://fenecdb.com/docs/integrations
|
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
fenecdb/__init__.py,sha256=mMrd365ZJfduOH8fsNi6v7UK5txkcEKjvy07zr1-V_4,3188
|
|
2
|
+
fenecdb/langchain.py,sha256=ncPhxU2Y8L1iHi9MwMxlWNWZyyv_HmCYFKcwN1woblE,12520
|
|
3
|
+
fenecdb/llama_index.py,sha256=KnY5h43SABEw0ZbIE_hCJ3qcqzoEVAxF5rZSJsL2cMY,13532
|
|
4
|
+
fenecdb-0.1.5.dist-info/METADATA,sha256=OTBTlazouURidyjdj2l9IVcEgfba_9HvTAJjYnyACx8,2941
|
|
5
|
+
fenecdb-0.1.5.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
|
|
6
|
+
fenecdb-0.1.5.dist-info/top_level.txt,sha256=Qr23SMzjRRS9xJQGoT06bXtReVRfrjhjuMO3vEclBCA,8
|
|
7
|
+
fenecdb-0.1.5.dist-info/RECORD,,
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
fenecdb
|