fenecdb 0.1.5__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- fenecdb-0.1.5/PKG-INFO +76 -0
- fenecdb-0.1.5/README.md +54 -0
- fenecdb-0.1.5/fenecdb/__init__.py +82 -0
- fenecdb-0.1.5/fenecdb/langchain.py +306 -0
- fenecdb-0.1.5/fenecdb/llama_index.py +332 -0
- fenecdb-0.1.5/fenecdb.egg-info/PKG-INFO +76 -0
- fenecdb-0.1.5/fenecdb.egg-info/SOURCES.txt +13 -0
- fenecdb-0.1.5/fenecdb.egg-info/dependency_links.txt +1 -0
- fenecdb-0.1.5/fenecdb.egg-info/requires.txt +13 -0
- fenecdb-0.1.5/fenecdb.egg-info/top_level.txt +1 -0
- fenecdb-0.1.5/pyproject.toml +37 -0
- fenecdb-0.1.5/setup.cfg +4 -0
- fenecdb-0.1.5/tests/test_client.py +36 -0
- fenecdb-0.1.5/tests/test_langchain.py +144 -0
- fenecdb-0.1.5/tests/test_llama_index.py +209 -0
fenecdb-0.1.5/PKG-INFO
ADDED
|
@@ -0,0 +1,76 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: fenecdb
|
|
3
|
+
Version: 0.1.5
|
|
4
|
+
Summary: fenecdb over HTTP: a client, and vector stores for LangChain and LlamaIndex
|
|
5
|
+
License-Expression: Apache-2.0
|
|
6
|
+
Project-URL: Homepage, https://fenecdb.com
|
|
7
|
+
Project-URL: Documentation, https://fenecdb.com/docs/integrations
|
|
8
|
+
Project-URL: Source, https://github.com/fenecdb/fenec
|
|
9
|
+
Keywords: fenecdb,vector,database,langchain,llama-index
|
|
10
|
+
Requires-Python: >=3.10
|
|
11
|
+
Description-Content-Type: text/markdown
|
|
12
|
+
Provides-Extra: langchain
|
|
13
|
+
Requires-Dist: langchain-core>=1.0; extra == "langchain"
|
|
14
|
+
Provides-Extra: llama-index
|
|
15
|
+
Requires-Dist: llama-index-core>=0.12; extra == "llama-index"
|
|
16
|
+
Provides-Extra: test
|
|
17
|
+
Requires-Dist: pytest>=8; extra == "test"
|
|
18
|
+
Requires-Dist: pytest-asyncio>=0.23; extra == "test"
|
|
19
|
+
Requires-Dist: langchain-core>=1.0; extra == "test"
|
|
20
|
+
Requires-Dist: langchain-tests>=1.0; extra == "test"
|
|
21
|
+
Requires-Dist: llama-index-core>=0.12; extra == "test"
|
|
22
|
+
|
|
23
|
+
# fenecdb for Python
|
|
24
|
+
|
|
25
|
+
A client for fenec-pg's HTTP endpoint (`fenec-pg --http <address>`), and
|
|
26
|
+
vector stores for LangChain and LlamaIndex on top of it.
|
|
27
|
+
|
|
28
|
+
```sh
|
|
29
|
+
pip install "fenecdb[langchain] @ git+https://github.com/fenecdb/fenec#subdirectory=integrations/python"
|
|
30
|
+
# or fenecdb[llama-index]
|
|
31
|
+
```
|
|
32
|
+
|
|
33
|
+
```python
|
|
34
|
+
from fenecdb import Client
|
|
35
|
+
|
|
36
|
+
db = Client("http://127.0.0.1:8080", token="...")
|
|
37
|
+
db.query("get articles select title near embed $1 limit 5", [[0.1, 0.2, 0.3]])
|
|
38
|
+
```
|
|
39
|
+
|
|
40
|
+
```python
|
|
41
|
+
from fenecdb.langchain import FenecVectorStore
|
|
42
|
+
|
|
43
|
+
store = FenecVectorStore(embeddings, "docs", url="http://127.0.0.1:8080", token="...",
|
|
44
|
+
metadata_fields={"source": "text"})
|
|
45
|
+
store.add_documents(documents)
|
|
46
|
+
store.similarity_search("how do I compact", k=4, filter={"source": "handbook"})
|
|
47
|
+
```
|
|
48
|
+
|
|
49
|
+
```python
|
|
50
|
+
from fenecdb.llama_index import FenecVectorStore
|
|
51
|
+
from llama_index.core import StorageContext, VectorStoreIndex
|
|
52
|
+
|
|
53
|
+
store = FenecVectorStore("docs", url="http://127.0.0.1:8080", token="...")
|
|
54
|
+
index = VectorStoreIndex.from_documents(
|
|
55
|
+
documents, storage_context=StorageContext.from_defaults(vector_store=store)
|
|
56
|
+
)
|
|
57
|
+
```
|
|
58
|
+
|
|
59
|
+
A document or node is a row -- its id, text, metadata as JSON and vector
|
|
60
|
+
under `@hnsw` -- in a collection created on the first write. Fields named in
|
|
61
|
+
`metadata_fields` get columns of their own under `@hash`, and filters over
|
|
62
|
+
them are answered by the index before the search runs.
|
|
63
|
+
|
|
64
|
+
With `full_text=True` the text is indexed for BM25 too, and both stores search
|
|
65
|
+
by the words alone or by the words and the vector fused -- LangChain's
|
|
66
|
+
`mode="text"` and `mode="hybrid"`, LlamaIndex's `TEXT_SEARCH` and `HYBRID`.
|
|
67
|
+
`quant="int8"` or `"bit"` keeps the graph over codes.
|
|
68
|
+
|
|
69
|
+
```python
|
|
70
|
+
store = FenecVectorStore(embeddings, "docs", url=..., token=..., full_text=True)
|
|
71
|
+
store.similarity_search("how do I compact", k=4, mode="hybrid")
|
|
72
|
+
```
|
|
73
|
+
|
|
74
|
+
`./run-tests.sh` runs LangChain's standard vector store suite and the tests
|
|
75
|
+
LlamaIndex's own integrations run against a fenec-pg it builds and starts.
|
|
76
|
+
Full reference: https://fenecdb.com/docs/integrations
|
fenecdb-0.1.5/README.md
ADDED
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
# fenecdb for Python
|
|
2
|
+
|
|
3
|
+
A client for fenec-pg's HTTP endpoint (`fenec-pg --http <address>`), and
|
|
4
|
+
vector stores for LangChain and LlamaIndex on top of it.
|
|
5
|
+
|
|
6
|
+
```sh
|
|
7
|
+
pip install "fenecdb[langchain] @ git+https://github.com/fenecdb/fenec#subdirectory=integrations/python"
|
|
8
|
+
# or fenecdb[llama-index]
|
|
9
|
+
```
|
|
10
|
+
|
|
11
|
+
```python
|
|
12
|
+
from fenecdb import Client
|
|
13
|
+
|
|
14
|
+
db = Client("http://127.0.0.1:8080", token="...")
|
|
15
|
+
db.query("get articles select title near embed $1 limit 5", [[0.1, 0.2, 0.3]])
|
|
16
|
+
```
|
|
17
|
+
|
|
18
|
+
```python
|
|
19
|
+
from fenecdb.langchain import FenecVectorStore
|
|
20
|
+
|
|
21
|
+
store = FenecVectorStore(embeddings, "docs", url="http://127.0.0.1:8080", token="...",
|
|
22
|
+
metadata_fields={"source": "text"})
|
|
23
|
+
store.add_documents(documents)
|
|
24
|
+
store.similarity_search("how do I compact", k=4, filter={"source": "handbook"})
|
|
25
|
+
```
|
|
26
|
+
|
|
27
|
+
```python
|
|
28
|
+
from fenecdb.llama_index import FenecVectorStore
|
|
29
|
+
from llama_index.core import StorageContext, VectorStoreIndex
|
|
30
|
+
|
|
31
|
+
store = FenecVectorStore("docs", url="http://127.0.0.1:8080", token="...")
|
|
32
|
+
index = VectorStoreIndex.from_documents(
|
|
33
|
+
documents, storage_context=StorageContext.from_defaults(vector_store=store)
|
|
34
|
+
)
|
|
35
|
+
```
|
|
36
|
+
|
|
37
|
+
A document or node is a row -- its id, text, metadata as JSON and vector
|
|
38
|
+
under `@hnsw` -- in a collection created on the first write. Fields named in
|
|
39
|
+
`metadata_fields` get columns of their own under `@hash`, and filters over
|
|
40
|
+
them are answered by the index before the search runs.
|
|
41
|
+
|
|
42
|
+
With `full_text=True` the text is indexed for BM25 too, and both stores search
|
|
43
|
+
by the words alone or by the words and the vector fused -- LangChain's
|
|
44
|
+
`mode="text"` and `mode="hybrid"`, LlamaIndex's `TEXT_SEARCH` and `HYBRID`.
|
|
45
|
+
`quant="int8"` or `"bit"` keeps the graph over codes.
|
|
46
|
+
|
|
47
|
+
```python
|
|
48
|
+
store = FenecVectorStore(embeddings, "docs", url=..., token=..., full_text=True)
|
|
49
|
+
store.similarity_search("how do I compact", k=4, mode="hybrid")
|
|
50
|
+
```
|
|
51
|
+
|
|
52
|
+
`./run-tests.sh` runs LangChain's standard vector store suite and the tests
|
|
53
|
+
LlamaIndex's own integrations run against a fenec-pg it builds and starts.
|
|
54
|
+
Full reference: https://fenecdb.com/docs/integrations
|
|
@@ -0,0 +1,82 @@
|
|
|
1
|
+
"""fenecdb over HTTP.
|
|
2
|
+
|
|
3
|
+
A client for fenec-pg's HTTP endpoint (`fenec-pg --http <address>`), and on
|
|
4
|
+
top of it vector stores for LangChain (`fenecdb.langchain`) and LlamaIndex
|
|
5
|
+
(`fenecdb.llama_index`).
|
|
6
|
+
|
|
7
|
+
from fenecdb import Client
|
|
8
|
+
db = Client("http://127.0.0.1:8080", token="...")
|
|
9
|
+
db.query("get articles near embed $1 limit 5", [[0.1, 0.2, 0.3]])
|
|
10
|
+
|
|
11
|
+
The client is the standard library alone, as the server is: a statement is
|
|
12
|
+
one `POST /query`, several are one `POST /batch` under one lock.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
import json
|
|
18
|
+
import urllib.error
|
|
19
|
+
import urllib.request
|
|
20
|
+
from typing import Any, Iterable, Sequence
|
|
21
|
+
|
|
22
|
+
__all__ = ["Client", "FenecError", "placeholders"]
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
class FenecError(Exception):
|
|
26
|
+
"""A statement the server refused: its message, and the HTTP status."""
|
|
27
|
+
|
|
28
|
+
def __init__(self, message: str, status: int):
|
|
29
|
+
super().__init__(message)
|
|
30
|
+
self.status = status
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
class Client:
|
|
34
|
+
"""fenec-pg's HTTP endpoint, one statement a request."""
|
|
35
|
+
|
|
36
|
+
def __init__(
|
|
37
|
+
self,
|
|
38
|
+
url: str = "http://127.0.0.1:8080",
|
|
39
|
+
token: str | None = None,
|
|
40
|
+
timeout: float = 30.0,
|
|
41
|
+
):
|
|
42
|
+
self.url = url.rstrip("/")
|
|
43
|
+
self.token = token
|
|
44
|
+
self.timeout = timeout
|
|
45
|
+
|
|
46
|
+
def query(self, fenecql: str, params: Sequence[Any] | None = None) -> Any:
|
|
47
|
+
"""Runs one FenecQL statement. Values go in as `$1`, `$2`, ... and
|
|
48
|
+
never into the text."""
|
|
49
|
+
body = {"query": fenecql, "params": list(params or [])}
|
|
50
|
+
return self._post("/query", json.dumps(body).encode(), "application/json")
|
|
51
|
+
|
|
52
|
+
def batch(self, statements: Iterable[tuple[str, Sequence[Any]]]) -> Any:
|
|
53
|
+
"""Runs statements in order under one write lock, as one block:
|
|
54
|
+
their writes -- a create, a drop or a `create index` among them --
|
|
55
|
+
all land or, at the first error, none of them do. A batch holding a
|
|
56
|
+
`compact` runs each statement on its own, and what ran before an
|
|
57
|
+
error stays."""
|
|
58
|
+
lines = [json.dumps({"query": q, "params": list(p)}) for q, p in statements]
|
|
59
|
+
return self._post("/batch", "\n".join(lines).encode(), "application/x-ndjson")
|
|
60
|
+
|
|
61
|
+
def _post(self, path: str, body: bytes, content_type: str) -> Any:
|
|
62
|
+
req = urllib.request.Request(self.url + path, data=body, method="POST")
|
|
63
|
+
req.add_header("Content-Type", content_type)
|
|
64
|
+
if self.token:
|
|
65
|
+
req.add_header("Authorization", f"Bearer {self.token}")
|
|
66
|
+
try:
|
|
67
|
+
with urllib.request.urlopen(req, timeout=self.timeout) as resp:
|
|
68
|
+
return json.load(resp)
|
|
69
|
+
except urllib.error.HTTPError as e:
|
|
70
|
+
raw = e.read()
|
|
71
|
+
try:
|
|
72
|
+
message = json.loads(raw).get("error", raw.decode(errors="replace"))
|
|
73
|
+
except (ValueError, AttributeError):
|
|
74
|
+
message = raw.decode(errors="replace")
|
|
75
|
+
raise FenecError(message, e.code) from None
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def placeholders(start: int, n: int) -> str:
|
|
79
|
+
"""`$start, ..., $(start+n-1)`: a list literal's worth of parameters.
|
|
80
|
+
FenecQL binds a parameter to a value, not to a list, so `in` takes one
|
|
81
|
+
per element -- `in [$2, $3, $4]`."""
|
|
82
|
+
return ", ".join(f"${i}" for i in range(start, start + n))
|
|
@@ -0,0 +1,306 @@
|
|
|
1
|
+
"""A LangChain vector store over fenecdb's HTTP endpoint.
|
|
2
|
+
|
|
3
|
+
from fenecdb.langchain import FenecVectorStore
|
|
4
|
+
|
|
5
|
+
store = FenecVectorStore(embeddings, "docs", url="http://127.0.0.1:8080", token="...")
|
|
6
|
+
store.add_documents(documents)
|
|
7
|
+
store.similarity_search("how do I compact", k=4)
|
|
8
|
+
|
|
9
|
+
A document is a row: its id, its text, its metadata as JSON, and its vector
|
|
10
|
+
under `@hnsw` -- over int8 or bit codes with `quant="int8"` or `"bit"`. The
|
|
11
|
+
collection is created on the first write, when the embedding's size is known.
|
|
12
|
+
A metadata field named in `metadata_fields` also gets a column of its own,
|
|
13
|
+
under `@hash`, and `filter` takes equality over those -- `filter={"source":
|
|
14
|
+
"handbook"}` is `where source = $2`, answered by the index, before the search
|
|
15
|
+
ranks what passed.
|
|
16
|
+
|
|
17
|
+
With `full_text=True` the text is indexed for BM25 as well (`@text`), and a
|
|
18
|
+
search takes `mode`: `"text"` ranks by the words alone (`match`), with no
|
|
19
|
+
embedding of the query, and `"hybrid"` ranks the words and the vector each to
|
|
20
|
+
their own depth and fuses the two rankings (`match ... near ... fuse`). A
|
|
21
|
+
retriever passes it through `search_kwargs={"mode": "hybrid"}`. The scores
|
|
22
|
+
are then BM25's, or the fusion's, not similarities.
|
|
23
|
+
|
|
24
|
+
Writing an id that is already there replaces it: the old row out and the new
|
|
25
|
+
one in, sent as one batch and run under one write lock.
|
|
26
|
+
"""
|
|
27
|
+
|
|
28
|
+
from __future__ import annotations
|
|
29
|
+
|
|
30
|
+
import json
|
|
31
|
+
import re
|
|
32
|
+
import uuid
|
|
33
|
+
from typing import Any, Callable, Iterable, Sequence
|
|
34
|
+
|
|
35
|
+
from langchain_core.documents import Document
|
|
36
|
+
from langchain_core.embeddings import Embeddings
|
|
37
|
+
from langchain_core.vectorstores import VectorStore
|
|
38
|
+
|
|
39
|
+
from fenecdb import Client, FenecError, placeholders
|
|
40
|
+
|
|
41
|
+
__all__ = ["FenecVectorStore"]
|
|
42
|
+
|
|
43
|
+
# Names go into the statement's text, so they are held to FenecQL's.
|
|
44
|
+
_NAME = re.compile(r"^[A-Za-z_][A-Za-z0-9_]*$")
|
|
45
|
+
_TYPES = {"text", "int", "float", "bool", "timestamp"}
|
|
46
|
+
_RESERVED = {"id", "lc_id", "content", "metadata", "embedding"}
|
|
47
|
+
_METRICS = {"cosine", "l2", "dot"}
|
|
48
|
+
_QUANT = {None, "int8", "bit"}
|
|
49
|
+
_MODES = {"similarity", "text", "hybrid"}
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
class FenecVectorStore(VectorStore):
|
|
53
|
+
"""A collection in fenecdb, holding LangChain documents."""
|
|
54
|
+
|
|
55
|
+
def __init__(
|
|
56
|
+
self,
|
|
57
|
+
embedding: Embeddings,
|
|
58
|
+
collection: str = "langchain",
|
|
59
|
+
*,
|
|
60
|
+
url: str = "http://127.0.0.1:8080",
|
|
61
|
+
token: str | None = None,
|
|
62
|
+
client: Client | None = None,
|
|
63
|
+
metric: str = "cosine",
|
|
64
|
+
metadata_fields: dict[str, str] | None = None,
|
|
65
|
+
quant: str | None = None,
|
|
66
|
+
full_text: bool = False,
|
|
67
|
+
):
|
|
68
|
+
if not _NAME.match(collection):
|
|
69
|
+
raise ValueError(f"not a collection name: {collection!r}")
|
|
70
|
+
if metric not in _METRICS:
|
|
71
|
+
raise ValueError(f"metric is one of {sorted(_METRICS)}, not {metric!r}")
|
|
72
|
+
if quant not in _QUANT:
|
|
73
|
+
raise ValueError(f"quant is None, 'int8' or 'bit', not {quant!r}")
|
|
74
|
+
if quant == "bit" and metric != "cosine":
|
|
75
|
+
raise ValueError("quant='bit' keeps signs, which only cosine can order by")
|
|
76
|
+
fields = dict(metadata_fields or {})
|
|
77
|
+
for name, ty in fields.items():
|
|
78
|
+
if not _NAME.match(name) or name in _RESERVED:
|
|
79
|
+
raise ValueError(f"not a metadata field this store can index: {name!r}")
|
|
80
|
+
if ty not in _TYPES:
|
|
81
|
+
raise ValueError(f"{name}: a field is one of {sorted(_TYPES)}, not {ty!r}")
|
|
82
|
+
self._embedding = embedding
|
|
83
|
+
self.collection = collection
|
|
84
|
+
self._client = client or Client(url, token)
|
|
85
|
+
self._metric = metric
|
|
86
|
+
self._fields = fields
|
|
87
|
+
self._quant = quant
|
|
88
|
+
self._full_text = full_text
|
|
89
|
+
self._dimension: int | None = None
|
|
90
|
+
|
|
91
|
+
@property
|
|
92
|
+
def embeddings(self) -> Embeddings:
|
|
93
|
+
return self._embedding
|
|
94
|
+
|
|
95
|
+
# ------------------------------------------------------------ writing
|
|
96
|
+
|
|
97
|
+
def add_texts(
|
|
98
|
+
self,
|
|
99
|
+
texts: Iterable[str],
|
|
100
|
+
metadatas: list[dict] | None = None,
|
|
101
|
+
*,
|
|
102
|
+
ids: list[str | None] | None = None,
|
|
103
|
+
**kwargs: Any,
|
|
104
|
+
) -> list[str]:
|
|
105
|
+
texts = list(texts)
|
|
106
|
+
if not texts:
|
|
107
|
+
return []
|
|
108
|
+
metadatas = list(metadatas) if metadatas else [{} for _ in texts]
|
|
109
|
+
# A document that came without an id gets one, and so does each of
|
|
110
|
+
# the documents without one among some that have theirs.
|
|
111
|
+
ids = [i or str(uuid.uuid4()) for i in (ids or [None] * len(texts))]
|
|
112
|
+
# numpy's floats are not JSON's.
|
|
113
|
+
vectors = [[float(x) for x in v] for v in self._embedding.embed_documents(texts)]
|
|
114
|
+
self._ensure(len(vectors[0]))
|
|
115
|
+
|
|
116
|
+
docs, params = [], []
|
|
117
|
+
for doc_id, text, meta, vector in zip(ids, texts, metadatas, vectors):
|
|
118
|
+
values = [doc_id, text, json.dumps(meta, default=str), vector]
|
|
119
|
+
values += [meta.get(f) for f in self._fields]
|
|
120
|
+
names = ["lc_id", "content", "metadata", "embedding", *self._fields]
|
|
121
|
+
first = len(params) + 1
|
|
122
|
+
docs.append(
|
|
123
|
+
"{" + ", ".join(f"{n}: ${first + i}" for i, n in enumerate(names)) + "}"
|
|
124
|
+
)
|
|
125
|
+
params += values
|
|
126
|
+
self._client.batch(
|
|
127
|
+
[
|
|
128
|
+
(f"del {self.collection} where lc_id in [{placeholders(1, len(ids))}]", ids),
|
|
129
|
+
(f"put {self.collection} [{', '.join(docs)}]", params),
|
|
130
|
+
]
|
|
131
|
+
)
|
|
132
|
+
return ids
|
|
133
|
+
|
|
134
|
+
def delete(self, ids: list[str] | None = None, **kwargs: Any) -> bool | None:
|
|
135
|
+
"""Deletes these ids, or every document when `ids` is None."""
|
|
136
|
+
if ids is None:
|
|
137
|
+
statement, params = f"del {self.collection}", []
|
|
138
|
+
else:
|
|
139
|
+
params = list(ids)
|
|
140
|
+
if not params:
|
|
141
|
+
return True
|
|
142
|
+
statement = (
|
|
143
|
+
f"del {self.collection} where lc_id in [{placeholders(1, len(params))}]"
|
|
144
|
+
)
|
|
145
|
+
self._run(statement, params)
|
|
146
|
+
return True
|
|
147
|
+
|
|
148
|
+
def drop_collection(self) -> None:
|
|
149
|
+
"""Drops the collection, documents and index together."""
|
|
150
|
+
self._client.query(f"drop collection if exists {self.collection}")
|
|
151
|
+
self._dimension = None
|
|
152
|
+
|
|
153
|
+
# ------------------------------------------------------------ reading
|
|
154
|
+
|
|
155
|
+
def get_by_ids(self, ids: Sequence[str], /) -> list[Document]:
|
|
156
|
+
ids = list(ids)
|
|
157
|
+
if not ids:
|
|
158
|
+
return []
|
|
159
|
+
rows = self._run(
|
|
160
|
+
f"get {self.collection} select lc_id, content, metadata "
|
|
161
|
+
f"where lc_id in [{placeholders(1, len(ids))}]",
|
|
162
|
+
ids,
|
|
163
|
+
)
|
|
164
|
+
return [self._document(r) for r in rows]
|
|
165
|
+
|
|
166
|
+
def similarity_search(
|
|
167
|
+
self, query: str, k: int = 4, filter: dict | None = None, **kwargs: Any
|
|
168
|
+
) -> list[Document]:
|
|
169
|
+
"""`mode` as `similarity_search_with_score` takes it."""
|
|
170
|
+
return [d for d, _ in self.similarity_search_with_score(query, k, filter, **kwargs)]
|
|
171
|
+
|
|
172
|
+
def similarity_search_with_score(
|
|
173
|
+
self,
|
|
174
|
+
query: str,
|
|
175
|
+
k: int = 4,
|
|
176
|
+
filter: dict | None = None,
|
|
177
|
+
*,
|
|
178
|
+
mode: str = "similarity",
|
|
179
|
+
**kwargs: Any,
|
|
180
|
+
) -> list[tuple[Document, float]]:
|
|
181
|
+
"""Documents with fenecdb's score: the cosine similarity under
|
|
182
|
+
`cosine`, higher is nearer; the distance under `l2`; the product
|
|
183
|
+
under `dot`. With `mode="text"` or `"hybrid"` (`full_text=True`)
|
|
184
|
+
BM25's score, or the fusion's."""
|
|
185
|
+
if mode not in _MODES:
|
|
186
|
+
raise ValueError(f"mode is one of {sorted(_MODES)}, not {mode!r}")
|
|
187
|
+
if mode == "similarity":
|
|
188
|
+
vector = self._embedding.embed_query(query)
|
|
189
|
+
return self.similarity_search_with_score_by_vector(vector, k, filter, **kwargs)
|
|
190
|
+
if not self._full_text:
|
|
191
|
+
raise ValueError(f"mode {mode!r} needs the store made with full_text=True")
|
|
192
|
+
if mode == "text":
|
|
193
|
+
rank, ahead = "match content $1", [query]
|
|
194
|
+
else:
|
|
195
|
+
vector = [float(x) for x in self._embedding.embed_query(query)]
|
|
196
|
+
rank, ahead = "match content $1 near embedding $2 fuse", [query, vector]
|
|
197
|
+
where, params = self._where(filter, first=len(ahead) + 1)
|
|
198
|
+
rows = self._run(
|
|
199
|
+
f"get {self.collection} select lc_id, content, metadata{where} "
|
|
200
|
+
f"{rank} limit {int(k)}",
|
|
201
|
+
[*ahead, *params],
|
|
202
|
+
)
|
|
203
|
+
return [(self._document(r), r["_score"]) for r in rows]
|
|
204
|
+
|
|
205
|
+
def similarity_search_by_vector(
|
|
206
|
+
self, embedding: list[float], k: int = 4, filter: dict | None = None, **kwargs: Any
|
|
207
|
+
) -> list[Document]:
|
|
208
|
+
pairs = self.similarity_search_with_score_by_vector(embedding, k, filter, **kwargs)
|
|
209
|
+
return [d for d, _ in pairs]
|
|
210
|
+
|
|
211
|
+
def similarity_search_with_score_by_vector(
|
|
212
|
+
self, embedding: list[float], k: int = 4, filter: dict | None = None, **kwargs: Any
|
|
213
|
+
) -> list[tuple[Document, float]]:
|
|
214
|
+
where, params = self._where(filter, first=2)
|
|
215
|
+
rows = self._run(
|
|
216
|
+
f"get {self.collection} select lc_id, content, metadata{where} "
|
|
217
|
+
f"near embedding $1 limit {int(k)}",
|
|
218
|
+
[[float(x) for x in embedding], *params],
|
|
219
|
+
)
|
|
220
|
+
return [(self._document(r), r["_score"]) for r in rows]
|
|
221
|
+
|
|
222
|
+
def _select_relevance_score_fn(self) -> Callable[[float], float]:
|
|
223
|
+
# Under `cosine` the score already is 1 - distance, LangChain's
|
|
224
|
+
# relevance; under `dot` the product is, for normalised vectors.
|
|
225
|
+
if self._metric == "l2":
|
|
226
|
+
return self._euclidean_relevance_score_fn
|
|
227
|
+
return lambda score: score
|
|
228
|
+
|
|
229
|
+
@classmethod
|
|
230
|
+
def from_texts(
|
|
231
|
+
cls,
|
|
232
|
+
texts: list[str],
|
|
233
|
+
embedding: Embeddings,
|
|
234
|
+
metadatas: list[dict] | None = None,
|
|
235
|
+
*,
|
|
236
|
+
ids: list[str] | None = None,
|
|
237
|
+
**kwargs: Any,
|
|
238
|
+
) -> "FenecVectorStore":
|
|
239
|
+
store = cls(embedding, **kwargs)
|
|
240
|
+
store.add_texts(texts, metadatas, ids=ids)
|
|
241
|
+
return store
|
|
242
|
+
|
|
243
|
+
# ------------------------------------------------------------ inside
|
|
244
|
+
|
|
245
|
+
def _ensure(self, dimension: int) -> None:
|
|
246
|
+
if self._dimension == dimension:
|
|
247
|
+
return
|
|
248
|
+
extra = "".join(f", {f} {t} @hash" for f, t in self._fields.items())
|
|
249
|
+
quant = f", quant={self._quant}" if self._quant else ""
|
|
250
|
+
content = "content text @text" if self._full_text else "content text"
|
|
251
|
+
self._client.query(
|
|
252
|
+
f"create collection if not exists {self.collection} ("
|
|
253
|
+
f"lc_id text @hash, {content}, metadata text, "
|
|
254
|
+
f"embedding vector<{dimension}> @hnsw({self._metric}{quant}){extra})"
|
|
255
|
+
)
|
|
256
|
+
if self._full_text:
|
|
257
|
+
# A collection made before without it takes the index now.
|
|
258
|
+
self._client.query(
|
|
259
|
+
f"create index if not exists on {self.collection} (content) @text"
|
|
260
|
+
)
|
|
261
|
+
self._dimension = dimension
|
|
262
|
+
|
|
263
|
+
def _run(self, statement: str, params: list) -> Any:
|
|
264
|
+
"""A statement over a collection that may not be there yet: a store
|
|
265
|
+
nothing was written to is an empty one, not an error."""
|
|
266
|
+
try:
|
|
267
|
+
return self._client.query(statement, params)
|
|
268
|
+
except FenecError as e:
|
|
269
|
+
if e.status == 404 and str(e).startswith("collection `"):
|
|
270
|
+
return []
|
|
271
|
+
raise
|
|
272
|
+
|
|
273
|
+
def _where(self, filter: dict | None, first: int) -> tuple[str, list]:
|
|
274
|
+
"""` where f = $n and g in [$m, ...]` over the indexed fields."""
|
|
275
|
+
if not filter:
|
|
276
|
+
return "", []
|
|
277
|
+
terms, params = [], []
|
|
278
|
+
for name, want in filter.items():
|
|
279
|
+
if name not in self._fields:
|
|
280
|
+
raise ValueError(
|
|
281
|
+
f"`{name}` is not in metadata_fields: only those can be filtered on"
|
|
282
|
+
)
|
|
283
|
+
if isinstance(want, dict):
|
|
284
|
+
if set(want) == {"$eq"}:
|
|
285
|
+
want = want["$eq"]
|
|
286
|
+
elif set(want) == {"$in"}:
|
|
287
|
+
want = list(want["$in"])
|
|
288
|
+
else:
|
|
289
|
+
raise ValueError(f"{name}: only $eq and $in are supported, not {want}")
|
|
290
|
+
if isinstance(want, (list, tuple, set)):
|
|
291
|
+
values = list(want)
|
|
292
|
+
n = first + len(params)
|
|
293
|
+
terms.append(f"{name} in [{placeholders(n, len(values))}]")
|
|
294
|
+
params += values
|
|
295
|
+
else:
|
|
296
|
+
terms.append(f"{name} = ${first + len(params)}")
|
|
297
|
+
params.append(want)
|
|
298
|
+
return " where " + " and ".join(terms), params
|
|
299
|
+
|
|
300
|
+
@staticmethod
|
|
301
|
+
def _document(row: dict) -> Document:
|
|
302
|
+
return Document(
|
|
303
|
+
id=row["lc_id"],
|
|
304
|
+
page_content=row["content"],
|
|
305
|
+
metadata=json.loads(row["metadata"] or "{}"),
|
|
306
|
+
)
|