fenecdb 0.1.5__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
fenecdb-0.1.5/PKG-INFO ADDED
@@ -0,0 +1,76 @@
1
+ Metadata-Version: 2.4
2
+ Name: fenecdb
3
+ Version: 0.1.5
4
+ Summary: fenecdb over HTTP: a client, and vector stores for LangChain and LlamaIndex
5
+ License-Expression: Apache-2.0
6
+ Project-URL: Homepage, https://fenecdb.com
7
+ Project-URL: Documentation, https://fenecdb.com/docs/integrations
8
+ Project-URL: Source, https://github.com/fenecdb/fenec
9
+ Keywords: fenecdb,vector,database,langchain,llama-index
10
+ Requires-Python: >=3.10
11
+ Description-Content-Type: text/markdown
12
+ Provides-Extra: langchain
13
+ Requires-Dist: langchain-core>=1.0; extra == "langchain"
14
+ Provides-Extra: llama-index
15
+ Requires-Dist: llama-index-core>=0.12; extra == "llama-index"
16
+ Provides-Extra: test
17
+ Requires-Dist: pytest>=8; extra == "test"
18
+ Requires-Dist: pytest-asyncio>=0.23; extra == "test"
19
+ Requires-Dist: langchain-core>=1.0; extra == "test"
20
+ Requires-Dist: langchain-tests>=1.0; extra == "test"
21
+ Requires-Dist: llama-index-core>=0.12; extra == "test"
22
+
23
+ # fenecdb for Python
24
+
25
+ A client for fenec-pg's HTTP endpoint (`fenec-pg --http <address>`), and
26
+ vector stores for LangChain and LlamaIndex on top of it.
27
+
28
+ ```sh
29
+ pip install "fenecdb[langchain] @ git+https://github.com/fenecdb/fenec#subdirectory=integrations/python"
30
+ # or fenecdb[llama-index]
31
+ ```
32
+
33
+ ```python
34
+ from fenecdb import Client
35
+
36
+ db = Client("http://127.0.0.1:8080", token="...")
37
+ db.query("get articles select title near embed $1 limit 5", [[0.1, 0.2, 0.3]])
38
+ ```
39
+
40
+ ```python
41
+ from fenecdb.langchain import FenecVectorStore
42
+
43
+ store = FenecVectorStore(embeddings, "docs", url="http://127.0.0.1:8080", token="...",
44
+ metadata_fields={"source": "text"})
45
+ store.add_documents(documents)
46
+ store.similarity_search("how do I compact", k=4, filter={"source": "handbook"})
47
+ ```
48
+
49
+ ```python
50
+ from fenecdb.llama_index import FenecVectorStore
51
+ from llama_index.core import StorageContext, VectorStoreIndex
52
+
53
+ store = FenecVectorStore("docs", url="http://127.0.0.1:8080", token="...")
54
+ index = VectorStoreIndex.from_documents(
55
+ documents, storage_context=StorageContext.from_defaults(vector_store=store)
56
+ )
57
+ ```
58
+
59
+ A document or node is a row -- its id, text, metadata as JSON and vector
60
+ under `@hnsw` -- in a collection created on the first write. Fields named in
61
+ `metadata_fields` get columns of their own under `@hash`, and filters over
62
+ them are answered by the index before the search runs.
63
+
64
+ With `full_text=True` the text is indexed for BM25 too, and both stores search
65
+ by the words alone or by the words and the vector fused -- LangChain's
66
+ `mode="text"` and `mode="hybrid"`, LlamaIndex's `TEXT_SEARCH` and `HYBRID`.
67
+ `quant="int8"` or `"bit"` keeps the graph over codes.
68
+
69
+ ```python
70
+ store = FenecVectorStore(embeddings, "docs", url=..., token=..., full_text=True)
71
+ store.similarity_search("how do I compact", k=4, mode="hybrid")
72
+ ```
73
+
74
+ `./run-tests.sh` runs LangChain's standard vector store suite and the tests
75
+ LlamaIndex's own integrations run against a fenec-pg it builds and starts.
76
+ Full reference: https://fenecdb.com/docs/integrations
@@ -0,0 +1,54 @@
1
+ # fenecdb for Python
2
+
3
+ A client for fenec-pg's HTTP endpoint (`fenec-pg --http <address>`), and
4
+ vector stores for LangChain and LlamaIndex on top of it.
5
+
6
+ ```sh
7
+ pip install "fenecdb[langchain] @ git+https://github.com/fenecdb/fenec#subdirectory=integrations/python"
8
+ # or fenecdb[llama-index]
9
+ ```
10
+
11
+ ```python
12
+ from fenecdb import Client
13
+
14
+ db = Client("http://127.0.0.1:8080", token="...")
15
+ db.query("get articles select title near embed $1 limit 5", [[0.1, 0.2, 0.3]])
16
+ ```
17
+
18
+ ```python
19
+ from fenecdb.langchain import FenecVectorStore
20
+
21
+ store = FenecVectorStore(embeddings, "docs", url="http://127.0.0.1:8080", token="...",
22
+ metadata_fields={"source": "text"})
23
+ store.add_documents(documents)
24
+ store.similarity_search("how do I compact", k=4, filter={"source": "handbook"})
25
+ ```
26
+
27
+ ```python
28
+ from fenecdb.llama_index import FenecVectorStore
29
+ from llama_index.core import StorageContext, VectorStoreIndex
30
+
31
+ store = FenecVectorStore("docs", url="http://127.0.0.1:8080", token="...")
32
+ index = VectorStoreIndex.from_documents(
33
+ documents, storage_context=StorageContext.from_defaults(vector_store=store)
34
+ )
35
+ ```
36
+
37
+ A document or node is a row -- its id, text, metadata as JSON and vector
38
+ under `@hnsw` -- in a collection created on the first write. Fields named in
39
+ `metadata_fields` get columns of their own under `@hash`, and filters over
40
+ them are answered by the index before the search runs.
41
+
42
+ With `full_text=True` the text is indexed for BM25 too, and both stores search
43
+ by the words alone or by the words and the vector fused -- LangChain's
44
+ `mode="text"` and `mode="hybrid"`, LlamaIndex's `TEXT_SEARCH` and `HYBRID`.
45
+ `quant="int8"` or `"bit"` keeps the graph over codes.
46
+
47
+ ```python
48
+ store = FenecVectorStore(embeddings, "docs", url=..., token=..., full_text=True)
49
+ store.similarity_search("how do I compact", k=4, mode="hybrid")
50
+ ```
51
+
52
+ `./run-tests.sh` runs LangChain's standard vector store suite and the tests
53
+ LlamaIndex's own integrations run against a fenec-pg it builds and starts.
54
+ Full reference: https://fenecdb.com/docs/integrations
@@ -0,0 +1,82 @@
1
+ """fenecdb over HTTP.
2
+
3
+ A client for fenec-pg's HTTP endpoint (`fenec-pg --http <address>`), and on
4
+ top of it vector stores for LangChain (`fenecdb.langchain`) and LlamaIndex
5
+ (`fenecdb.llama_index`).
6
+
7
+ from fenecdb import Client
8
+ db = Client("http://127.0.0.1:8080", token="...")
9
+ db.query("get articles near embed $1 limit 5", [[0.1, 0.2, 0.3]])
10
+
11
+ The client is the standard library alone, as the server is: a statement is
12
+ one `POST /query`, several are one `POST /batch` under one lock.
13
+ """
14
+
15
+ from __future__ import annotations
16
+
17
+ import json
18
+ import urllib.error
19
+ import urllib.request
20
+ from typing import Any, Iterable, Sequence
21
+
22
+ __all__ = ["Client", "FenecError", "placeholders"]
23
+
24
+
25
+ class FenecError(Exception):
26
+ """A statement the server refused: its message, and the HTTP status."""
27
+
28
+ def __init__(self, message: str, status: int):
29
+ super().__init__(message)
30
+ self.status = status
31
+
32
+
33
+ class Client:
34
+ """fenec-pg's HTTP endpoint, one statement a request."""
35
+
36
+ def __init__(
37
+ self,
38
+ url: str = "http://127.0.0.1:8080",
39
+ token: str | None = None,
40
+ timeout: float = 30.0,
41
+ ):
42
+ self.url = url.rstrip("/")
43
+ self.token = token
44
+ self.timeout = timeout
45
+
46
+ def query(self, fenecql: str, params: Sequence[Any] | None = None) -> Any:
47
+ """Runs one FenecQL statement. Values go in as `$1`, `$2`, ... and
48
+ never into the text."""
49
+ body = {"query": fenecql, "params": list(params or [])}
50
+ return self._post("/query", json.dumps(body).encode(), "application/json")
51
+
52
+ def batch(self, statements: Iterable[tuple[str, Sequence[Any]]]) -> Any:
53
+ """Runs statements in order under one write lock, as one block:
54
+ their writes -- a create, a drop or a `create index` among them --
55
+ all land or, at the first error, none of them do. A batch holding a
56
+ `compact` runs each statement on its own, and what ran before an
57
+ error stays."""
58
+ lines = [json.dumps({"query": q, "params": list(p)}) for q, p in statements]
59
+ return self._post("/batch", "\n".join(lines).encode(), "application/x-ndjson")
60
+
61
+ def _post(self, path: str, body: bytes, content_type: str) -> Any:
62
+ req = urllib.request.Request(self.url + path, data=body, method="POST")
63
+ req.add_header("Content-Type", content_type)
64
+ if self.token:
65
+ req.add_header("Authorization", f"Bearer {self.token}")
66
+ try:
67
+ with urllib.request.urlopen(req, timeout=self.timeout) as resp:
68
+ return json.load(resp)
69
+ except urllib.error.HTTPError as e:
70
+ raw = e.read()
71
+ try:
72
+ message = json.loads(raw).get("error", raw.decode(errors="replace"))
73
+ except (ValueError, AttributeError):
74
+ message = raw.decode(errors="replace")
75
+ raise FenecError(message, e.code) from None
76
+
77
+
78
+ def placeholders(start: int, n: int) -> str:
79
+ """`$start, ..., $(start+n-1)`: a list literal's worth of parameters.
80
+ FenecQL binds a parameter to a value, not to a list, so `in` takes one
81
+ per element -- `in [$2, $3, $4]`."""
82
+ return ", ".join(f"${i}" for i in range(start, start + n))
@@ -0,0 +1,306 @@
1
+ """A LangChain vector store over fenecdb's HTTP endpoint.
2
+
3
+ from fenecdb.langchain import FenecVectorStore
4
+
5
+ store = FenecVectorStore(embeddings, "docs", url="http://127.0.0.1:8080", token="...")
6
+ store.add_documents(documents)
7
+ store.similarity_search("how do I compact", k=4)
8
+
9
+ A document is a row: its id, its text, its metadata as JSON, and its vector
10
+ under `@hnsw` -- over int8 or bit codes with `quant="int8"` or `"bit"`. The
11
+ collection is created on the first write, when the embedding's size is known.
12
+ A metadata field named in `metadata_fields` also gets a column of its own,
13
+ under `@hash`, and `filter` takes equality over those -- `filter={"source":
14
+ "handbook"}` is `where source = $2`, answered by the index, before the search
15
+ ranks what passed.
16
+
17
+ With `full_text=True` the text is indexed for BM25 as well (`@text`), and a
18
+ search takes `mode`: `"text"` ranks by the words alone (`match`), with no
19
+ embedding of the query, and `"hybrid"` ranks the words and the vector each to
20
+ their own depth and fuses the two rankings (`match ... near ... fuse`). A
21
+ retriever passes it through `search_kwargs={"mode": "hybrid"}`. The scores
22
+ are then BM25's, or the fusion's, not similarities.
23
+
24
+ Writing an id that is already there replaces it: the old row out and the new
25
+ one in, sent as one batch and run under one write lock.
26
+ """
27
+
28
+ from __future__ import annotations
29
+
30
+ import json
31
+ import re
32
+ import uuid
33
+ from typing import Any, Callable, Iterable, Sequence
34
+
35
+ from langchain_core.documents import Document
36
+ from langchain_core.embeddings import Embeddings
37
+ from langchain_core.vectorstores import VectorStore
38
+
39
+ from fenecdb import Client, FenecError, placeholders
40
+
41
+ __all__ = ["FenecVectorStore"]
42
+
43
+ # Names go into the statement's text, so they are held to FenecQL's.
44
+ _NAME = re.compile(r"^[A-Za-z_][A-Za-z0-9_]*$")
45
+ _TYPES = {"text", "int", "float", "bool", "timestamp"}
46
+ _RESERVED = {"id", "lc_id", "content", "metadata", "embedding"}
47
+ _METRICS = {"cosine", "l2", "dot"}
48
+ _QUANT = {None, "int8", "bit"}
49
+ _MODES = {"similarity", "text", "hybrid"}
50
+
51
+
52
+ class FenecVectorStore(VectorStore):
53
+ """A collection in fenecdb, holding LangChain documents."""
54
+
55
+ def __init__(
56
+ self,
57
+ embedding: Embeddings,
58
+ collection: str = "langchain",
59
+ *,
60
+ url: str = "http://127.0.0.1:8080",
61
+ token: str | None = None,
62
+ client: Client | None = None,
63
+ metric: str = "cosine",
64
+ metadata_fields: dict[str, str] | None = None,
65
+ quant: str | None = None,
66
+ full_text: bool = False,
67
+ ):
68
+ if not _NAME.match(collection):
69
+ raise ValueError(f"not a collection name: {collection!r}")
70
+ if metric not in _METRICS:
71
+ raise ValueError(f"metric is one of {sorted(_METRICS)}, not {metric!r}")
72
+ if quant not in _QUANT:
73
+ raise ValueError(f"quant is None, 'int8' or 'bit', not {quant!r}")
74
+ if quant == "bit" and metric != "cosine":
75
+ raise ValueError("quant='bit' keeps signs, which only cosine can order by")
76
+ fields = dict(metadata_fields or {})
77
+ for name, ty in fields.items():
78
+ if not _NAME.match(name) or name in _RESERVED:
79
+ raise ValueError(f"not a metadata field this store can index: {name!r}")
80
+ if ty not in _TYPES:
81
+ raise ValueError(f"{name}: a field is one of {sorted(_TYPES)}, not {ty!r}")
82
+ self._embedding = embedding
83
+ self.collection = collection
84
+ self._client = client or Client(url, token)
85
+ self._metric = metric
86
+ self._fields = fields
87
+ self._quant = quant
88
+ self._full_text = full_text
89
+ self._dimension: int | None = None
90
+
91
+ @property
92
+ def embeddings(self) -> Embeddings:
93
+ return self._embedding
94
+
95
+ # ------------------------------------------------------------ writing
96
+
97
+ def add_texts(
98
+ self,
99
+ texts: Iterable[str],
100
+ metadatas: list[dict] | None = None,
101
+ *,
102
+ ids: list[str | None] | None = None,
103
+ **kwargs: Any,
104
+ ) -> list[str]:
105
+ texts = list(texts)
106
+ if not texts:
107
+ return []
108
+ metadatas = list(metadatas) if metadatas else [{} for _ in texts]
109
+ # A document that came without an id gets one, and so does each of
110
+ # the documents without one among some that have theirs.
111
+ ids = [i or str(uuid.uuid4()) for i in (ids or [None] * len(texts))]
112
+ # numpy's floats are not JSON's.
113
+ vectors = [[float(x) for x in v] for v in self._embedding.embed_documents(texts)]
114
+ self._ensure(len(vectors[0]))
115
+
116
+ docs, params = [], []
117
+ for doc_id, text, meta, vector in zip(ids, texts, metadatas, vectors):
118
+ values = [doc_id, text, json.dumps(meta, default=str), vector]
119
+ values += [meta.get(f) for f in self._fields]
120
+ names = ["lc_id", "content", "metadata", "embedding", *self._fields]
121
+ first = len(params) + 1
122
+ docs.append(
123
+ "{" + ", ".join(f"{n}: ${first + i}" for i, n in enumerate(names)) + "}"
124
+ )
125
+ params += values
126
+ self._client.batch(
127
+ [
128
+ (f"del {self.collection} where lc_id in [{placeholders(1, len(ids))}]", ids),
129
+ (f"put {self.collection} [{', '.join(docs)}]", params),
130
+ ]
131
+ )
132
+ return ids
133
+
134
+ def delete(self, ids: list[str] | None = None, **kwargs: Any) -> bool | None:
135
+ """Deletes these ids, or every document when `ids` is None."""
136
+ if ids is None:
137
+ statement, params = f"del {self.collection}", []
138
+ else:
139
+ params = list(ids)
140
+ if not params:
141
+ return True
142
+ statement = (
143
+ f"del {self.collection} where lc_id in [{placeholders(1, len(params))}]"
144
+ )
145
+ self._run(statement, params)
146
+ return True
147
+
148
+ def drop_collection(self) -> None:
149
+ """Drops the collection, documents and index together."""
150
+ self._client.query(f"drop collection if exists {self.collection}")
151
+ self._dimension = None
152
+
153
+ # ------------------------------------------------------------ reading
154
+
155
+ def get_by_ids(self, ids: Sequence[str], /) -> list[Document]:
156
+ ids = list(ids)
157
+ if not ids:
158
+ return []
159
+ rows = self._run(
160
+ f"get {self.collection} select lc_id, content, metadata "
161
+ f"where lc_id in [{placeholders(1, len(ids))}]",
162
+ ids,
163
+ )
164
+ return [self._document(r) for r in rows]
165
+
166
+ def similarity_search(
167
+ self, query: str, k: int = 4, filter: dict | None = None, **kwargs: Any
168
+ ) -> list[Document]:
169
+ """`mode` as `similarity_search_with_score` takes it."""
170
+ return [d for d, _ in self.similarity_search_with_score(query, k, filter, **kwargs)]
171
+
172
+ def similarity_search_with_score(
173
+ self,
174
+ query: str,
175
+ k: int = 4,
176
+ filter: dict | None = None,
177
+ *,
178
+ mode: str = "similarity",
179
+ **kwargs: Any,
180
+ ) -> list[tuple[Document, float]]:
181
+ """Documents with fenecdb's score: the cosine similarity under
182
+ `cosine`, higher is nearer; the distance under `l2`; the product
183
+ under `dot`. With `mode="text"` or `"hybrid"` (`full_text=True`)
184
+ BM25's score, or the fusion's."""
185
+ if mode not in _MODES:
186
+ raise ValueError(f"mode is one of {sorted(_MODES)}, not {mode!r}")
187
+ if mode == "similarity":
188
+ vector = self._embedding.embed_query(query)
189
+ return self.similarity_search_with_score_by_vector(vector, k, filter, **kwargs)
190
+ if not self._full_text:
191
+ raise ValueError(f"mode {mode!r} needs the store made with full_text=True")
192
+ if mode == "text":
193
+ rank, ahead = "match content $1", [query]
194
+ else:
195
+ vector = [float(x) for x in self._embedding.embed_query(query)]
196
+ rank, ahead = "match content $1 near embedding $2 fuse", [query, vector]
197
+ where, params = self._where(filter, first=len(ahead) + 1)
198
+ rows = self._run(
199
+ f"get {self.collection} select lc_id, content, metadata{where} "
200
+ f"{rank} limit {int(k)}",
201
+ [*ahead, *params],
202
+ )
203
+ return [(self._document(r), r["_score"]) for r in rows]
204
+
205
+ def similarity_search_by_vector(
206
+ self, embedding: list[float], k: int = 4, filter: dict | None = None, **kwargs: Any
207
+ ) -> list[Document]:
208
+ pairs = self.similarity_search_with_score_by_vector(embedding, k, filter, **kwargs)
209
+ return [d for d, _ in pairs]
210
+
211
+ def similarity_search_with_score_by_vector(
212
+ self, embedding: list[float], k: int = 4, filter: dict | None = None, **kwargs: Any
213
+ ) -> list[tuple[Document, float]]:
214
+ where, params = self._where(filter, first=2)
215
+ rows = self._run(
216
+ f"get {self.collection} select lc_id, content, metadata{where} "
217
+ f"near embedding $1 limit {int(k)}",
218
+ [[float(x) for x in embedding], *params],
219
+ )
220
+ return [(self._document(r), r["_score"]) for r in rows]
221
+
222
+ def _select_relevance_score_fn(self) -> Callable[[float], float]:
223
+ # Under `cosine` the score already is 1 - distance, LangChain's
224
+ # relevance; under `dot` the product is, for normalised vectors.
225
+ if self._metric == "l2":
226
+ return self._euclidean_relevance_score_fn
227
+ return lambda score: score
228
+
229
+ @classmethod
230
+ def from_texts(
231
+ cls,
232
+ texts: list[str],
233
+ embedding: Embeddings,
234
+ metadatas: list[dict] | None = None,
235
+ *,
236
+ ids: list[str] | None = None,
237
+ **kwargs: Any,
238
+ ) -> "FenecVectorStore":
239
+ store = cls(embedding, **kwargs)
240
+ store.add_texts(texts, metadatas, ids=ids)
241
+ return store
242
+
243
+ # ------------------------------------------------------------ inside
244
+
245
+ def _ensure(self, dimension: int) -> None:
246
+ if self._dimension == dimension:
247
+ return
248
+ extra = "".join(f", {f} {t} @hash" for f, t in self._fields.items())
249
+ quant = f", quant={self._quant}" if self._quant else ""
250
+ content = "content text @text" if self._full_text else "content text"
251
+ self._client.query(
252
+ f"create collection if not exists {self.collection} ("
253
+ f"lc_id text @hash, {content}, metadata text, "
254
+ f"embedding vector<{dimension}> @hnsw({self._metric}{quant}){extra})"
255
+ )
256
+ if self._full_text:
257
+ # A collection made before without it takes the index now.
258
+ self._client.query(
259
+ f"create index if not exists on {self.collection} (content) @text"
260
+ )
261
+ self._dimension = dimension
262
+
263
+ def _run(self, statement: str, params: list) -> Any:
264
+ """A statement over a collection that may not be there yet: a store
265
+ nothing was written to is an empty one, not an error."""
266
+ try:
267
+ return self._client.query(statement, params)
268
+ except FenecError as e:
269
+ if e.status == 404 and str(e).startswith("collection `"):
270
+ return []
271
+ raise
272
+
273
+ def _where(self, filter: dict | None, first: int) -> tuple[str, list]:
274
+ """` where f = $n and g in [$m, ...]` over the indexed fields."""
275
+ if not filter:
276
+ return "", []
277
+ terms, params = [], []
278
+ for name, want in filter.items():
279
+ if name not in self._fields:
280
+ raise ValueError(
281
+ f"`{name}` is not in metadata_fields: only those can be filtered on"
282
+ )
283
+ if isinstance(want, dict):
284
+ if set(want) == {"$eq"}:
285
+ want = want["$eq"]
286
+ elif set(want) == {"$in"}:
287
+ want = list(want["$in"])
288
+ else:
289
+ raise ValueError(f"{name}: only $eq and $in are supported, not {want}")
290
+ if isinstance(want, (list, tuple, set)):
291
+ values = list(want)
292
+ n = first + len(params)
293
+ terms.append(f"{name} in [{placeholders(n, len(values))}]")
294
+ params += values
295
+ else:
296
+ terms.append(f"{name} = ${first + len(params)}")
297
+ params.append(want)
298
+ return " where " + " and ".join(terms), params
299
+
300
+ @staticmethod
301
+ def _document(row: dict) -> Document:
302
+ return Document(
303
+ id=row["lc_id"],
304
+ page_content=row["content"],
305
+ metadata=json.loads(row["metadata"] or "{}"),
306
+ )