agent-framework-duckdb 1.0.0a261002__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_framework_duckdb-1.0.0a261002/LICENSE +21 -0
- agent_framework_duckdb-1.0.0a261002/PKG-INFO +180 -0
- agent_framework_duckdb-1.0.0a261002/README.md +154 -0
- agent_framework_duckdb-1.0.0a261002/agent_framework_duckdb/__init__.py +14 -0
- agent_framework_duckdb-1.0.0a261002/agent_framework_duckdb/_vector_store.py +788 -0
- agent_framework_duckdb-1.0.0a261002/agent_framework_duckdb/py.typed +0 -0
- agent_framework_duckdb-1.0.0a261002/pyproject.toml +61 -0
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) Microsoft Corporation.
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,180 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: agent-framework-duckdb
|
|
3
|
+
Version: 1.0.0a261002
|
|
4
|
+
Summary: DuckDB vector stores for Microsoft Agent Framework.
|
|
5
|
+
Author-email: Microsoft <af-support@microsoft.com>
|
|
6
|
+
Requires-Python: >=3.10
|
|
7
|
+
Description-Content-Type: text/markdown
|
|
8
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
9
|
+
Classifier: Development Status :: 3 - Alpha
|
|
10
|
+
Classifier: Intended Audience :: Developers
|
|
11
|
+
Classifier: Programming Language :: Python :: 3
|
|
12
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
13
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.14
|
|
17
|
+
Classifier: Typing :: Typed
|
|
18
|
+
License-File: LICENSE
|
|
19
|
+
Requires-Dist: agent-framework-core>=1.19.0,<2
|
|
20
|
+
Requires-Dist: duckdb>=1.4.1,<1.6
|
|
21
|
+
Requires-Dist: pytz>=2024.1,<2027
|
|
22
|
+
Project-URL: homepage, https://aka.ms/agent-framework
|
|
23
|
+
Project-URL: issues, https://github.com/microsoft/agent-framework/issues
|
|
24
|
+
Project-URL: source, https://github.com/microsoft/agent-framework/tree/main/python
|
|
25
|
+
|
|
26
|
+
# Agent Framework DuckDB vector store
|
|
27
|
+
|
|
28
|
+
Store and search typed records in DuckDB using the
|
|
29
|
+
[Microsoft Agent Framework](https://learn.microsoft.com/agent-framework/)
|
|
30
|
+
vector-store APIs. This alpha package uses the **DuckDB Python client only**:
|
|
31
|
+
local files, DuckDB-supported connection URIs (including MotherDuck), and
|
|
32
|
+
caller-provided `duckdb.DuckDBPyConnection` objects use the same implementation.
|
|
33
|
+
|
|
34
|
+
## Installation
|
|
35
|
+
|
|
36
|
+
```bash
|
|
37
|
+
pip install agent-framework-duckdb --pre
|
|
38
|
+
```
|
|
39
|
+
|
|
40
|
+
Python 3.10+ and DuckDB 1.4.1–1.5.x are required. The package also installs
|
|
41
|
+
`pytz`, which DuckDB uses when returning timezone-aware timestamps. Import
|
|
42
|
+
`DuckDBStore` and `DuckDBCollection` directly from `agent_framework_duckdb`;
|
|
43
|
+
this alpha package is not part of `agent-framework-core[all]`.
|
|
44
|
+
|
|
45
|
+
## Connections and persistence
|
|
46
|
+
|
|
47
|
+
`DuckDBStore()` (and a directly constructed `DuckDBCollection`) defaults to
|
|
48
|
+
**`agent-framework.duckdb` in the current working directory**. It is a normal
|
|
49
|
+
persistent database file, not an in-memory database or temporary file. Give
|
|
50
|
+
each application a suitable writable location with
|
|
51
|
+
`connection_string="path/to/vectors.duckdb"`; parent directories must already
|
|
52
|
+
exist. The connection is opened lazily on the first async operation.
|
|
53
|
+
`aclose()` or an async context manager drains pending operations and releases
|
|
54
|
+
the file. Opening a *new* store against the same filename restores its tables
|
|
55
|
+
and records.
|
|
56
|
+
|
|
57
|
+
`DUCKDB_CONNECTION_STRING` can instead provide a filename or a URI. Settings
|
|
58
|
+
precedence is **explicit `connection_string` > selected `.env` file >
|
|
59
|
+
environment > persistent default**. Select a file with `env_file_path` and
|
|
60
|
+
optionally `env_file_encoding`; `.env` files are never discovered implicitly.
|
|
61
|
+
An explicitly empty connection string is an error. The URI is held as an AF
|
|
62
|
+
`SecretString` and never printed or logged by the connector.
|
|
63
|
+
|
|
64
|
+
To use [MotherDuck](https://motherduck.com/docs/getting-started/interfaces/client-apis/python/installation-authentication/),
|
|
65
|
+
configure the *same* DuckDB client with a `md:` URI and its usual credentials,
|
|
66
|
+
for example:
|
|
67
|
+
|
|
68
|
+
```bash
|
|
69
|
+
export DUCKDB_CONNECTION_STRING='md:my_db'
|
|
70
|
+
export MOTHERDUCK_TOKEN='<your access token>'
|
|
71
|
+
```
|
|
72
|
+
|
|
73
|
+
```python
|
|
74
|
+
from agent_framework_duckdb import DuckDBStore
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
async def list_remote_tables() -> None:
|
|
78
|
+
async with DuckDBStore() as store:
|
|
79
|
+
print(await store.list_collection_names())
|
|
80
|
+
```
|
|
81
|
+
|
|
82
|
+
For an application-managed token, pass
|
|
83
|
+
`DuckDBStore(connection_string="md:my_db",
|
|
84
|
+
config={"motherduck_token": SecretString(token)})` (import `SecretString`
|
|
85
|
+
from `agent_framework`). `config` is forwarded to `duckdb.connect`; it is
|
|
86
|
+
not loaded from the connector's `.env` file. Keep tokens outside source control
|
|
87
|
+
and avoid embedding them in a URI. Other DuckDB-supported services use their
|
|
88
|
+
documented URI/configuration or an injected client; no service-specific
|
|
89
|
+
adapter is installed. DuckDB/MotherDuck may need network access and compatible
|
|
90
|
+
client/extension versions; authentication and service availability are managed
|
|
91
|
+
by DuckDB, not by this connector.
|
|
92
|
+
|
|
93
|
+
Alternatively, use `DuckDBStore(client=duckdb.connect(...))` or
|
|
94
|
+
`DuckDBCollection(Record, client=...)` to **borrow** an existing connection.
|
|
95
|
+
The caller owns and closes that connection; `client` cannot be combined with
|
|
96
|
+
`connection_string`, `config`, or `.env` options. Collections obtained from a
|
|
97
|
+
store borrow its client, so keep the store open while using them. All connector
|
|
98
|
+
operations on one store/collection run serially on one worker thread, not on
|
|
99
|
+
the event loop. Coordinate any *external* use of an injected connection
|
|
100
|
+
yourself. Cancelling an awaiter does not stop a DuckDB query already running
|
|
101
|
+
on that thread; `aclose()` waits for it to finish.
|
|
102
|
+
|
|
103
|
+
## Example
|
|
104
|
+
|
|
105
|
+
```python
|
|
106
|
+
from dataclasses import dataclass
|
|
107
|
+
from typing import Annotated
|
|
108
|
+
|
|
109
|
+
from agent_framework import Filter, VectorStoreField, vectorstoremodel
|
|
110
|
+
from agent_framework_duckdb import DuckDBStore
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
@vectorstoremodel(collection_name="articles")
|
|
114
|
+
@dataclass
|
|
115
|
+
class Article:
|
|
116
|
+
id: Annotated[str, VectorStoreField("key")]
|
|
117
|
+
text: Annotated[str, VectorStoreField("data")]
|
|
118
|
+
embedding: Annotated[list[float] | None, VectorStoreField("vector", dimensions=3)] = None
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
async def save_and_search() -> None:
|
|
122
|
+
async with DuckDBStore(connection_string="articles.duckdb") as store:
|
|
123
|
+
collection = store.get_collection(Article)
|
|
124
|
+
await collection.ensure_collection_exists()
|
|
125
|
+
await collection.upsert(
|
|
126
|
+
[Article("one", "DuckDB persists records", [1, 0, 0])],
|
|
127
|
+
generate_vectors=False,
|
|
128
|
+
)
|
|
129
|
+
|
|
130
|
+
async with DuckDBStore(connection_string="articles.duckdb") as reopened:
|
|
131
|
+
collection = reopened.get_collection(Article)
|
|
132
|
+
assert (await collection.get(["one"]))[0].text == "DuckDB persists records"
|
|
133
|
+
results = await collection.search(
|
|
134
|
+
vector=[1, 0, 0],
|
|
135
|
+
filter=Filter("text", "contains_text", "persists"),
|
|
136
|
+
)
|
|
137
|
+
async for result in results:
|
|
138
|
+
print(result["record"].text, result["score"])
|
|
139
|
+
```
|
|
140
|
+
|
|
141
|
+
Run a complete local example with
|
|
142
|
+
`uv run --package agent-framework-duckdb python packages/duckdb/samples/duckdb_vectors.py`
|
|
143
|
+
from the `python/` directory. It writes a default database file, reopens it,
|
|
144
|
+
and cleans up only its sample table.
|
|
145
|
+
|
|
146
|
+
## Capabilities and limits
|
|
147
|
+
|
|
148
|
+
- Tables are created by `ensure_collection_exists()`; `collection_exists()`,
|
|
149
|
+
`list_collection_names()`, and `ensure_collection_deleted()` provide the
|
|
150
|
+
corresponding lifecycle. Existing tables are **not migrated or reindexed**.
|
|
151
|
+
Identifiers are quoted, values are parameterized, and collection names refer
|
|
152
|
+
to tables in the connection's current database and schema.
|
|
153
|
+
- Batch upsert, retrieval, filtered/paged listing, and delete support string,
|
|
154
|
+
signed 64-bit integer, and UUID keys; string and UUID keys can be generated
|
|
155
|
+
when declared auto-generated. Records support string, integer, float,
|
|
156
|
+
boolean, UUID, bytes, date, timezone-aware datetime, JSON list/dict data,
|
|
157
|
+
and nullable dense vectors. Use `generate_vectors=False` for supplied
|
|
158
|
+
embeddings or pass a local `embedding_generator`. Retrieval excludes vectors
|
|
159
|
+
unless `include_vectors=True`.
|
|
160
|
+
- Exact SQL vector search supports `DEFAULT`/`cosine_distance`,
|
|
161
|
+
`cosine_similarity`, `euclidean_distance`, `dot_prod`, and
|
|
162
|
+
`negative_dot_prod`. The default score is cosine **distance**, so lower is
|
|
163
|
+
better and `score_threshold` is a maximum; similarity/dot-product scores
|
|
164
|
+
use a minimum threshold. Null vectors are omitted. Results are filtered
|
|
165
|
+
and paged **in DuckDB**, with the primary key as a stable tie-breaker.
|
|
166
|
+
- Portable filters support scalar equality/inequality, null/presence checks,
|
|
167
|
+
ordered scalar comparisons, `in`/`not_in`, text contains/prefix/suffix,
|
|
168
|
+
and AND/OR/NOT groups. They do not coerce booleans into numbers, and text
|
|
169
|
+
wildcards are literal. JSON equality/collection membership and nested
|
|
170
|
+
paths are not supported.
|
|
171
|
+
- No approximate vector indexes, keyword/hybrid search, full-text or explicit
|
|
172
|
+
data indexes, binary/sparse vectors, auto-generated integer keys, or
|
|
173
|
+
server-side vectorization are provided. Unsupported options raise errors.
|
|
174
|
+
DuckDB is an in-process database with file locking: multiple writer
|
|
175
|
+
**processes** cannot concurrently write the same local file. Local records
|
|
176
|
+
are unencrypted; choose and protect the database file appropriately.
|
|
177
|
+
Connector-created batch upserts are transactional. On a borrowed connection
|
|
178
|
+
the connector does not start/commit/roll back a caller transaction; without
|
|
179
|
+
one, a failed batch may have partially persisted.
|
|
180
|
+
|
|
@@ -0,0 +1,154 @@
|
|
|
1
|
+
# Agent Framework DuckDB vector store
|
|
2
|
+
|
|
3
|
+
Store and search typed records in DuckDB using the
|
|
4
|
+
[Microsoft Agent Framework](https://learn.microsoft.com/agent-framework/)
|
|
5
|
+
vector-store APIs. This alpha package uses the **DuckDB Python client only**:
|
|
6
|
+
local files, DuckDB-supported connection URIs (including MotherDuck), and
|
|
7
|
+
caller-provided `duckdb.DuckDBPyConnection` objects use the same implementation.
|
|
8
|
+
|
|
9
|
+
## Installation
|
|
10
|
+
|
|
11
|
+
```bash
|
|
12
|
+
pip install agent-framework-duckdb --pre
|
|
13
|
+
```
|
|
14
|
+
|
|
15
|
+
Python 3.10+ and DuckDB 1.4.1–1.5.x are required. The package also installs
|
|
16
|
+
`pytz`, which DuckDB uses when returning timezone-aware timestamps. Import
|
|
17
|
+
`DuckDBStore` and `DuckDBCollection` directly from `agent_framework_duckdb`;
|
|
18
|
+
this alpha package is not part of `agent-framework-core[all]`.
|
|
19
|
+
|
|
20
|
+
## Connections and persistence
|
|
21
|
+
|
|
22
|
+
`DuckDBStore()` (and a directly constructed `DuckDBCollection`) defaults to
|
|
23
|
+
**`agent-framework.duckdb` in the current working directory**. It is a normal
|
|
24
|
+
persistent database file, not an in-memory database or temporary file. Give
|
|
25
|
+
each application a suitable writable location with
|
|
26
|
+
`connection_string="path/to/vectors.duckdb"`; parent directories must already
|
|
27
|
+
exist. The connection is opened lazily on the first async operation.
|
|
28
|
+
`aclose()` or an async context manager drains pending operations and releases
|
|
29
|
+
the file. Opening a *new* store against the same filename restores its tables
|
|
30
|
+
and records.
|
|
31
|
+
|
|
32
|
+
`DUCKDB_CONNECTION_STRING` can instead provide a filename or a URI. Settings
|
|
33
|
+
precedence is **explicit `connection_string` > selected `.env` file >
|
|
34
|
+
environment > persistent default**. Select a file with `env_file_path` and
|
|
35
|
+
optionally `env_file_encoding`; `.env` files are never discovered implicitly.
|
|
36
|
+
An explicitly empty connection string is an error. The URI is held as an AF
|
|
37
|
+
`SecretString` and never printed or logged by the connector.
|
|
38
|
+
|
|
39
|
+
To use [MotherDuck](https://motherduck.com/docs/getting-started/interfaces/client-apis/python/installation-authentication/),
|
|
40
|
+
configure the *same* DuckDB client with a `md:` URI and its usual credentials,
|
|
41
|
+
for example:
|
|
42
|
+
|
|
43
|
+
```bash
|
|
44
|
+
export DUCKDB_CONNECTION_STRING='md:my_db'
|
|
45
|
+
export MOTHERDUCK_TOKEN='<your access token>'
|
|
46
|
+
```
|
|
47
|
+
|
|
48
|
+
```python
|
|
49
|
+
from agent_framework_duckdb import DuckDBStore
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
async def list_remote_tables() -> None:
|
|
53
|
+
async with DuckDBStore() as store:
|
|
54
|
+
print(await store.list_collection_names())
|
|
55
|
+
```
|
|
56
|
+
|
|
57
|
+
For an application-managed token, pass
|
|
58
|
+
`DuckDBStore(connection_string="md:my_db",
|
|
59
|
+
config={"motherduck_token": SecretString(token)})` (import `SecretString`
|
|
60
|
+
from `agent_framework`). `config` is forwarded to `duckdb.connect`; it is
|
|
61
|
+
not loaded from the connector's `.env` file. Keep tokens outside source control
|
|
62
|
+
and avoid embedding them in a URI. Other DuckDB-supported services use their
|
|
63
|
+
documented URI/configuration or an injected client; no service-specific
|
|
64
|
+
adapter is installed. DuckDB/MotherDuck may need network access and compatible
|
|
65
|
+
client/extension versions; authentication and service availability are managed
|
|
66
|
+
by DuckDB, not by this connector.
|
|
67
|
+
|
|
68
|
+
Alternatively, use `DuckDBStore(client=duckdb.connect(...))` or
|
|
69
|
+
`DuckDBCollection(Record, client=...)` to **borrow** an existing connection.
|
|
70
|
+
The caller owns and closes that connection; `client` cannot be combined with
|
|
71
|
+
`connection_string`, `config`, or `.env` options. Collections obtained from a
|
|
72
|
+
store borrow its client, so keep the store open while using them. All connector
|
|
73
|
+
operations on one store/collection run serially on one worker thread, not on
|
|
74
|
+
the event loop. Coordinate any *external* use of an injected connection
|
|
75
|
+
yourself. Cancelling an awaiter does not stop a DuckDB query already running
|
|
76
|
+
on that thread; `aclose()` waits for it to finish.
|
|
77
|
+
|
|
78
|
+
## Example
|
|
79
|
+
|
|
80
|
+
```python
|
|
81
|
+
from dataclasses import dataclass
|
|
82
|
+
from typing import Annotated
|
|
83
|
+
|
|
84
|
+
from agent_framework import Filter, VectorStoreField, vectorstoremodel
|
|
85
|
+
from agent_framework_duckdb import DuckDBStore
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
@vectorstoremodel(collection_name="articles")
|
|
89
|
+
@dataclass
|
|
90
|
+
class Article:
|
|
91
|
+
id: Annotated[str, VectorStoreField("key")]
|
|
92
|
+
text: Annotated[str, VectorStoreField("data")]
|
|
93
|
+
embedding: Annotated[list[float] | None, VectorStoreField("vector", dimensions=3)] = None
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
async def save_and_search() -> None:
|
|
97
|
+
async with DuckDBStore(connection_string="articles.duckdb") as store:
|
|
98
|
+
collection = store.get_collection(Article)
|
|
99
|
+
await collection.ensure_collection_exists()
|
|
100
|
+
await collection.upsert(
|
|
101
|
+
[Article("one", "DuckDB persists records", [1, 0, 0])],
|
|
102
|
+
generate_vectors=False,
|
|
103
|
+
)
|
|
104
|
+
|
|
105
|
+
async with DuckDBStore(connection_string="articles.duckdb") as reopened:
|
|
106
|
+
collection = reopened.get_collection(Article)
|
|
107
|
+
assert (await collection.get(["one"]))[0].text == "DuckDB persists records"
|
|
108
|
+
results = await collection.search(
|
|
109
|
+
vector=[1, 0, 0],
|
|
110
|
+
filter=Filter("text", "contains_text", "persists"),
|
|
111
|
+
)
|
|
112
|
+
async for result in results:
|
|
113
|
+
print(result["record"].text, result["score"])
|
|
114
|
+
```
|
|
115
|
+
|
|
116
|
+
Run a complete local example with
|
|
117
|
+
`uv run --package agent-framework-duckdb python packages/duckdb/samples/duckdb_vectors.py`
|
|
118
|
+
from the `python/` directory. It writes a default database file, reopens it,
|
|
119
|
+
and cleans up only its sample table.
|
|
120
|
+
|
|
121
|
+
## Capabilities and limits
|
|
122
|
+
|
|
123
|
+
- Tables are created by `ensure_collection_exists()`; `collection_exists()`,
|
|
124
|
+
`list_collection_names()`, and `ensure_collection_deleted()` provide the
|
|
125
|
+
corresponding lifecycle. Existing tables are **not migrated or reindexed**.
|
|
126
|
+
Identifiers are quoted, values are parameterized, and collection names refer
|
|
127
|
+
to tables in the connection's current database and schema.
|
|
128
|
+
- Batch upsert, retrieval, filtered/paged listing, and delete support string,
|
|
129
|
+
signed 64-bit integer, and UUID keys; string and UUID keys can be generated
|
|
130
|
+
when declared auto-generated. Records support string, integer, float,
|
|
131
|
+
boolean, UUID, bytes, date, timezone-aware datetime, JSON list/dict data,
|
|
132
|
+
and nullable dense vectors. Use `generate_vectors=False` for supplied
|
|
133
|
+
embeddings or pass a local `embedding_generator`. Retrieval excludes vectors
|
|
134
|
+
unless `include_vectors=True`.
|
|
135
|
+
- Exact SQL vector search supports `DEFAULT`/`cosine_distance`,
|
|
136
|
+
`cosine_similarity`, `euclidean_distance`, `dot_prod`, and
|
|
137
|
+
`negative_dot_prod`. The default score is cosine **distance**, so lower is
|
|
138
|
+
better and `score_threshold` is a maximum; similarity/dot-product scores
|
|
139
|
+
use a minimum threshold. Null vectors are omitted. Results are filtered
|
|
140
|
+
and paged **in DuckDB**, with the primary key as a stable tie-breaker.
|
|
141
|
+
- Portable filters support scalar equality/inequality, null/presence checks,
|
|
142
|
+
ordered scalar comparisons, `in`/`not_in`, text contains/prefix/suffix,
|
|
143
|
+
and AND/OR/NOT groups. They do not coerce booleans into numbers, and text
|
|
144
|
+
wildcards are literal. JSON equality/collection membership and nested
|
|
145
|
+
paths are not supported.
|
|
146
|
+
- No approximate vector indexes, keyword/hybrid search, full-text or explicit
|
|
147
|
+
data indexes, binary/sparse vectors, auto-generated integer keys, or
|
|
148
|
+
server-side vectorization are provided. Unsupported options raise errors.
|
|
149
|
+
DuckDB is an in-process database with file locking: multiple writer
|
|
150
|
+
**processes** cannot concurrently write the same local file. Local records
|
|
151
|
+
are unencrypted; choose and protect the database file appropriately.
|
|
152
|
+
Connector-created batch upserts are transactional. On a borrowed connection
|
|
153
|
+
the connector does not start/commit/roll back a caller transaction; without
|
|
154
|
+
one, a failed batch may have partially persisted.
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
# Copyright (c) Microsoft. All rights reserved.
|
|
2
|
+
|
|
3
|
+
"""DuckDB vector stores for Microsoft Agent Framework."""
|
|
4
|
+
|
|
5
|
+
import importlib.metadata
|
|
6
|
+
|
|
7
|
+
from ._vector_store import DuckDBCollection, DuckDBSettings, DuckDBStore
|
|
8
|
+
|
|
9
|
+
try:
|
|
10
|
+
__version__ = importlib.metadata.version(__name__)
|
|
11
|
+
except importlib.metadata.PackageNotFoundError:
|
|
12
|
+
__version__ = "0.0.0"
|
|
13
|
+
|
|
14
|
+
__all__ = ["DuckDBCollection", "DuckDBSettings", "DuckDBStore", "__version__"]
|
|
@@ -0,0 +1,788 @@
|
|
|
1
|
+
# Copyright (c) Microsoft. All rights reserved.
|
|
2
|
+
|
|
3
|
+
"""Persistent DuckDB collections with exact dense-vector search."""
|
|
4
|
+
|
|
5
|
+
from __future__ import annotations
|
|
6
|
+
|
|
7
|
+
import asyncio
|
|
8
|
+
import json
|
|
9
|
+
import math
|
|
10
|
+
import threading
|
|
11
|
+
from collections.abc import Callable, Mapping, Sequence
|
|
12
|
+
from concurrent.futures import Future, ThreadPoolExecutor
|
|
13
|
+
from datetime import date, datetime
|
|
14
|
+
from typing import Any, ClassVar, Generic, cast
|
|
15
|
+
from uuid import UUID, uuid4
|
|
16
|
+
|
|
17
|
+
import duckdb
|
|
18
|
+
from agent_framework import (
|
|
19
|
+
BaseVectorCollection,
|
|
20
|
+
BaseVectorSearch,
|
|
21
|
+
BaseVectorStore,
|
|
22
|
+
Filter,
|
|
23
|
+
FilterGroup,
|
|
24
|
+
SearchResults,
|
|
25
|
+
SecretString,
|
|
26
|
+
VectorStoreCollectionDefinition,
|
|
27
|
+
VectorStoreField,
|
|
28
|
+
load_settings,
|
|
29
|
+
)
|
|
30
|
+
from agent_framework._vector_filters import FilterExpression
|
|
31
|
+
from agent_framework._vectors import EmbeddingClient, SearchType, Vector
|
|
32
|
+
from agent_framework.exceptions import IntegrationException
|
|
33
|
+
from typing_extensions import TypedDict, TypeVar
|
|
34
|
+
|
|
35
|
+
KeyT = TypeVar("KeyT", default=Any)
|
|
36
|
+
ModelT = TypeVar("ModelT", default=Any)
|
|
37
|
+
ResultT = TypeVar("ResultT")
|
|
38
|
+
|
|
39
|
+
_DEFAULT_DATABASE = "agent-framework.duckdb"
|
|
40
|
+
_DATA_TYPES = {
|
|
41
|
+
"str": "VARCHAR",
|
|
42
|
+
"int": "BIGINT",
|
|
43
|
+
"float": "DOUBLE",
|
|
44
|
+
"bool": "BOOLEAN",
|
|
45
|
+
"UUID": "UUID",
|
|
46
|
+
"bytes": "BLOB",
|
|
47
|
+
"date": "DATE",
|
|
48
|
+
"datetime": "TIMESTAMPTZ",
|
|
49
|
+
"list": "JSON",
|
|
50
|
+
"dict": "JSON",
|
|
51
|
+
}
|
|
52
|
+
_VECTOR_TYPES = {"float": "DOUBLE", "float64": "DOUBLE", "float32": "FLOAT"}
|
|
53
|
+
_METRICS = {
|
|
54
|
+
"DEFAULT": ("array_cosine_distance", False),
|
|
55
|
+
"cosine_distance": ("array_cosine_distance", False),
|
|
56
|
+
"cosine_similarity": ("array_cosine_similarity", True),
|
|
57
|
+
"euclidean_distance": ("array_distance", False),
|
|
58
|
+
"dot_prod": ("array_inner_product", True),
|
|
59
|
+
"negative_dot_prod": ("-array_inner_product", False),
|
|
60
|
+
}
|
|
61
|
+
_ORDER_OPERATORS = {"gt": ">", "gte": ">=", "lt": "<", "lte": "<="}
|
|
62
|
+
_ASCII_FOLD = str.maketrans("ABCDEFGHIJKLMNOPQRSTUVWXYZ", "abcdefghijklmnopqrstuvwxyz")
|
|
63
|
+
_TABLE_SCOPE = (
|
|
64
|
+
"FROM information_schema.tables "
|
|
65
|
+
"WHERE table_catalog = current_database() AND table_schema = current_schema() "
|
|
66
|
+
"AND table_type = 'BASE TABLE'"
|
|
67
|
+
)
|
|
68
|
+
_LIST_TABLES = f"SELECT table_name {_TABLE_SCOPE} ORDER BY table_name"
|
|
69
|
+
# DuckDB folds only ASCII identifier case; C collation avoids a connection's Unicode NOCASE setting.
|
|
70
|
+
_TABLE_EXISTS = (
|
|
71
|
+
f"SELECT 1 {_TABLE_SCOPE} "
|
|
72
|
+
"AND (translate(table_name, 'ABCDEFGHIJKLMNOPQRSTUVWXYZ', 'abcdefghijklmnopqrstuvwxyz') COLLATE \"C\") = "
|
|
73
|
+
"(translate(?, 'ABCDEFGHIJKLMNOPQRSTUVWXYZ', 'abcdefghijklmnopqrstuvwxyz') COLLATE \"C\")"
|
|
74
|
+
)
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
class DuckDBSettings(TypedDict, total=False):
|
|
78
|
+
"""Connection settings resolved from arguments, a selected .env file, or ``DUCKDB_`` variables."""
|
|
79
|
+
|
|
80
|
+
connection_string: SecretString | None
|
|
81
|
+
"""Local database filename or DuckDB URI, resolved from ``DUCKDB_CONNECTION_STRING``."""
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def _identifier(name: str) -> str:
|
|
85
|
+
if not isinstance(name, str) or not name or "\0" in name:
|
|
86
|
+
raise ValueError("DuckDB identifiers must be nonempty strings without NUL.")
|
|
87
|
+
return '"' + name.replace('"', '""') + '"'
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def _check_options(options: Mapping[str, Any] | None) -> None:
|
|
91
|
+
if options:
|
|
92
|
+
raise NotImplementedError(f"Unsupported DuckDB operation option(s): {', '.join(sorted(options))}.")
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def _execute_statement(
|
|
96
|
+
connection: duckdb.DuckDBPyConnection,
|
|
97
|
+
statement: str,
|
|
98
|
+
parameters: Sequence[Any] = (),
|
|
99
|
+
) -> None:
|
|
100
|
+
connection.execute(statement, parameters)
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
def _column_type(field: VectorStoreField) -> str:
|
|
104
|
+
if field.field_type == "vector":
|
|
105
|
+
if type(field.dimensions) is not int or field.dimensions <= 0:
|
|
106
|
+
raise ValueError(f"Vector field '{field.name}' needs positive integer dimensions.")
|
|
107
|
+
if field.type_ not in _VECTOR_TYPES:
|
|
108
|
+
raise NotImplementedError(f"Unsupported DuckDB vector type '{field.type_}'.")
|
|
109
|
+
return f"{_VECTOR_TYPES[field.type_]}[{field.dimensions}]"
|
|
110
|
+
if field.type_ not in _DATA_TYPES:
|
|
111
|
+
raise NotImplementedError(f"Field '{field.name}' needs a supported explicit type; got '{field.type_}'.")
|
|
112
|
+
return _DATA_TYPES[field.type_]
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def _validate_json(value: Any) -> None:
|
|
116
|
+
if value is None or type(value) in (str, int, bool):
|
|
117
|
+
return
|
|
118
|
+
if type(value) is float:
|
|
119
|
+
if not math.isfinite(value):
|
|
120
|
+
raise ValueError("JSON values must contain finite numbers.")
|
|
121
|
+
return
|
|
122
|
+
if type(value) is list:
|
|
123
|
+
for item in cast(list[Any], value):
|
|
124
|
+
_validate_json(item)
|
|
125
|
+
return
|
|
126
|
+
if type(value) is dict:
|
|
127
|
+
for key, item in cast(dict[Any, Any], value).items():
|
|
128
|
+
if not isinstance(key, str):
|
|
129
|
+
raise TypeError("JSON object keys must be strings.")
|
|
130
|
+
_validate_json(item)
|
|
131
|
+
return
|
|
132
|
+
raise TypeError("JSON fields require JSON-compatible lists or string-keyed dictionaries.")
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
def _prepare_vector(field: VectorStoreField, value: Any) -> list[float] | None:
|
|
136
|
+
if value is None:
|
|
137
|
+
return None
|
|
138
|
+
if not isinstance(value, Sequence) or isinstance(value, (str, bytes, bytearray)):
|
|
139
|
+
raise TypeError(f"Vector field '{field.name}' requires a dense numeric sequence.")
|
|
140
|
+
elements = cast(Sequence[Any], value)
|
|
141
|
+
if len(elements) != field.dimensions:
|
|
142
|
+
raise ValueError(f"Vector field '{field.name}' expects {field.dimensions} dimensions; got {len(elements)}.")
|
|
143
|
+
if any(type(item) not in (float, int) or not math.isfinite(item) for item in elements):
|
|
144
|
+
raise ValueError(f"Vector field '{field.name}' requires finite numeric elements without NULLs.")
|
|
145
|
+
if field.distance_function in (None, "DEFAULT", "cosine_distance", "cosine_similarity") and not any(elements):
|
|
146
|
+
raise ValueError(f"Cosine vector field '{field.name}' requires a nonzero vector.")
|
|
147
|
+
return [float(item) for item in elements]
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
def _prepare_value(field: VectorStoreField, value: Any) -> Any:
|
|
151
|
+
if field.field_type == "vector":
|
|
152
|
+
return _prepare_vector(field, value)
|
|
153
|
+
if value is None:
|
|
154
|
+
if field.field_type == "key":
|
|
155
|
+
raise ValueError("DuckDB keys cannot be null.")
|
|
156
|
+
return None
|
|
157
|
+
kind = field.type_
|
|
158
|
+
if kind == "str" and isinstance(value, str):
|
|
159
|
+
return value
|
|
160
|
+
if kind == "int" and type(value) is int:
|
|
161
|
+
if not -(2**63) <= value < 2**63:
|
|
162
|
+
raise ValueError(f"Field '{field.name}' exceeds the BIGINT range.")
|
|
163
|
+
return value
|
|
164
|
+
if kind == "float" and type(value) in (float, int):
|
|
165
|
+
if not math.isfinite(value):
|
|
166
|
+
raise ValueError(f"Field '{field.name}' requires a finite number.")
|
|
167
|
+
return float(value)
|
|
168
|
+
if kind == "bool" and type(value) is bool:
|
|
169
|
+
return value
|
|
170
|
+
if kind == "UUID" and isinstance(value, (str, UUID)):
|
|
171
|
+
return UUID(str(value))
|
|
172
|
+
if kind == "bytes" and isinstance(value, bytes):
|
|
173
|
+
return value
|
|
174
|
+
if kind == "date" and type(value) is date:
|
|
175
|
+
return value
|
|
176
|
+
if kind == "date" and isinstance(value, str):
|
|
177
|
+
return date.fromisoformat(value)
|
|
178
|
+
if kind == "datetime":
|
|
179
|
+
parsed = datetime.fromisoformat(value.replace("Z", "+00:00")) if isinstance(value, str) else value
|
|
180
|
+
if isinstance(parsed, datetime):
|
|
181
|
+
if parsed.tzinfo is None or parsed.utcoffset() is None:
|
|
182
|
+
raise ValueError(f"Datetime field '{field.name}' requires a timezone.")
|
|
183
|
+
return parsed
|
|
184
|
+
if kind in ("list", "dict") and type(value) is (list if kind == "list" else dict):
|
|
185
|
+
_validate_json(value)
|
|
186
|
+
return json.dumps(value, allow_nan=False)
|
|
187
|
+
raise TypeError(f"Field '{field.name}' requires a value of type '{kind}'.")
|
|
188
|
+
|
|
189
|
+
|
|
190
|
+
def _create_client(
|
|
191
|
+
connection_string: str | SecretString | None,
|
|
192
|
+
*,
|
|
193
|
+
client: duckdb.DuckDBPyConnection | None,
|
|
194
|
+
config: Mapping[str, str | SecretString] | None,
|
|
195
|
+
env_file_path: str | None,
|
|
196
|
+
env_file_encoding: str | None,
|
|
197
|
+
) -> _Client:
|
|
198
|
+
if client is not None:
|
|
199
|
+
if any(value is not None for value in (connection_string, config, env_file_path, env_file_encoding)):
|
|
200
|
+
raise ValueError("client cannot be combined with connection_string, config, or .env options.")
|
|
201
|
+
if not isinstance(client, duckdb.DuckDBPyConnection):
|
|
202
|
+
raise TypeError("client must be a DuckDBPyConnection.")
|
|
203
|
+
return _Client(None, client=client, config=None)
|
|
204
|
+
settings = load_settings(
|
|
205
|
+
DuckDBSettings,
|
|
206
|
+
env_prefix="DUCKDB_",
|
|
207
|
+
connection_string=connection_string,
|
|
208
|
+
env_file_path=env_file_path,
|
|
209
|
+
env_file_encoding=env_file_encoding,
|
|
210
|
+
)
|
|
211
|
+
address = settings.get("connection_string")
|
|
212
|
+
if address is not None and not isinstance(address, SecretString):
|
|
213
|
+
raise TypeError("connection_string must be a string or SecretString.")
|
|
214
|
+
if address is None:
|
|
215
|
+
address = SecretString(_DEFAULT_DATABASE)
|
|
216
|
+
if not address.get_secret_value().strip():
|
|
217
|
+
raise ValueError("connection_string must not be empty.")
|
|
218
|
+
if config is not None:
|
|
219
|
+
if not isinstance(config, Mapping):
|
|
220
|
+
raise TypeError("config must be a mapping of DuckDB settings.")
|
|
221
|
+
for key, value in config.items():
|
|
222
|
+
if not isinstance(key, str) or not key or not isinstance(value, (str, SecretString)):
|
|
223
|
+
raise TypeError("DuckDB config keys and values must be nonempty string keys and string values.")
|
|
224
|
+
return _Client(address, client=None, config=config)
|
|
225
|
+
|
|
226
|
+
|
|
227
|
+
class _Client:
|
|
228
|
+
"""Run every operation on one worker, including connection setup and shutdown."""
|
|
229
|
+
|
|
230
|
+
def __init__(
|
|
231
|
+
self,
|
|
232
|
+
address: SecretString | None,
|
|
233
|
+
*,
|
|
234
|
+
client: duckdb.DuckDBPyConnection | None,
|
|
235
|
+
config: Mapping[str, str | SecretString] | None,
|
|
236
|
+
) -> None:
|
|
237
|
+
self.owned = client is None
|
|
238
|
+
self._address = address
|
|
239
|
+
self._connection = client
|
|
240
|
+
self._config = {key: SecretString(value) for key, value in (config or {}).items()}
|
|
241
|
+
self._executor = ThreadPoolExecutor(max_workers=1, thread_name_prefix="af-duckdb")
|
|
242
|
+
self._lifecycle_lock = threading.Lock()
|
|
243
|
+
self._close_future: Future[None] | None = None
|
|
244
|
+
|
|
245
|
+
def _execute(self, operation: Callable[[duckdb.DuckDBPyConnection], ResultT]) -> ResultT:
|
|
246
|
+
connection = self._connection
|
|
247
|
+
if connection is None:
|
|
248
|
+
if self._address is None:
|
|
249
|
+
raise RuntimeError("A DuckDB connection or database path is required.")
|
|
250
|
+
try:
|
|
251
|
+
connection = duckdb.connect(
|
|
252
|
+
database=self._address.get_secret_value(),
|
|
253
|
+
config={key: value.get_secret_value() for key, value in self._config.items()},
|
|
254
|
+
)
|
|
255
|
+
except duckdb.Error:
|
|
256
|
+
connection = None
|
|
257
|
+
if connection is None:
|
|
258
|
+
raise IntegrationException("DuckDB connection failed. Verify the connection string and configuration.")
|
|
259
|
+
self._connection = connection
|
|
260
|
+
try:
|
|
261
|
+
return operation(connection)
|
|
262
|
+
except duckdb.Error as exc:
|
|
263
|
+
raise IntegrationException("DuckDB operation failed; inspect the chained driver exception.") from exc
|
|
264
|
+
|
|
265
|
+
async def run(self, operation: Callable[[duckdb.DuckDBPyConnection], ResultT]) -> ResultT:
|
|
266
|
+
"""Submit one complete database operation without blocking the event loop."""
|
|
267
|
+
with self._lifecycle_lock:
|
|
268
|
+
if self._close_future is not None:
|
|
269
|
+
raise RuntimeError("The DuckDB client is closed.")
|
|
270
|
+
future = self._executor.submit(self._execute, operation)
|
|
271
|
+
return await asyncio.wrap_future(future)
|
|
272
|
+
|
|
273
|
+
def _close(self) -> None:
|
|
274
|
+
if self.owned and self._connection is not None:
|
|
275
|
+
try:
|
|
276
|
+
self._connection.close()
|
|
277
|
+
except duckdb.Error as exc:
|
|
278
|
+
raise IntegrationException("DuckDB close failed; inspect the chained driver exception.") from exc
|
|
279
|
+
|
|
280
|
+
async def aclose(self) -> None:
|
|
281
|
+
"""Wait for queued work, close owned resources, and leave injected connections open."""
|
|
282
|
+
with self._lifecycle_lock:
|
|
283
|
+
if self._close_future is None:
|
|
284
|
+
self._close_future = self._executor.submit(self._close)
|
|
285
|
+
self._executor.shutdown(wait=False)
|
|
286
|
+
close_future = self._close_future
|
|
287
|
+
await asyncio.shield(asyncio.wrap_future(close_future))
|
|
288
|
+
|
|
289
|
+
|
|
290
|
+
class _FilterCompiler:
|
|
291
|
+
"""Compile supported portable filters into two-valued, parameterized SQL."""
|
|
292
|
+
|
|
293
|
+
def __init__(self, definition: VectorStoreCollectionDefinition) -> None:
|
|
294
|
+
self.definition = definition
|
|
295
|
+
self.parameters: list[Any] = []
|
|
296
|
+
|
|
297
|
+
def compile(self, expression: FilterExpression | None) -> tuple[str, list[Any]]:
|
|
298
|
+
"""Return a SQL predicate and its bound arguments."""
|
|
299
|
+
if expression is None:
|
|
300
|
+
return "TRUE", []
|
|
301
|
+
return self._condition(expression), self.parameters
|
|
302
|
+
|
|
303
|
+
def _equality(self, field: VectorStoreField, column: str, value: Any) -> str:
|
|
304
|
+
if field.type_ in ("list", "dict"):
|
|
305
|
+
raise NotImplementedError("DuckDB JSON equality and collection filters are not supported.")
|
|
306
|
+
if value is None:
|
|
307
|
+
return f"{column} IS NULL"
|
|
308
|
+
if field.type_ in ("int", "float") and type(value) in (int, float):
|
|
309
|
+
if not math.isfinite(value):
|
|
310
|
+
raise ValueError("Numeric filter operands must be finite.")
|
|
311
|
+
adapted = value
|
|
312
|
+
else:
|
|
313
|
+
try:
|
|
314
|
+
adapted = _prepare_value(field, value)
|
|
315
|
+
except TypeError:
|
|
316
|
+
return "FALSE"
|
|
317
|
+
self.parameters.append(adapted)
|
|
318
|
+
return f"{column} IS NOT DISTINCT FROM ?"
|
|
319
|
+
|
|
320
|
+
def _condition(self, expression: FilterExpression) -> str:
|
|
321
|
+
if isinstance(expression, FilterGroup):
|
|
322
|
+
parts = [self._condition(child) for child in expression.filters]
|
|
323
|
+
if expression.operator == "not":
|
|
324
|
+
return f"(NOT ({parts[0]}))"
|
|
325
|
+
joiner = " AND " if expression.operator == "and" else " OR "
|
|
326
|
+
return f"({joiner.join(parts)})"
|
|
327
|
+
if not isinstance(expression, Filter):
|
|
328
|
+
raise TypeError("filter must be a Filter or FilterGroup.")
|
|
329
|
+
if "." in expression.field_name:
|
|
330
|
+
raise NotImplementedError("DuckDB does not support nested filter paths.")
|
|
331
|
+
field = self.definition.try_get_field(expression.field_name)
|
|
332
|
+
if field is None:
|
|
333
|
+
raise ValueError(f"Unknown DuckDB field '{expression.field_name}'.")
|
|
334
|
+
if field.field_type == "vector":
|
|
335
|
+
raise NotImplementedError("Filtering vector columns is not supported.")
|
|
336
|
+
column = f"t.{_identifier(field.storage_name or field.name)}"
|
|
337
|
+
op, value = expression.operator, expression.value
|
|
338
|
+
if op == "exists":
|
|
339
|
+
return "TRUE"
|
|
340
|
+
if op == "is_null":
|
|
341
|
+
return f"{column} IS NULL"
|
|
342
|
+
if op == "is_not_null":
|
|
343
|
+
return f"{column} IS NOT NULL"
|
|
344
|
+
if op in ("eq", "ne"):
|
|
345
|
+
equality = self._equality(field, column, value)
|
|
346
|
+
return equality if op == "eq" else f"(NOT ({equality}))"
|
|
347
|
+
if op in ("in", "not_in"):
|
|
348
|
+
parts = [self._equality(field, column, item) for item in value]
|
|
349
|
+
choices = f"({' OR '.join(parts)})" if parts else "FALSE"
|
|
350
|
+
membership = choices if op == "in" else f"(NOT {choices})"
|
|
351
|
+
return f"({column} IS NOT NULL AND {membership})"
|
|
352
|
+
if op in (*_ORDER_OPERATORS, "between"):
|
|
353
|
+
if field.type_ not in ("str", "int", "float", "date", "datetime"):
|
|
354
|
+
raise NotImplementedError(f"Ordered filtering is not supported for '{field.type_}'.")
|
|
355
|
+
operands: Sequence[Any] = value if op == "between" else [value]
|
|
356
|
+
for operand in operands:
|
|
357
|
+
if operand is None or type(operand) is bool:
|
|
358
|
+
raise TypeError("Ordered filter operands must be non-null scalars of the column's type.")
|
|
359
|
+
if field.type_ in ("int", "float") and type(operand) in (int, float):
|
|
360
|
+
if not math.isfinite(operand):
|
|
361
|
+
raise ValueError("Ordered filter numbers must be finite.")
|
|
362
|
+
self.parameters.append(operand)
|
|
363
|
+
else:
|
|
364
|
+
self.parameters.append(_prepare_value(field, operand))
|
|
365
|
+
if op == "between":
|
|
366
|
+
return f"({column} BETWEEN ? AND ?) IS TRUE"
|
|
367
|
+
return f"({column} {_ORDER_OPERATORS[op]} ?) IS TRUE"
|
|
368
|
+
if op in ("contains_text", "starts_with", "ends_with"):
|
|
369
|
+
if field.type_ != "str":
|
|
370
|
+
raise TypeError("Text filters require a string field.")
|
|
371
|
+
if not isinstance(value, str):
|
|
372
|
+
raise TypeError("Text filters require a string operand.")
|
|
373
|
+
escaped = value.replace("!", "!!").replace("%", "!%").replace("_", "!_")
|
|
374
|
+
pattern = ("%" if op != "starts_with" else "") + escaped + ("%" if op != "ends_with" else "")
|
|
375
|
+
self.parameters.append(pattern)
|
|
376
|
+
return f"({column} LIKE ? ESCAPE '!') IS TRUE"
|
|
377
|
+
raise NotImplementedError(f"Unsupported DuckDB filter operator '{op}'.")
|
|
378
|
+
|
|
379
|
+
|
|
380
|
+
class DuckDBCollection(BaseVectorCollection[KeyT, ModelT], BaseVectorSearch[KeyT, ModelT], Generic[KeyT, ModelT]):
|
|
381
|
+
"""One DuckDB table with exact dense-vector search and typed record codecs."""
|
|
382
|
+
|
|
383
|
+
supported_key_types: ClassVar[set[str] | None] = {"str", "int", "UUID"}
|
|
384
|
+
supported_vector_types: ClassVar[set[str] | None] = set(_VECTOR_TYPES)
|
|
385
|
+
supported_search_types: ClassVar[set[SearchType]] = {"vector"}
|
|
386
|
+
|
|
387
|
+
def __init__(
|
|
388
|
+
self,
|
|
389
|
+
record_type: type[ModelT],
|
|
390
|
+
*,
|
|
391
|
+
connection_string: str | SecretString | None = None,
|
|
392
|
+
client: duckdb.DuckDBPyConnection | None = None,
|
|
393
|
+
config: Mapping[str, str | SecretString] | None = None,
|
|
394
|
+
definition: VectorStoreCollectionDefinition | None = None,
|
|
395
|
+
collection_name: str | None = None,
|
|
396
|
+
embedding_generator: EmbeddingClient | None = None,
|
|
397
|
+
env_file_path: str | None = None,
|
|
398
|
+
env_file_encoding: str | None = None,
|
|
399
|
+
_shared_client: _Client | None = None,
|
|
400
|
+
) -> None:
|
|
401
|
+
"""Initialize a collection; the first async operation opens an owned connection.
|
|
402
|
+
|
|
403
|
+
Args:
|
|
404
|
+
record_type: Registered typed model or ``dict`` with an explicit definition.
|
|
405
|
+
connection_string: Persistent filename or DuckDB URI. Defaults to ``agent-framework.duckdb``.
|
|
406
|
+
client: Borrowed DuckDB connection; mutually exclusive with connection settings.
|
|
407
|
+
config: DuckDB connection settings, optionally wrapping secret values in ``SecretString``.
|
|
408
|
+
definition: Explicit definition for dictionary records.
|
|
409
|
+
collection_name: Table name, overriding the model definition.
|
|
410
|
+
embedding_generator: Local embedding client for generated vectors.
|
|
411
|
+
env_file_path: Optional selected .env file for ``DUCKDB_CONNECTION_STRING``.
|
|
412
|
+
env_file_encoding: Encoding of the selected .env file.
|
|
413
|
+
_shared_client: Internal connection owned by a parent store.
|
|
414
|
+
"""
|
|
415
|
+
super().__init__(
|
|
416
|
+
record_type,
|
|
417
|
+
definition=definition,
|
|
418
|
+
collection_name=collection_name,
|
|
419
|
+
embedding_generator=embedding_generator,
|
|
420
|
+
managed_client=client is None and _shared_client is None,
|
|
421
|
+
)
|
|
422
|
+
self._table = _identifier(self.collection_name)
|
|
423
|
+
names = [field.storage_name or field.name for field in self.definition.fields]
|
|
424
|
+
if len({name.translate(_ASCII_FOLD) for name in names}) != len(names):
|
|
425
|
+
raise ValueError("DuckDB column names must be unique ignoring case.")
|
|
426
|
+
self._fields = tuple(self.definition.fields)
|
|
427
|
+
for name in names:
|
|
428
|
+
_identifier(name)
|
|
429
|
+
if _shared_client is not None and any(
|
|
430
|
+
value is not None for value in (connection_string, client, config, env_file_path, env_file_encoding)
|
|
431
|
+
):
|
|
432
|
+
raise ValueError("A store collection cannot override its shared connection settings.")
|
|
433
|
+
self._client = _shared_client or _create_client(
|
|
434
|
+
connection_string,
|
|
435
|
+
client=client,
|
|
436
|
+
config=config,
|
|
437
|
+
env_file_path=env_file_path,
|
|
438
|
+
env_file_encoding=env_file_encoding,
|
|
439
|
+
)
|
|
440
|
+
self._owns_wrapper = _shared_client is None
|
|
441
|
+
|
|
442
|
+
def _validate_data_model(self) -> None:
|
|
443
|
+
super()._validate_data_model()
|
|
444
|
+
for field in self.definition.fields:
|
|
445
|
+
_column_type(field)
|
|
446
|
+
if field.is_full_text_indexed or field.is_indexed:
|
|
447
|
+
raise NotImplementedError("DuckDB full-text and explicit data indexes are not supported.")
|
|
448
|
+
if field.field_type == "vector":
|
|
449
|
+
if field.index_kind not in ("default", "flat"):
|
|
450
|
+
raise NotImplementedError(f"Unsupported DuckDB vector index kind '{field.index_kind}'.")
|
|
451
|
+
if field.distance_function not in _METRICS:
|
|
452
|
+
raise NotImplementedError(f"Unsupported DuckDB distance function '{field.distance_function}'.")
|
|
453
|
+
if any(name.startswith("duckdb.") for name in field.provider_annotations):
|
|
454
|
+
raise NotImplementedError("DuckDB provider annotations are not supported.")
|
|
455
|
+
key = self.definition.key_field
|
|
456
|
+
if key.is_auto_generated and key.type_ == "int":
|
|
457
|
+
raise NotImplementedError("DuckDB auto-generated integer keys are not supported.")
|
|
458
|
+
|
|
459
|
+
async def __aexit__(self, exc_type: Any, exc_value: Any, traceback: Any) -> None:
|
|
460
|
+
"""Release a direct collection's connection; store collections leave ownership to the store."""
|
|
461
|
+
await self.aclose()
|
|
462
|
+
|
|
463
|
+
async def aclose(self) -> None:
|
|
464
|
+
"""Close this collection's wrapper, leaving injected or store-owned connections open."""
|
|
465
|
+
if self._owns_wrapper:
|
|
466
|
+
await self._client.aclose()
|
|
467
|
+
|
|
468
|
+
async def collection_exists(self, *, operation_options: Mapping[str, Any] | None = None) -> bool:
|
|
469
|
+
"""Check for this table in the active DuckDB database and schema."""
|
|
470
|
+
_check_options(operation_options)
|
|
471
|
+
return await self._client.run(
|
|
472
|
+
lambda connection: connection.execute(_TABLE_EXISTS, [self.collection_name]).fetchone() is not None
|
|
473
|
+
)
|
|
474
|
+
|
|
475
|
+
async def ensure_collection_exists(self, *, operation_options: Mapping[str, Any] | None = None) -> None:
|
|
476
|
+
"""Create the table if absent; existing schemas are not modified."""
|
|
477
|
+
_check_options(operation_options)
|
|
478
|
+
columns = [
|
|
479
|
+
f"{_identifier(field.storage_name or field.name)} {_column_type(field)}"
|
|
480
|
+
+ (" PRIMARY KEY" if field.field_type == "key" else "")
|
|
481
|
+
for field in self._fields
|
|
482
|
+
]
|
|
483
|
+
await self._client.run(
|
|
484
|
+
lambda connection: _execute_statement(
|
|
485
|
+
connection, f"CREATE TABLE IF NOT EXISTS {self._table} ({', '.join(columns)})"
|
|
486
|
+
)
|
|
487
|
+
)
|
|
488
|
+
|
|
489
|
+
async def ensure_collection_deleted(self, *, operation_options: Mapping[str, Any] | None = None) -> None:
|
|
490
|
+
"""Drop this table if present, without affecting other tables."""
|
|
491
|
+
_check_options(operation_options)
|
|
492
|
+
await self._client.run(lambda connection: _execute_statement(connection, f"DROP TABLE IF EXISTS {self._table}"))
|
|
493
|
+
|
|
494
|
+
def _deserialize_store_models_to_dicts(
|
|
495
|
+
self,
|
|
496
|
+
records: Sequence[Any],
|
|
497
|
+
*,
|
|
498
|
+
context: Mapping[str, Any] | None = None,
|
|
499
|
+
) -> Sequence[dict[str, Any]]:
|
|
500
|
+
decoded = super()._deserialize_store_models_to_dicts(records, context=context)
|
|
501
|
+
for record in decoded:
|
|
502
|
+
for field in self._fields:
|
|
503
|
+
name = field.storage_name or field.name
|
|
504
|
+
value = record.get(name)
|
|
505
|
+
if field.field_type == "vector" and isinstance(value, tuple):
|
|
506
|
+
record[name] = list(cast(tuple[Any, ...], value))
|
|
507
|
+
elif field.type_ in ("list", "dict") and value is not None:
|
|
508
|
+
record[name] = json.loads(value)
|
|
509
|
+
return decoded
|
|
510
|
+
|
|
511
|
+
def _columns(self, include_vectors: bool) -> list[str]:
|
|
512
|
+
return [
|
|
513
|
+
field.storage_name or field.name
|
|
514
|
+
for field in self._fields
|
|
515
|
+
if include_vectors or field.field_type != "vector"
|
|
516
|
+
]
|
|
517
|
+
|
|
518
|
+
def _order_by(self, order_by: Mapping[str, bool] | None) -> str:
|
|
519
|
+
parts: list[str] = []
|
|
520
|
+
for name, ascending in (order_by or {}).items():
|
|
521
|
+
if type(ascending) is not bool:
|
|
522
|
+
raise TypeError("Order directions must be booleans.")
|
|
523
|
+
if "." in name:
|
|
524
|
+
raise NotImplementedError("DuckDB does not support nested order paths.")
|
|
525
|
+
field = self.definition.try_get_field(name)
|
|
526
|
+
if field is None:
|
|
527
|
+
raise ValueError(f"Unknown DuckDB order field '{name}'.")
|
|
528
|
+
if field.field_type == "vector" or field.type_ in ("list", "dict"):
|
|
529
|
+
raise NotImplementedError(f"Ordering field '{name}' is not supported.")
|
|
530
|
+
direction = "ASC" if ascending else "DESC"
|
|
531
|
+
parts.append(f"t.{_identifier(field.storage_name or field.name)} {direction} NULLS LAST")
|
|
532
|
+
if self.definition.key_field.name not in (order_by or {}):
|
|
533
|
+
parts.append(f"t.{_identifier(self.definition.key_field_storage_name)} ASC")
|
|
534
|
+
return ", ".join(parts)
|
|
535
|
+
|
|
536
|
+
async def _inner_upsert(
|
|
537
|
+
self,
|
|
538
|
+
records: Sequence[Any],
|
|
539
|
+
*,
|
|
540
|
+
operation_options: Mapping[str, Any] | None = None,
|
|
541
|
+
) -> Sequence[KeyT]:
|
|
542
|
+
_check_options(operation_options)
|
|
543
|
+
key = self.definition.key_field
|
|
544
|
+
key_name = self.definition.key_field_storage_name
|
|
545
|
+
names = [field.storage_name or field.name for field in self._fields]
|
|
546
|
+
rows: list[tuple[Any, ...]] = []
|
|
547
|
+
keys: list[KeyT] = []
|
|
548
|
+
for record in records:
|
|
549
|
+
if key_name not in record:
|
|
550
|
+
if not key.is_auto_generated:
|
|
551
|
+
raise ValueError(f"Record is missing vector store field '{key.name}'.")
|
|
552
|
+
record = {**record, key_name: str(uuid4()) if key.type_ == "str" else uuid4()}
|
|
553
|
+
row = tuple(_prepare_value(field, record[name]) for field, name in zip(self._fields, names, strict=True))
|
|
554
|
+
rows.append(row)
|
|
555
|
+
keys.append(cast(KeyT, row[names.index(key_name)]))
|
|
556
|
+
if not rows:
|
|
557
|
+
return []
|
|
558
|
+
identifiers = ", ".join(map(_identifier, names))
|
|
559
|
+
placeholders = ", ".join("?" for _ in names)
|
|
560
|
+
updates = [name for name in names if name != key_name]
|
|
561
|
+
conflict = (
|
|
562
|
+
f"DO UPDATE SET {', '.join(f'{_identifier(name)} = EXCLUDED.{_identifier(name)}' for name in updates)}" # nosec B608
|
|
563
|
+
if updates
|
|
564
|
+
else "DO NOTHING"
|
|
565
|
+
)
|
|
566
|
+
statement = (
|
|
567
|
+
f"INSERT INTO {self._table} ({identifiers}) VALUES ({placeholders}) " # ruff: ignore[hardcoded-sql-expression] # nosec B608
|
|
568
|
+
f"ON CONFLICT ({_identifier(key_name)}) {conflict}"
|
|
569
|
+
)
|
|
570
|
+
|
|
571
|
+
def write(connection: duckdb.DuckDBPyConnection) -> None:
|
|
572
|
+
if self._client.owned:
|
|
573
|
+
connection.execute("BEGIN TRANSACTION")
|
|
574
|
+
try:
|
|
575
|
+
connection.executemany(statement, rows)
|
|
576
|
+
if self._client.owned:
|
|
577
|
+
connection.execute("COMMIT")
|
|
578
|
+
except Exception:
|
|
579
|
+
if self._client.owned:
|
|
580
|
+
connection.execute("ROLLBACK")
|
|
581
|
+
raise
|
|
582
|
+
|
|
583
|
+
await self._client.run(write)
|
|
584
|
+
return keys
|
|
585
|
+
|
|
586
|
+
async def _inner_get(
|
|
587
|
+
self,
|
|
588
|
+
*,
|
|
589
|
+
keys: Sequence[KeyT] | None = None,
|
|
590
|
+
filter: FilterExpression | None = None,
|
|
591
|
+
top: int = 10,
|
|
592
|
+
skip: int = 0,
|
|
593
|
+
order_by: Mapping[str, bool] | None = None,
|
|
594
|
+
include_vectors: bool = False,
|
|
595
|
+
operation_options: Mapping[str, Any] | None = None,
|
|
596
|
+
) -> Sequence[Any]:
|
|
597
|
+
_check_options(operation_options)
|
|
598
|
+
columns = self._columns(include_vectors)
|
|
599
|
+
selected = ", ".join(f"t.{_identifier(name)}" for name in columns)
|
|
600
|
+
statement = f"SELECT {selected} FROM {self._table} AS t WHERE " # ruff: ignore[hardcoded-sql-expression] # nosec B608
|
|
601
|
+
adapted: list[Any] = []
|
|
602
|
+
if keys is not None:
|
|
603
|
+
if order_by:
|
|
604
|
+
raise ValueError("order_by applies only to filtered retrieval, not key lookup.")
|
|
605
|
+
adapted = [_prepare_value(self.definition.key_field, key) for key in keys]
|
|
606
|
+
if not adapted:
|
|
607
|
+
return []
|
|
608
|
+
placeholders = ", ".join("?" for _ in adapted)
|
|
609
|
+
statement += f"t.{_identifier(self.definition.key_field_storage_name)} IN ({placeholders})"
|
|
610
|
+
params: list[Any] = adapted
|
|
611
|
+
else:
|
|
612
|
+
predicate, params = _FilterCompiler(self.definition).compile(filter)
|
|
613
|
+
statement += f"{predicate} ORDER BY {self._order_by(order_by)} LIMIT ? OFFSET ?"
|
|
614
|
+
params.extend((top, skip))
|
|
615
|
+
|
|
616
|
+
def read(connection: duckdb.DuckDBPyConnection) -> list[dict[str, Any]]:
|
|
617
|
+
rows = connection.execute(statement, params).fetchall()
|
|
618
|
+
return [dict(zip(columns, row, strict=True)) for row in rows]
|
|
619
|
+
|
|
620
|
+
records = await self._client.run(read)
|
|
621
|
+
if keys is not None:
|
|
622
|
+
key_name = self.definition.key_field_storage_name
|
|
623
|
+
by_key = {record[key_name]: record for record in records}
|
|
624
|
+
return [by_key[key] for key in adapted if key in by_key]
|
|
625
|
+
return records
|
|
626
|
+
|
|
627
|
+
async def _inner_delete(self, keys: Sequence[KeyT], *, operation_options: Mapping[str, Any] | None = None) -> None:
|
|
628
|
+
_check_options(operation_options)
|
|
629
|
+
adapted = [_prepare_value(self.definition.key_field, key) for key in keys]
|
|
630
|
+
if adapted:
|
|
631
|
+
statement = (
|
|
632
|
+
f"DELETE FROM {self._table} WHERE {_identifier(self.definition.key_field_storage_name)} " # ruff: ignore[hardcoded-sql-expression] # nosec B608
|
|
633
|
+
f"IN ({', '.join('?' for _ in adapted)})"
|
|
634
|
+
)
|
|
635
|
+
await self._client.run(lambda connection: _execute_statement(connection, statement, adapted))
|
|
636
|
+
|
|
637
|
+
async def _inner_search(
|
|
638
|
+
self,
|
|
639
|
+
*,
|
|
640
|
+
search_type: SearchType,
|
|
641
|
+
filter: FilterExpression | None = None,
|
|
642
|
+
values: Any | None = None,
|
|
643
|
+
vector: Vector | None = None,
|
|
644
|
+
top: int = 3,
|
|
645
|
+
skip: int = 0,
|
|
646
|
+
include_vectors: bool = False,
|
|
647
|
+
vector_property_name: str | None = None,
|
|
648
|
+
additional_property_name: str | None = None,
|
|
649
|
+
score_threshold: float | None = None,
|
|
650
|
+
operation_options: Mapping[str, Any] | None = None,
|
|
651
|
+
) -> SearchResults[Any]:
|
|
652
|
+
_check_options(operation_options)
|
|
653
|
+
if search_type != "vector" or additional_property_name is not None:
|
|
654
|
+
raise NotImplementedError("DuckDB supports dense vector search only, not keyword-hybrid search.")
|
|
655
|
+
if vector is None:
|
|
656
|
+
raise NotImplementedError("DuckDB cannot vectorize values; supply a vector or embedding generator.")
|
|
657
|
+
field = self.definition.try_get_vector_field(vector_property_name)
|
|
658
|
+
if field is None:
|
|
659
|
+
raise ValueError("Select a vector_property_name from the collection definition.")
|
|
660
|
+
query_vector = _prepare_vector(field, vector)
|
|
661
|
+
metric, descending = _METRICS[field.distance_function or "DEFAULT"]
|
|
662
|
+
column = f"t.{_identifier(field.storage_name or field.name)}"
|
|
663
|
+
distance = f"{metric}({column}, CAST(? AS {_column_type(field)}))"
|
|
664
|
+
if metric in ("array_cosine_distance", "array_cosine_similarity"):
|
|
665
|
+
distance = f"CASE WHEN array_inner_product({column}, {column}) = 0 THEN NULL ELSE {distance} END"
|
|
666
|
+
predicate, filter_params = _FilterCompiler(self.definition).compile(filter)
|
|
667
|
+
columns = self._columns(include_vectors)
|
|
668
|
+
selected = ", ".join(f"t.{_identifier(name)}" for name in columns)
|
|
669
|
+
comparison = ">=" if descending else "<="
|
|
670
|
+
order = "DESC" if descending else "ASC"
|
|
671
|
+
statement = (
|
|
672
|
+
f"SELECT {selected}, scores.value FROM {self._table} AS t " # ruff: ignore[hardcoded-sql-expression] # nosec B608
|
|
673
|
+
f"CROSS JOIN LATERAL (SELECT {distance} AS value) AS scores "
|
|
674
|
+
f"WHERE {column} IS NOT NULL AND ({predicate}) AND isfinite(scores.value)"
|
|
675
|
+
)
|
|
676
|
+
parameters: list[Any] = [query_vector, *filter_params]
|
|
677
|
+
if score_threshold is not None:
|
|
678
|
+
if type(score_threshold) not in (float, int) or not math.isfinite(score_threshold):
|
|
679
|
+
raise ValueError("score_threshold must be a finite number.")
|
|
680
|
+
statement += f" AND scores.value {comparison} ?"
|
|
681
|
+
parameters.append(score_threshold)
|
|
682
|
+
statement += (
|
|
683
|
+
f" ORDER BY scores.value {order}, t.{_identifier(self.definition.key_field_storage_name)} ASC "
|
|
684
|
+
"LIMIT ? OFFSET ?"
|
|
685
|
+
)
|
|
686
|
+
parameters.extend((top, skip))
|
|
687
|
+
|
|
688
|
+
def search(connection: duckdb.DuckDBPyConnection) -> list[dict[str, Any]]:
|
|
689
|
+
rows = connection.execute(statement, parameters).fetchall()
|
|
690
|
+
return [{"record": dict(zip(columns, row[:-1], strict=True)), "score": float(row[-1])} for row in rows]
|
|
691
|
+
|
|
692
|
+
return SearchResults(
|
|
693
|
+
await self._client.run(search),
|
|
694
|
+
metadata={"distance_function": field.distance_function or "DEFAULT", "approximate": False},
|
|
695
|
+
)
|
|
696
|
+
|
|
697
|
+
def _get_record_from_result(self, result: Any) -> Any:
|
|
698
|
+
return result["record"]
|
|
699
|
+
|
|
700
|
+
def _get_score_from_result(self, result: Any) -> float | None:
|
|
701
|
+
return float(result["score"])
|
|
702
|
+
|
|
703
|
+
|
|
704
|
+
class DuckDBStore(BaseVectorStore):
|
|
705
|
+
"""Create DuckDB collections sharing one serial, worker-owned connection."""
|
|
706
|
+
|
|
707
|
+
def __init__(
|
|
708
|
+
self,
|
|
709
|
+
*,
|
|
710
|
+
connection_string: str | SecretString | None = None,
|
|
711
|
+
client: duckdb.DuckDBPyConnection | None = None,
|
|
712
|
+
config: Mapping[str, str | SecretString] | None = None,
|
|
713
|
+
embedding_generator: EmbeddingClient | None = None,
|
|
714
|
+
env_file_path: str | None = None,
|
|
715
|
+
env_file_encoding: str | None = None,
|
|
716
|
+
) -> None:
|
|
717
|
+
"""Initialize a persistent local store or a DuckDB-client-supported service.
|
|
718
|
+
|
|
719
|
+
Args:
|
|
720
|
+
connection_string: On-disk filename or DuckDB URI; defaults to ``agent-framework.duckdb``.
|
|
721
|
+
client: Borrowed DuckDB connection; cannot be combined with other connection settings.
|
|
722
|
+
config: Additional DuckDB connection settings. Wrap credential values in ``SecretString``.
|
|
723
|
+
embedding_generator: Default local embedding client for collections.
|
|
724
|
+
env_file_path: Optional selected .env file for ``DUCKDB_CONNECTION_STRING``.
|
|
725
|
+
env_file_encoding: Encoding of the selected .env file.
|
|
726
|
+
"""
|
|
727
|
+
super().__init__(embedding_generator=embedding_generator, managed_client=client is None)
|
|
728
|
+
self._client = _create_client(
|
|
729
|
+
connection_string,
|
|
730
|
+
client=client,
|
|
731
|
+
config=config,
|
|
732
|
+
env_file_path=env_file_path,
|
|
733
|
+
env_file_encoding=env_file_encoding,
|
|
734
|
+
)
|
|
735
|
+
|
|
736
|
+
def get_collection(
|
|
737
|
+
self,
|
|
738
|
+
record_type: type[ModelT],
|
|
739
|
+
*,
|
|
740
|
+
definition: VectorStoreCollectionDefinition | None = None,
|
|
741
|
+
collection_name: str | None = None,
|
|
742
|
+
embedding_generator: EmbeddingClient | None = None,
|
|
743
|
+
) -> DuckDBCollection[Any, ModelT]:
|
|
744
|
+
"""Create a collection that borrows this store's connection and lifecycle."""
|
|
745
|
+
return DuckDBCollection(
|
|
746
|
+
record_type,
|
|
747
|
+
definition=definition,
|
|
748
|
+
collection_name=collection_name,
|
|
749
|
+
embedding_generator=embedding_generator if embedding_generator is not None else self.embedding_generator,
|
|
750
|
+
_shared_client=self._client,
|
|
751
|
+
)
|
|
752
|
+
|
|
753
|
+
async def list_collection_names(self, *, operation_options: Mapping[str, Any] | None = None) -> Sequence[str]:
|
|
754
|
+
"""List base tables in the active database and schema."""
|
|
755
|
+
_check_options(operation_options)
|
|
756
|
+
return await self._client.run(
|
|
757
|
+
lambda connection: [str(row[0]) for row in connection.execute(_LIST_TABLES).fetchall()]
|
|
758
|
+
)
|
|
759
|
+
|
|
760
|
+
async def collection_exists(
|
|
761
|
+
self,
|
|
762
|
+
collection_name: str,
|
|
763
|
+
*,
|
|
764
|
+
operation_options: Mapping[str, Any] | None = None,
|
|
765
|
+
) -> bool:
|
|
766
|
+
"""Check for a table using DuckDB's case-insensitive identifier semantics."""
|
|
767
|
+
_check_options(operation_options)
|
|
768
|
+
return await self._client.run(
|
|
769
|
+
lambda connection: connection.execute(_TABLE_EXISTS, [collection_name]).fetchone() is not None
|
|
770
|
+
)
|
|
771
|
+
|
|
772
|
+
async def _inner_ensure_collection_deleted(
|
|
773
|
+
self,
|
|
774
|
+
collection_name: str,
|
|
775
|
+
*,
|
|
776
|
+
operation_options: Mapping[str, Any] | None = None,
|
|
777
|
+
) -> None:
|
|
778
|
+
_check_options(operation_options)
|
|
779
|
+
table = _identifier(collection_name)
|
|
780
|
+
await self._client.run(lambda connection: _execute_statement(connection, f"DROP TABLE IF EXISTS {table}"))
|
|
781
|
+
|
|
782
|
+
async def __aexit__(self, exc_type: Any, exc_value: Any, traceback: Any) -> None:
|
|
783
|
+
"""Close owned connections and wait for the database file to be released."""
|
|
784
|
+
await self.aclose()
|
|
785
|
+
|
|
786
|
+
async def aclose(self) -> None:
|
|
787
|
+
"""Drain queued work and close only a connector-created connection."""
|
|
788
|
+
await self._client.aclose()
|
|
File without changes
|
|
@@ -0,0 +1,61 @@
|
|
|
1
|
+
[project]
|
|
2
|
+
name = "agent-framework-duckdb"
|
|
3
|
+
description = "DuckDB vector stores for Microsoft Agent Framework."
|
|
4
|
+
authors = [{ name = "Microsoft", email = "af-support@microsoft.com" }]
|
|
5
|
+
readme = "README.md"
|
|
6
|
+
requires-python = ">=3.10"
|
|
7
|
+
version = "1.0.0a261002"
|
|
8
|
+
license-files = ["LICENSE"]
|
|
9
|
+
urls.homepage = "https://aka.ms/agent-framework"
|
|
10
|
+
urls.source = "https://github.com/microsoft/agent-framework/tree/main/python"
|
|
11
|
+
urls.issues = "https://github.com/microsoft/agent-framework/issues"
|
|
12
|
+
classifiers = [
|
|
13
|
+
"License :: OSI Approved :: MIT License",
|
|
14
|
+
"Development Status :: 3 - Alpha",
|
|
15
|
+
"Intended Audience :: Developers",
|
|
16
|
+
"Programming Language :: Python :: 3",
|
|
17
|
+
"Programming Language :: Python :: 3.10",
|
|
18
|
+
"Programming Language :: Python :: 3.11",
|
|
19
|
+
"Programming Language :: Python :: 3.12",
|
|
20
|
+
"Programming Language :: Python :: 3.13",
|
|
21
|
+
"Programming Language :: Python :: 3.14",
|
|
22
|
+
"Typing :: Typed",
|
|
23
|
+
]
|
|
24
|
+
dependencies = [
|
|
25
|
+
"agent-framework-core>=1.19.0,<2",
|
|
26
|
+
"duckdb>=1.4.1,<1.6",
|
|
27
|
+
"pytz>=2024.1,<2027",
|
|
28
|
+
]
|
|
29
|
+
|
|
30
|
+
[tool.ruff]
|
|
31
|
+
extend = "../../pyproject.toml"
|
|
32
|
+
|
|
33
|
+
[tool.ruff.lint.per-file-ignores]
|
|
34
|
+
"samples/**" = ["D", "INP", "TD", "ERA001", "RUF", "S", "T201", "CPY"]
|
|
35
|
+
"tests/**" = ["D", "INP", "TD", "ERA001", "RUF", "S"]
|
|
36
|
+
|
|
37
|
+
[tool.pyright]
|
|
38
|
+
extends = "../../pyproject.toml"
|
|
39
|
+
include = ["agent_framework_duckdb"]
|
|
40
|
+
|
|
41
|
+
[tool.pytest.ini_options]
|
|
42
|
+
testpaths = ["tests"]
|
|
43
|
+
addopts = "-ra -q -r fEX"
|
|
44
|
+
asyncio_mode = "auto"
|
|
45
|
+
asyncio_default_fixture_loop_scope = "function"
|
|
46
|
+
timeout = 60
|
|
47
|
+
|
|
48
|
+
[tool.coverage.run]
|
|
49
|
+
omit = ["**/__init__.py"]
|
|
50
|
+
|
|
51
|
+
[tool.poe]
|
|
52
|
+
executor.type = "uv"
|
|
53
|
+
include = "../../shared_tasks.toml"
|
|
54
|
+
|
|
55
|
+
[tool.poe.tasks.test]
|
|
56
|
+
help = "Run DuckDB connector unit tests."
|
|
57
|
+
cmd = 'pytest -m "not integration" --cov=agent_framework_duckdb --cov-report=term-missing:skip-covered tests'
|
|
58
|
+
|
|
59
|
+
[build-system]
|
|
60
|
+
requires = ["flit-core>=3.11,<4.0"]
|
|
61
|
+
build-backend = "flit_core.buildapi"
|