hybriddb 0.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- hybriddb-0.3.0/.gitignore +11 -0
- hybriddb-0.3.0/LICENSE +21 -0
- hybriddb-0.3.0/PKG-INFO +231 -0
- hybriddb-0.3.0/README.md +191 -0
- hybriddb-0.3.0/docs/API.md +272 -0
- hybriddb-0.3.0/docs/BENCHMARKS.md +93 -0
- hybriddb-0.3.0/docs/RELEASE.md +117 -0
- hybriddb-0.3.0/docs/spec.md +1368 -0
- hybriddb-0.3.0/docs/superpowers/plans/2026-05-30-hybriddb-architecture-split.md +463 -0
- hybriddb-0.3.0/docs/superpowers/specs/2026-05-21-hybriddb-benchmark-design.md +248 -0
- hybriddb-0.3.0/docs/superpowers/specs/2026-05-30-hybriddb-architecture-split-design.md +531 -0
- hybriddb-0.3.0/hybriddb/__init__.py +31 -0
- hybriddb-0.3.0/hybriddb/analytics.py +236 -0
- hybriddb-0.3.0/hybriddb/async_api.py +68 -0
- hybriddb-0.3.0/hybriddb/crud.py +275 -0
- hybriddb-0.3.0/hybriddb/db.py +259 -0
- hybriddb-0.3.0/hybriddb/embedding.py +60 -0
- hybriddb-0.3.0/hybriddb/facades.py +49 -0
- hybriddb-0.3.0/hybriddb/graph.py +641 -0
- hybriddb-0.3.0/hybriddb/journal.py +192 -0
- hybriddb-0.3.0/hybriddb/maintenance.py +277 -0
- hybriddb-0.3.0/hybriddb/schema.py +327 -0
- hybriddb-0.3.0/hybriddb/search.py +259 -0
- hybriddb-0.3.0/hybriddb/types.py +38 -0
- hybriddb-0.3.0/hybriddb/utils.py +78 -0
- hybriddb-0.3.0/pyproject.toml +68 -0
- hybriddb-0.3.0/scripts/compare_results.py +134 -0
- hybriddb-0.3.0/scripts/run_benchmarks.sh +41 -0
- hybriddb-0.3.0/tests/__init__.py +0 -0
- hybriddb-0.3.0/tests/benchmarks/__init__.py +0 -0
- hybriddb-0.3.0/tests/benchmarks/conftest.py +76 -0
- hybriddb-0.3.0/tests/benchmarks/helpers.py +168 -0
- hybriddb-0.3.0/tests/benchmarks/test_analytics.py +74 -0
- hybriddb-0.3.0/tests/benchmarks/test_concurrent.py +115 -0
- hybriddb-0.3.0/tests/benchmarks/test_graph.py +81 -0
- hybriddb-0.3.0/tests/benchmarks/test_search.py +105 -0
- hybriddb-0.3.0/tests/benchmarks/test_storage.py +85 -0
- hybriddb-0.3.0/tests/test_db.py +884 -0
- hybriddb-0.3.0/uv.lock +2786 -0
hybriddb-0.3.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Eddy Vinck
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
hybriddb-0.3.0/PKG-INFO
ADDED
|
@@ -0,0 +1,231 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: hybriddb
|
|
3
|
+
Version: 0.3.0
|
|
4
|
+
Summary: Embedded, local, open-source hybrid search for AI agents — SQLite + FTS5 + ChromaDB with self-healing journal
|
|
5
|
+
Author: Eddy Xu
|
|
6
|
+
License: MIT
|
|
7
|
+
License-File: LICENSE
|
|
8
|
+
Keywords: chromadb,embeddings,fts5,hybrid-search,keyword-search,sqlite,vector-search
|
|
9
|
+
Classifier: Development Status :: 3 - Alpha
|
|
10
|
+
Classifier: Intended Audience :: Developers
|
|
11
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
12
|
+
Classifier: Programming Language :: Python :: 3
|
|
13
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
16
|
+
Classifier: Topic :: Database
|
|
17
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
18
|
+
Requires-Python: >=3.11
|
|
19
|
+
Requires-Dist: chromadb>=0.5.0
|
|
20
|
+
Provides-Extra: all
|
|
21
|
+
Requires-Dist: duckdb>=1.0.0; extra == 'all'
|
|
22
|
+
Requires-Dist: networkx>=3.0; extra == 'all'
|
|
23
|
+
Provides-Extra: analytics
|
|
24
|
+
Requires-Dist: duckdb>=1.0.0; extra == 'analytics'
|
|
25
|
+
Provides-Extra: benchmark
|
|
26
|
+
Requires-Dist: numpy>=1.24.0; extra == 'benchmark'
|
|
27
|
+
Requires-Dist: pytest-benchmark>=4.0.0; extra == 'benchmark'
|
|
28
|
+
Requires-Dist: pytest-timeout>=2.3.0; extra == 'benchmark'
|
|
29
|
+
Requires-Dist: sentence-transformers>=3.0.0; extra == 'benchmark'
|
|
30
|
+
Provides-Extra: dev
|
|
31
|
+
Requires-Dist: duckdb>=1.0.0; extra == 'dev'
|
|
32
|
+
Requires-Dist: networkx>=3.0; extra == 'dev'
|
|
33
|
+
Requires-Dist: pytest-asyncio>=0.24.0; extra == 'dev'
|
|
34
|
+
Requires-Dist: pytest>=8.0.0; extra == 'dev'
|
|
35
|
+
Provides-Extra: graph
|
|
36
|
+
Requires-Dist: networkx>=3.0; extra == 'graph'
|
|
37
|
+
Provides-Extra: sentence-transformers
|
|
38
|
+
Requires-Dist: sentence-transformers>=3.0.0; extra == 'sentence-transformers'
|
|
39
|
+
Description-Content-Type: text/markdown
|
|
40
|
+
|
|
41
|
+
# HybridDB
|
|
42
|
+
|
|
43
|
+
> **Purposefully built for AI Agents.** HybridDB gives agents persistent, searchable memory — every conversation turn is indexed and retrievable via keyword, vector, or hybrid search. Used in production by the [Executive Assistant](https://github.com/open-assistants-lab) agent system.
|
|
44
|
+
|
|
45
|
+
> **Embedded. Local. Open source.** No cloud APIs, no vector DB services, no internet connection required. Runs entirely on-device with SQLite + ChromaDB + your choice of local embedding model. Ships as a single Python package with zero external infrastructure dependencies.
|
|
46
|
+
|
|
47
|
+
**SQLite + FTS5 + ChromaDB with a self-healing journal.** One Python class that gives you keyword search, vector search, SQL queries, and structured filtering — all kept in sync automatically.
|
|
48
|
+
|
|
49
|
+
```python
|
|
50
|
+
from hybriddb import HybridDB, LONGTEXT, TEXT
|
|
51
|
+
|
|
52
|
+
db = HybridDB("./my_data")
|
|
53
|
+
db.create_table("docs", {"title": TEXT, "body": LONGTEXT})
|
|
54
|
+
|
|
55
|
+
db.insert("docs", {"title": "Getting Started", "body": "A guide to using HybridDB..."})
|
|
56
|
+
db.insert("docs", {"title": "API Reference", "body": "Full API documentation..."})
|
|
57
|
+
|
|
58
|
+
# Search every text column
|
|
59
|
+
db.search("docs", "getting started")
|
|
60
|
+
|
|
61
|
+
# Search one column
|
|
62
|
+
db.search("docs", "body", "how do I begin", mode="hybrid")
|
|
63
|
+
|
|
64
|
+
# Structured query with parameters
|
|
65
|
+
db.query("docs", where="title LIKE ?", params=("%start%",))
|
|
66
|
+
```
|
|
67
|
+
|
|
68
|
+
## Why HybridDB?
|
|
69
|
+
|
|
70
|
+
Every serious project that needs **both** keyword and semantic search ends up wiring SQLite + FTS5 + ChromaDB together. You handle schema creation, FTS5 triggers, ChromaDB collection management, keeping them in sync, recovering from crashes, rebuilding indexes...
|
|
71
|
+
|
|
72
|
+
HybridDB does all of that once, done right.
|
|
73
|
+
|
|
74
|
+
| Feature | Status |
|
|
75
|
+
|---------|--------|
|
|
76
|
+
| SQL CRUD (insert, update, delete, get, query) | ✅ |
|
|
77
|
+
| FTS5 keyword search with BM25 scoring | ✅ |
|
|
78
|
+
| ChromaDB semantic/vector search with HNSW | ✅ |
|
|
79
|
+
| Hybrid search (RRF fusion of keyword + semantic) | ✅ |
|
|
80
|
+
| Recency-weighted scoring | ✅ |
|
|
81
|
+
| Schema management (create, add/drop/rename columns) | ✅ |
|
|
82
|
+
| Self-healing journal (crash recovery) | ✅ |
|
|
83
|
+
| Sync + async APIs | ✅ |
|
|
84
|
+
| No external API dependencies (works offline) | ✅ |
|
|
85
|
+
| Embedding model pluggable (sentence-transformers, OpenAI, custom) | ✅ |
|
|
86
|
+
|
|
87
|
+
## Documentation
|
|
88
|
+
|
|
89
|
+
- [API reference](docs/API.md) — stable public methods, sync/async examples, graph and OLAP facades
|
|
90
|
+
- [Benchmarks](docs/BENCHMARKS.md) — smoke vs full benchmark commands and expected runtime behavior
|
|
91
|
+
- [Release guide](docs/RELEASE.md) — local build, wheel smoke test, TestPyPI/PyPI publishing
|
|
92
|
+
|
|
93
|
+
## Installation
|
|
94
|
+
|
|
95
|
+
```bash
|
|
96
|
+
pip install hybriddb
|
|
97
|
+
```
|
|
98
|
+
|
|
99
|
+
HybridDB uses ChromaDB's bundled local MiniLM embedding by default. No API key required.
|
|
100
|
+
|
|
101
|
+
## Core Concepts
|
|
102
|
+
|
|
103
|
+
### Column Types
|
|
104
|
+
|
|
105
|
+
HybridDB maps Python-friendly types to SQLite storage and automatically sets up the right search indexes:
|
|
106
|
+
|
|
107
|
+
| Type | SQLite | FTS5 | ChromaDB | Use for |
|
|
108
|
+
|------|--------|------|----------|---------|
|
|
109
|
+
| `TEXT` | TEXT | ✅ | — | Names, titles, short strings |
|
|
110
|
+
| `LONGTEXT` | TEXT | ✅ | ✅ | Documents, messages, memory content |
|
|
111
|
+
| `INTEGER` | INTEGER | — | — | Counts, ages, IDs |
|
|
112
|
+
| `REAL` | REAL | — | — | Prices, scores, confidence values |
|
|
113
|
+
| `BOOLEAN` | INTEGER | — | — | Flags, status indicators |
|
|
114
|
+
| `JSON` | TEXT | — | — | Tags, metadata, structured data |
|
|
115
|
+
|
|
116
|
+
**TEXT** columns get automated FTS5 keyword search.
|
|
117
|
+
**LONGTEXT** columns get FTS5 + ChromaDB semantic search.
|
|
118
|
+
|
|
119
|
+
### Search Modes
|
|
120
|
+
|
|
121
|
+
```python
|
|
122
|
+
from hybriddb import HYBRID, LONGTEXT, TEXT, Column, SearchMode
|
|
123
|
+
|
|
124
|
+
db.create_table("docs", {"title": Column(TEXT), "body": LONGTEXT})
|
|
125
|
+
|
|
126
|
+
# Keyword only — fast, exact, great for names and titles
|
|
127
|
+
db.search("contacts", "name", "Alice", mode="keyword")
|
|
128
|
+
|
|
129
|
+
# Semantic only — finds "9am standup" when searching for "morning meetings"
|
|
130
|
+
db.search("memories", "content", "team rituals", mode=SearchMode.SEMANTIC)
|
|
131
|
+
|
|
132
|
+
# Hybrid — best of both, RRF fusion, the default
|
|
133
|
+
db.search("docs", "body", "getting started guide", mode=SearchMode.HYBRID)
|
|
134
|
+
db.search("docs", "body", "getting started guide", mode=HYBRID)
|
|
135
|
+
|
|
136
|
+
# Search across ALL text columns at once
|
|
137
|
+
db.search("contacts", "engineering manager")
|
|
138
|
+
db.search_columns("contacts", "engineering manager")
|
|
139
|
+
```
|
|
140
|
+
|
|
141
|
+
### Async API
|
|
142
|
+
|
|
143
|
+
All core operations have async wrappers that run blocking SQLite/ChromaDB work in a worker thread:
|
|
144
|
+
|
|
145
|
+
```python
|
|
146
|
+
await db.acreate_table("messages", {"content": LONGTEXT})
|
|
147
|
+
await db.ainsert("messages", {"content": "async-safe memory"})
|
|
148
|
+
results = await db.asearch("messages", "content", "memory")
|
|
149
|
+
```
|
|
150
|
+
|
|
151
|
+
### Public Cursor
|
|
152
|
+
|
|
153
|
+
For small custom SQL reads or migrations, use the public cursor context manager:
|
|
154
|
+
|
|
155
|
+
```python
|
|
156
|
+
with db.cursor() as cur:
|
|
157
|
+
cur.execute("SELECT COUNT(*) FROM messages")
|
|
158
|
+
count = cur.fetchone()[0]
|
|
159
|
+
```
|
|
160
|
+
|
|
161
|
+
### Namespaced Advanced APIs
|
|
162
|
+
|
|
163
|
+
Graph and OLAP helpers remain available on `HybridDB`, with namespaced facades for discovery:
|
|
164
|
+
|
|
165
|
+
```python
|
|
166
|
+
node_id = db.graph.add_node("Alice", type="person")
|
|
167
|
+
rows = db.olap.query("SELECT COUNT(*) AS total FROM messages")
|
|
168
|
+
```
|
|
169
|
+
|
|
170
|
+
### Recency Scoring
|
|
171
|
+
|
|
172
|
+
Boost recent content over older content:
|
|
173
|
+
|
|
174
|
+
```python
|
|
175
|
+
results = db.search(
|
|
176
|
+
"messages", "content", "project update",
|
|
177
|
+
recency_weight=0.3, # 30% weight to recency
|
|
178
|
+
recency_column="timestamp"
|
|
179
|
+
)
|
|
180
|
+
```
|
|
181
|
+
|
|
182
|
+
### Self-Healing Journal
|
|
183
|
+
|
|
184
|
+
All ChromaDB mutations (adds, updates, deletes) are journaled in SQLite. On insert with `sync=True` (default), the journal is processed immediately. On `sync=False`, journal entries are deferred:
|
|
185
|
+
|
|
186
|
+
```python
|
|
187
|
+
# Batch insert — defer ChromaDB sync for speed
|
|
188
|
+
db.insert_batch("contacts", big_list_of_rows, sync=False)
|
|
189
|
+
db.process_journal() # Sync everything at once
|
|
190
|
+
```
|
|
191
|
+
|
|
192
|
+
If your process crashes mid-write, the journal replays pending entries on next startup. No ghosts, no drift.
|
|
193
|
+
|
|
194
|
+
### Health & Maintenance
|
|
195
|
+
|
|
196
|
+
```python
|
|
197
|
+
# Check if SQLite and ChromaDB are in sync
|
|
198
|
+
health = db.health("contacts")
|
|
199
|
+
# {"sqlite_rows": 5000, "chroma_docs": {"contacts_bio": 5000}, "status": "ok"}
|
|
200
|
+
|
|
201
|
+
# Reconcile: delete ghosts, add missing docs
|
|
202
|
+
result = db.reconcile("contacts")
|
|
203
|
+
# {"ghosts_deleted": 0, "missing_added": 3, "metadata_updated": 0}
|
|
204
|
+
```
|
|
205
|
+
|
|
206
|
+
## Custom Embedding Models
|
|
207
|
+
|
|
208
|
+
By default, HybridDB uses ChromaDB's bundled local MiniLM embedding. Plug in any embedding function if you want a specific model or provider:
|
|
209
|
+
|
|
210
|
+
```python
|
|
211
|
+
from sentence_transformers import SentenceTransformer
|
|
212
|
+
|
|
213
|
+
model = SentenceTransformer("all-MiniLM-L6-v2")
|
|
214
|
+
db = HybridDB("./data", embedding_fn=lambda text: model.encode(text).tolist())
|
|
215
|
+
```
|
|
216
|
+
|
|
217
|
+
Works with any embedding provider — OpenAI, Cohere, Hugging Face, local models.
|
|
218
|
+
|
|
219
|
+
## License
|
|
220
|
+
|
|
221
|
+
MIT — see [LICENSE](LICENSE).
|
|
222
|
+
|
|
223
|
+
## Author
|
|
224
|
+
|
|
225
|
+
Eddy Xu
|
|
226
|
+
|
|
227
|
+
Inspired by [claude-mem](https://github.com/thedotmack/claude-mem) by [Matt Mack](https://github.com/thedotmack).
|
|
228
|
+
|
|
229
|
+
## Status
|
|
230
|
+
|
|
231
|
+
Alpha — actively developed, API may evolve. Core CRUD and search are stable with full test coverage (35+ tests). Currently used in production in the Executive Assistant agent system.
|
hybriddb-0.3.0/README.md
ADDED
|
@@ -0,0 +1,191 @@
|
|
|
1
|
+
# HybridDB
|
|
2
|
+
|
|
3
|
+
> **Purposefully built for AI Agents.** HybridDB gives agents persistent, searchable memory — every conversation turn is indexed and retrievable via keyword, vector, or hybrid search. Used in production by the [Executive Assistant](https://github.com/open-assistants-lab) agent system.
|
|
4
|
+
|
|
5
|
+
> **Embedded. Local. Open source.** No cloud APIs, no vector DB services, no internet connection required. Runs entirely on-device with SQLite + ChromaDB + your choice of local embedding model. Ships as a single Python package with zero external infrastructure dependencies.
|
|
6
|
+
|
|
7
|
+
**SQLite + FTS5 + ChromaDB with a self-healing journal.** One Python class that gives you keyword search, vector search, SQL queries, and structured filtering — all kept in sync automatically.
|
|
8
|
+
|
|
9
|
+
```python
|
|
10
|
+
from hybriddb import HybridDB, LONGTEXT, TEXT
|
|
11
|
+
|
|
12
|
+
db = HybridDB("./my_data")
|
|
13
|
+
db.create_table("docs", {"title": TEXT, "body": LONGTEXT})
|
|
14
|
+
|
|
15
|
+
db.insert("docs", {"title": "Getting Started", "body": "A guide to using HybridDB..."})
|
|
16
|
+
db.insert("docs", {"title": "API Reference", "body": "Full API documentation..."})
|
|
17
|
+
|
|
18
|
+
# Search every text column
|
|
19
|
+
db.search("docs", "getting started")
|
|
20
|
+
|
|
21
|
+
# Search one column
|
|
22
|
+
db.search("docs", "body", "how do I begin", mode="hybrid")
|
|
23
|
+
|
|
24
|
+
# Structured query with parameters
|
|
25
|
+
db.query("docs", where="title LIKE ?", params=("%start%",))
|
|
26
|
+
```
|
|
27
|
+
|
|
28
|
+
## Why HybridDB?
|
|
29
|
+
|
|
30
|
+
Every serious project that needs **both** keyword and semantic search ends up wiring SQLite + FTS5 + ChromaDB together. You handle schema creation, FTS5 triggers, ChromaDB collection management, keeping them in sync, recovering from crashes, rebuilding indexes...
|
|
31
|
+
|
|
32
|
+
HybridDB does all of that once, done right.
|
|
33
|
+
|
|
34
|
+
| Feature | Status |
|
|
35
|
+
|---------|--------|
|
|
36
|
+
| SQL CRUD (insert, update, delete, get, query) | ✅ |
|
|
37
|
+
| FTS5 keyword search with BM25 scoring | ✅ |
|
|
38
|
+
| ChromaDB semantic/vector search with HNSW | ✅ |
|
|
39
|
+
| Hybrid search (RRF fusion of keyword + semantic) | ✅ |
|
|
40
|
+
| Recency-weighted scoring | ✅ |
|
|
41
|
+
| Schema management (create, add/drop/rename columns) | ✅ |
|
|
42
|
+
| Self-healing journal (crash recovery) | ✅ |
|
|
43
|
+
| Sync + async APIs | ✅ |
|
|
44
|
+
| No external API dependencies (works offline) | ✅ |
|
|
45
|
+
| Embedding model pluggable (sentence-transformers, OpenAI, custom) | ✅ |
|
|
46
|
+
|
|
47
|
+
## Documentation
|
|
48
|
+
|
|
49
|
+
- [API reference](docs/API.md) — stable public methods, sync/async examples, graph and OLAP facades
|
|
50
|
+
- [Benchmarks](docs/BENCHMARKS.md) — smoke vs full benchmark commands and expected runtime behavior
|
|
51
|
+
- [Release guide](docs/RELEASE.md) — local build, wheel smoke test, TestPyPI/PyPI publishing
|
|
52
|
+
|
|
53
|
+
## Installation
|
|
54
|
+
|
|
55
|
+
```bash
|
|
56
|
+
pip install hybriddb
|
|
57
|
+
```
|
|
58
|
+
|
|
59
|
+
HybridDB uses ChromaDB's bundled local MiniLM embedding by default. No API key required.
|
|
60
|
+
|
|
61
|
+
## Core Concepts
|
|
62
|
+
|
|
63
|
+
### Column Types
|
|
64
|
+
|
|
65
|
+
HybridDB maps Python-friendly types to SQLite storage and automatically sets up the right search indexes:
|
|
66
|
+
|
|
67
|
+
| Type | SQLite | FTS5 | ChromaDB | Use for |
|
|
68
|
+
|------|--------|------|----------|---------|
|
|
69
|
+
| `TEXT` | TEXT | ✅ | — | Names, titles, short strings |
|
|
70
|
+
| `LONGTEXT` | TEXT | ✅ | ✅ | Documents, messages, memory content |
|
|
71
|
+
| `INTEGER` | INTEGER | — | — | Counts, ages, IDs |
|
|
72
|
+
| `REAL` | REAL | — | — | Prices, scores, confidence values |
|
|
73
|
+
| `BOOLEAN` | INTEGER | — | — | Flags, status indicators |
|
|
74
|
+
| `JSON` | TEXT | — | — | Tags, metadata, structured data |
|
|
75
|
+
|
|
76
|
+
**TEXT** columns get automated FTS5 keyword search.
|
|
77
|
+
**LONGTEXT** columns get FTS5 + ChromaDB semantic search.
|
|
78
|
+
|
|
79
|
+
### Search Modes
|
|
80
|
+
|
|
81
|
+
```python
|
|
82
|
+
from hybriddb import HYBRID, LONGTEXT, TEXT, Column, SearchMode
|
|
83
|
+
|
|
84
|
+
db.create_table("docs", {"title": Column(TEXT), "body": LONGTEXT})
|
|
85
|
+
|
|
86
|
+
# Keyword only — fast, exact, great for names and titles
|
|
87
|
+
db.search("contacts", "name", "Alice", mode="keyword")
|
|
88
|
+
|
|
89
|
+
# Semantic only — finds "9am standup" when searching for "morning meetings"
|
|
90
|
+
db.search("memories", "content", "team rituals", mode=SearchMode.SEMANTIC)
|
|
91
|
+
|
|
92
|
+
# Hybrid — best of both, RRF fusion, the default
|
|
93
|
+
db.search("docs", "body", "getting started guide", mode=SearchMode.HYBRID)
|
|
94
|
+
db.search("docs", "body", "getting started guide", mode=HYBRID)
|
|
95
|
+
|
|
96
|
+
# Search across ALL text columns at once
|
|
97
|
+
db.search("contacts", "engineering manager")
|
|
98
|
+
db.search_columns("contacts", "engineering manager")
|
|
99
|
+
```
|
|
100
|
+
|
|
101
|
+
### Async API
|
|
102
|
+
|
|
103
|
+
All core operations have async wrappers that run blocking SQLite/ChromaDB work in a worker thread:
|
|
104
|
+
|
|
105
|
+
```python
|
|
106
|
+
await db.acreate_table("messages", {"content": LONGTEXT})
|
|
107
|
+
await db.ainsert("messages", {"content": "async-safe memory"})
|
|
108
|
+
results = await db.asearch("messages", "content", "memory")
|
|
109
|
+
```
|
|
110
|
+
|
|
111
|
+
### Public Cursor
|
|
112
|
+
|
|
113
|
+
For small custom SQL reads or migrations, use the public cursor context manager:
|
|
114
|
+
|
|
115
|
+
```python
|
|
116
|
+
with db.cursor() as cur:
|
|
117
|
+
cur.execute("SELECT COUNT(*) FROM messages")
|
|
118
|
+
count = cur.fetchone()[0]
|
|
119
|
+
```
|
|
120
|
+
|
|
121
|
+
### Namespaced Advanced APIs
|
|
122
|
+
|
|
123
|
+
Graph and OLAP helpers remain available on `HybridDB`, with namespaced facades for discovery:
|
|
124
|
+
|
|
125
|
+
```python
|
|
126
|
+
node_id = db.graph.add_node("Alice", type="person")
|
|
127
|
+
rows = db.olap.query("SELECT COUNT(*) AS total FROM messages")
|
|
128
|
+
```
|
|
129
|
+
|
|
130
|
+
### Recency Scoring
|
|
131
|
+
|
|
132
|
+
Boost recent content over older content:
|
|
133
|
+
|
|
134
|
+
```python
|
|
135
|
+
results = db.search(
|
|
136
|
+
"messages", "content", "project update",
|
|
137
|
+
recency_weight=0.3, # 30% weight to recency
|
|
138
|
+
recency_column="timestamp"
|
|
139
|
+
)
|
|
140
|
+
```
|
|
141
|
+
|
|
142
|
+
### Self-Healing Journal
|
|
143
|
+
|
|
144
|
+
All ChromaDB mutations (adds, updates, deletes) are journaled in SQLite. On insert with `sync=True` (default), the journal is processed immediately. On `sync=False`, journal entries are deferred:
|
|
145
|
+
|
|
146
|
+
```python
|
|
147
|
+
# Batch insert — defer ChromaDB sync for speed
|
|
148
|
+
db.insert_batch("contacts", big_list_of_rows, sync=False)
|
|
149
|
+
db.process_journal() # Sync everything at once
|
|
150
|
+
```
|
|
151
|
+
|
|
152
|
+
If your process crashes mid-write, the journal replays pending entries on next startup. No ghosts, no drift.
|
|
153
|
+
|
|
154
|
+
### Health & Maintenance
|
|
155
|
+
|
|
156
|
+
```python
|
|
157
|
+
# Check if SQLite and ChromaDB are in sync
|
|
158
|
+
health = db.health("contacts")
|
|
159
|
+
# {"sqlite_rows": 5000, "chroma_docs": {"contacts_bio": 5000}, "status": "ok"}
|
|
160
|
+
|
|
161
|
+
# Reconcile: delete ghosts, add missing docs
|
|
162
|
+
result = db.reconcile("contacts")
|
|
163
|
+
# {"ghosts_deleted": 0, "missing_added": 3, "metadata_updated": 0}
|
|
164
|
+
```
|
|
165
|
+
|
|
166
|
+
## Custom Embedding Models
|
|
167
|
+
|
|
168
|
+
By default, HybridDB uses ChromaDB's bundled local MiniLM embedding. Plug in any embedding function if you want a specific model or provider:
|
|
169
|
+
|
|
170
|
+
```python
|
|
171
|
+
from sentence_transformers import SentenceTransformer
|
|
172
|
+
|
|
173
|
+
model = SentenceTransformer("all-MiniLM-L6-v2")
|
|
174
|
+
db = HybridDB("./data", embedding_fn=lambda text: model.encode(text).tolist())
|
|
175
|
+
```
|
|
176
|
+
|
|
177
|
+
Works with any embedding provider — OpenAI, Cohere, Hugging Face, local models.
|
|
178
|
+
|
|
179
|
+
## License
|
|
180
|
+
|
|
181
|
+
MIT — see [LICENSE](LICENSE).
|
|
182
|
+
|
|
183
|
+
## Author
|
|
184
|
+
|
|
185
|
+
Eddy Xu
|
|
186
|
+
|
|
187
|
+
Inspired by [claude-mem](https://github.com/thedotmack/claude-mem) by [Matt Mack](https://github.com/thedotmack).
|
|
188
|
+
|
|
189
|
+
## Status
|
|
190
|
+
|
|
191
|
+
Alpha — actively developed, API may evolve. Core CRUD and search are stable with full test coverage (35+ tests). Currently used in production in the Executive Assistant agent system.
|