hybriddb 0.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- hybriddb/__init__.py +31 -0
- hybriddb/analytics.py +236 -0
- hybriddb/async_api.py +68 -0
- hybriddb/crud.py +275 -0
- hybriddb/db.py +259 -0
- hybriddb/embedding.py +60 -0
- hybriddb/facades.py +49 -0
- hybriddb/graph.py +641 -0
- hybriddb/journal.py +192 -0
- hybriddb/maintenance.py +277 -0
- hybriddb/schema.py +327 -0
- hybriddb/search.py +259 -0
- hybriddb/types.py +38 -0
- hybriddb/utils.py +78 -0
- hybriddb-0.3.0.dist-info/METADATA +231 -0
- hybriddb-0.3.0.dist-info/RECORD +18 -0
- hybriddb-0.3.0.dist-info/WHEEL +4 -0
- hybriddb-0.3.0.dist-info/licenses/LICENSE +21 -0
hybriddb/__init__.py
ADDED
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
"""HybridDB: SQLite + FTS5 + ChromaDB + Graph + DuckDB with self-healing journal.
|
|
2
|
+
|
|
3
|
+
TEXT columns get keyword search. LONGTEXT columns get keyword + semantic search.
|
|
4
|
+
Graph capabilities: SQLite-backed nodes/edges, recursive CTE traversal, NetworkX algorithms.
|
|
5
|
+
Analytics: DuckDB columnar store synced via unified journal for fast OLAP queries.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
__version__ = "0.3.0"
|
|
9
|
+
|
|
10
|
+
from hybriddb.embedding import default_embedding_fn
|
|
11
|
+
from hybriddb.db import HybridDB
|
|
12
|
+
from hybriddb.types import (
|
|
13
|
+
BOOLEAN,
|
|
14
|
+
HYBRID,
|
|
15
|
+
INTEGER,
|
|
16
|
+
JSON,
|
|
17
|
+
KEYWORD,
|
|
18
|
+
LONGTEXT,
|
|
19
|
+
REAL,
|
|
20
|
+
SEMANTIC,
|
|
21
|
+
TEXT,
|
|
22
|
+
Column,
|
|
23
|
+
EmbeddingModelError,
|
|
24
|
+
SearchMode,
|
|
25
|
+
)
|
|
26
|
+
|
|
27
|
+
__all__ = [
|
|
28
|
+
"HybridDB", "SearchMode", "EmbeddingModelError", "Column", "default_embedding_fn",
|
|
29
|
+
"TEXT", "LONGTEXT", "INTEGER", "REAL", "BOOLEAN", "JSON",
|
|
30
|
+
"KEYWORD", "SEMANTIC", "HYBRID",
|
|
31
|
+
]
|
hybriddb/analytics.py
ADDED
|
@@ -0,0 +1,236 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
import logging
|
|
5
|
+
import shutil
|
|
6
|
+
import sqlite3
|
|
7
|
+
import struct
|
|
8
|
+
import tempfile
|
|
9
|
+
import uuid
|
|
10
|
+
from collections import defaultdict
|
|
11
|
+
from datetime import UTC, datetime
|
|
12
|
+
from pathlib import Path
|
|
13
|
+
from typing import Any
|
|
14
|
+
|
|
15
|
+
from hybriddb.embedding import EMBEDDING_DIM
|
|
16
|
+
from hybriddb.types import Column, SearchMode
|
|
17
|
+
from hybriddb.utils import (
|
|
18
|
+
CHROMA_BATCH,
|
|
19
|
+
JOURNAL_CAP,
|
|
20
|
+
RRF_K,
|
|
21
|
+
_CHROMA_INDEX_MAX_ELEMENTS,
|
|
22
|
+
_CHROMA_INDEX_MAX_M0,
|
|
23
|
+
_CHROMA_INDEX_WARN_FACTOR,
|
|
24
|
+
_CHROMA_REBUILD_BATCH,
|
|
25
|
+
_SKIP_SEARCH_COLUMNS,
|
|
26
|
+
_SYSTEM_TABLES,
|
|
27
|
+
_coerce_search_mode,
|
|
28
|
+
_column_spec,
|
|
29
|
+
_is_safe_identifier,
|
|
30
|
+
_now_iso,
|
|
31
|
+
_sanitize_fts_query,
|
|
32
|
+
_validate_identifier,
|
|
33
|
+
_validate_order_by,
|
|
34
|
+
)
|
|
35
|
+
|
|
36
|
+
logger = logging.getLogger("hybriddb")
|
|
37
|
+
|
|
38
|
+
class AnalyticsMixin:
|
|
39
|
+
def _init_duckdb(self) -> None:
|
|
40
|
+
self._duckdb_path = ""
|
|
41
|
+
self._duckdb_synced_tables: dict[str, dict] = {}
|
|
42
|
+
self._duckdb_conn = None
|
|
43
|
+
|
|
44
|
+
try:
|
|
45
|
+
import duckdb
|
|
46
|
+
|
|
47
|
+
self._duckdb_path = str((self.path / "analytics.duckdb").resolve())
|
|
48
|
+
self._duckdb_conn = duckdb.connect(self._duckdb_path)
|
|
49
|
+
self._duckdb_conn.execute("SET threads = 4")
|
|
50
|
+
self._duckdb_conn.execute("""
|
|
51
|
+
CREATE TABLE IF NOT EXISTS _duckdb_sync (
|
|
52
|
+
table_name TEXT PRIMARY KEY,
|
|
53
|
+
synced_count INTEGER DEFAULT 0
|
|
54
|
+
)
|
|
55
|
+
""")
|
|
56
|
+
existing = self._duckdb_conn.execute(
|
|
57
|
+
"SELECT table_name FROM _duckdb_sync"
|
|
58
|
+
).fetchall()
|
|
59
|
+
for (tname,) in existing:
|
|
60
|
+
quoted_tname = self._duckdb_quote_identifier(tname)
|
|
61
|
+
count = self._duckdb_conn.execute(
|
|
62
|
+
f"SELECT count(*) FROM {quoted_tname}"
|
|
63
|
+
).fetchone()[0]
|
|
64
|
+
cols_info = self._duckdb_conn.execute(
|
|
65
|
+
"SELECT column_name, data_type FROM information_schema.columns "
|
|
66
|
+
"WHERE table_name = ?",
|
|
67
|
+
(tname,),
|
|
68
|
+
).fetchall()
|
|
69
|
+
columns = {c: t for c, t in cols_info}
|
|
70
|
+
self._duckdb_synced_tables[tname] = {"columns": columns, "count": count}
|
|
71
|
+
except ImportError:
|
|
72
|
+
pass
|
|
73
|
+
except Exception as e:
|
|
74
|
+
logger.warning("duckdb.init_failed error=%s", e)
|
|
75
|
+
self._duckdb_path = ""
|
|
76
|
+
self._duckdb_conn = None
|
|
77
|
+
|
|
78
|
+
def _auto_register_duckdb_tables(self) -> None:
|
|
79
|
+
if not self._duckdb_path:
|
|
80
|
+
return
|
|
81
|
+
app_tables = self.list_tables()
|
|
82
|
+
new_tables = [t for t in app_tables if t not in self._duckdb_synced_tables]
|
|
83
|
+
if not new_tables:
|
|
84
|
+
return
|
|
85
|
+
for table in new_tables:
|
|
86
|
+
self.register_duckdb_table(table)
|
|
87
|
+
logger.info("duckdb.auto_registered tables=%s count=%d", new_tables, len(new_tables))
|
|
88
|
+
|
|
89
|
+
def register_duckdb_table(self, table: str) -> bool:
|
|
90
|
+
_validate_identifier(table, "table")
|
|
91
|
+
if not self._duckdb_path or self._duckdb_conn is None:
|
|
92
|
+
return False
|
|
93
|
+
|
|
94
|
+
meta = self._table_meta(table)
|
|
95
|
+
if not meta:
|
|
96
|
+
logger.warning("duckdb.register_missing_table table=%s", table)
|
|
97
|
+
return False
|
|
98
|
+
|
|
99
|
+
cols = []
|
|
100
|
+
for col_name, col_type in meta["columns"].items():
|
|
101
|
+
base = col_type.replace("_PK", "")
|
|
102
|
+
quoted_col = self._duckdb_quote_identifier(col_name)
|
|
103
|
+
if base == "BOOLEAN":
|
|
104
|
+
cols.append(f"{quoted_col} INTEGER")
|
|
105
|
+
elif base == "JSON":
|
|
106
|
+
cols.append(f"{quoted_col} TEXT")
|
|
107
|
+
elif base in ("TEXT", "LONGTEXT"):
|
|
108
|
+
cols.append(f"{quoted_col} TEXT")
|
|
109
|
+
elif base == "INTEGER" or base == "INTEGER_PK":
|
|
110
|
+
cols.append(f"{quoted_col} BIGINT")
|
|
111
|
+
else:
|
|
112
|
+
cols.append(f"{quoted_col} {base}")
|
|
113
|
+
if "id" not in meta["columns"]:
|
|
114
|
+
cols.insert(0, "id BIGINT")
|
|
115
|
+
|
|
116
|
+
quoted_table = self._duckdb_quote_identifier(table)
|
|
117
|
+
with self._db_lock:
|
|
118
|
+
dk = self._duckdb_conn
|
|
119
|
+
dk.execute(f"DROP TABLE IF EXISTS {quoted_table}")
|
|
120
|
+
dk.execute(f"CREATE TABLE {quoted_table} ({', '.join(cols)})")
|
|
121
|
+
dk.execute(
|
|
122
|
+
"INSERT OR REPLACE INTO _duckdb_sync (table_name, synced_count) VALUES (?, 0)",
|
|
123
|
+
(table,),
|
|
124
|
+
)
|
|
125
|
+
|
|
126
|
+
self._duckdb_synced_tables[table] = {"columns": dict(meta["columns"]), "count": 0}
|
|
127
|
+
self._full_sync_duckdb_table(table)
|
|
128
|
+
return True
|
|
129
|
+
|
|
130
|
+
@staticmethod
|
|
131
|
+
def _duckdb_quote_identifier(identifier: str) -> str:
|
|
132
|
+
return '"' + identifier.replace('"', '""') + '"'
|
|
133
|
+
|
|
134
|
+
def _refresh_duckdb_table_if_registered(self, table: str) -> None:
|
|
135
|
+
if table in self._duckdb_synced_tables:
|
|
136
|
+
self.register_duckdb_table(table)
|
|
137
|
+
|
|
138
|
+
def _full_sync_duckdb_table(self, table: str) -> None:
|
|
139
|
+
if not self._duckdb_path or table not in self._duckdb_synced_tables:
|
|
140
|
+
return
|
|
141
|
+
|
|
142
|
+
with self._db_lock:
|
|
143
|
+
dk = self._duckdb_conn
|
|
144
|
+
quoted_table = self._duckdb_quote_identifier(table)
|
|
145
|
+
dk.execute(f"DELETE FROM {quoted_table}")
|
|
146
|
+
try:
|
|
147
|
+
dk.execute("DETACH src")
|
|
148
|
+
except Exception:
|
|
149
|
+
pass
|
|
150
|
+
try:
|
|
151
|
+
dk.execute(f"ATTACH '{self._db_path}' AS src (TYPE sqlite)")
|
|
152
|
+
dk.execute(f"INSERT INTO {quoted_table} SELECT * FROM src.{quoted_table}")
|
|
153
|
+
finally:
|
|
154
|
+
dk.execute("DETACH src")
|
|
155
|
+
count = dk.execute(f"SELECT count(*) FROM {quoted_table}").fetchone()[0]
|
|
156
|
+
dk.execute(
|
|
157
|
+
"UPDATE _duckdb_sync SET synced_count = ? WHERE table_name = ?",
|
|
158
|
+
(count, table),
|
|
159
|
+
)
|
|
160
|
+
self._duckdb_synced_tables[table]["count"] = count
|
|
161
|
+
|
|
162
|
+
def unregister_duckdb_table(self, table: str) -> bool:
|
|
163
|
+
_validate_identifier(table, "table")
|
|
164
|
+
if table not in self._duckdb_synced_tables:
|
|
165
|
+
return False
|
|
166
|
+
|
|
167
|
+
with self._db_lock:
|
|
168
|
+
dk = self._duckdb_conn
|
|
169
|
+
dk.execute(f"DROP TABLE IF EXISTS {self._duckdb_quote_identifier(table)}")
|
|
170
|
+
dk.execute("DELETE FROM _duckdb_sync WHERE table_name = ?", (table,))
|
|
171
|
+
self._duckdb_synced_tables.pop(table, None)
|
|
172
|
+
return True
|
|
173
|
+
|
|
174
|
+
def analytics(self, sql: str) -> list[dict]:
|
|
175
|
+
if not self._duckdb_path:
|
|
176
|
+
raise RuntimeError(
|
|
177
|
+
"DuckDB analytics not available — DuckDB initialization failed "
|
|
178
|
+
"or module not installed"
|
|
179
|
+
)
|
|
180
|
+
with self._db_lock:
|
|
181
|
+
dk = self._duckdb_conn
|
|
182
|
+
result = dk.execute(sql)
|
|
183
|
+
columns = [desc[0] for desc in result.description]
|
|
184
|
+
rows = [dict(zip(columns, row)) for row in result.fetchall()]
|
|
185
|
+
return rows
|
|
186
|
+
|
|
187
|
+
def _sync_duckdb_from_journal(self, entries: list[dict]) -> None:
|
|
188
|
+
if not self._duckdb_path or not self._duckdb_synced_tables:
|
|
189
|
+
return
|
|
190
|
+
|
|
191
|
+
by_table: dict[str, dict[str, list[int]]] = {}
|
|
192
|
+
seen_ids: set[int] = set()
|
|
193
|
+
|
|
194
|
+
for e in entries:
|
|
195
|
+
if e["id"] in seen_ids:
|
|
196
|
+
continue
|
|
197
|
+
seen_ids.add(e["id"])
|
|
198
|
+
tbl = e["app_table"]
|
|
199
|
+
if tbl not in self._duckdb_synced_tables:
|
|
200
|
+
continue
|
|
201
|
+
if tbl not in by_table:
|
|
202
|
+
by_table[tbl] = {"add": [], "delete": []}
|
|
203
|
+
row_id = e["row_id"]
|
|
204
|
+
if row_id is not None:
|
|
205
|
+
by_table[tbl]["delete"].append(row_id)
|
|
206
|
+
if e["op"] != "row_delete":
|
|
207
|
+
by_table[tbl]["add"].append(row_id)
|
|
208
|
+
|
|
209
|
+
if not by_table:
|
|
210
|
+
return
|
|
211
|
+
|
|
212
|
+
with self._db_lock:
|
|
213
|
+
dk = self._duckdb_conn
|
|
214
|
+
dk.execute(f"ATTACH '{self._db_path}' AS src (TYPE sqlite)")
|
|
215
|
+
for tbl, ops in by_table.items():
|
|
216
|
+
quoted_tbl = self._duckdb_quote_identifier(tbl)
|
|
217
|
+
if ops["delete"]:
|
|
218
|
+
ids = ",".join(str(i) for i in ops["delete"])
|
|
219
|
+
dk.execute(f"DELETE FROM {quoted_tbl} WHERE id IN ({ids})")
|
|
220
|
+
if ops["add"]:
|
|
221
|
+
ids = ",".join(str(i) for i in ops["add"])
|
|
222
|
+
dk.execute(
|
|
223
|
+
f"INSERT INTO {quoted_tbl} SELECT * FROM src.{quoted_tbl} WHERE id IN ({ids})"
|
|
224
|
+
)
|
|
225
|
+
count = dk.execute(f"SELECT count(*) FROM {quoted_tbl}").fetchone()[0]
|
|
226
|
+
dk.execute(
|
|
227
|
+
"UPDATE _duckdb_sync SET synced_count = ? WHERE table_name = ?",
|
|
228
|
+
(count, tbl),
|
|
229
|
+
)
|
|
230
|
+
self._duckdb_synced_tables[tbl]["count"] = count
|
|
231
|
+
dk.execute("DETACH src")
|
|
232
|
+
|
|
233
|
+
def sync_duckdb_table(self, table: str) -> None:
|
|
234
|
+
_validate_identifier(table, "table")
|
|
235
|
+
self._full_sync_duckdb_table(table)
|
|
236
|
+
|
hybriddb/async_api.py
ADDED
|
@@ -0,0 +1,68 @@
|
|
|
1
|
+
"""Async wrappers for HybridDB."""
|
|
2
|
+
|
|
3
|
+
import asyncio
|
|
4
|
+
from typing import Any
|
|
5
|
+
|
|
6
|
+
from hybriddb.types import Column
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
class AsyncMixin:
|
|
10
|
+
async def acreate_table(self, table: str, columns: dict[str, str | Column]) -> None:
|
|
11
|
+
await asyncio.to_thread(self.create_table, table, columns)
|
|
12
|
+
|
|
13
|
+
async def aadd_column(self, table: str, column: str, col_type: str) -> None:
|
|
14
|
+
await asyncio.to_thread(self.add_column, table, column, col_type)
|
|
15
|
+
|
|
16
|
+
async def adrop_column(self, table: str, column: str) -> None:
|
|
17
|
+
await asyncio.to_thread(self.drop_column, table, column)
|
|
18
|
+
|
|
19
|
+
async def arename_column(self, table: str, old_name: str, new_name: str) -> None:
|
|
20
|
+
await asyncio.to_thread(self.rename_column, table, old_name, new_name)
|
|
21
|
+
|
|
22
|
+
async def ainsert(self, table: str, data: dict, sync: bool = True) -> int | str:
|
|
23
|
+
return await asyncio.to_thread(self.insert, table, data, sync)
|
|
24
|
+
|
|
25
|
+
async def ainsert_batch(self, table: str, rows: list[dict], sync: bool = True) -> list[int | str]:
|
|
26
|
+
return await asyncio.to_thread(self.insert_batch, table, rows, sync)
|
|
27
|
+
|
|
28
|
+
async def aupdate(self, table: str, row_id: int | str, data: dict, sync: bool = True) -> bool:
|
|
29
|
+
return await asyncio.to_thread(self.update, table, row_id, data, sync)
|
|
30
|
+
|
|
31
|
+
async def adelete(self, table: str, row_id: int | str, sync: bool = True) -> bool:
|
|
32
|
+
return await asyncio.to_thread(self.delete, table, row_id, sync)
|
|
33
|
+
|
|
34
|
+
async def aget(self, table: str, row_id: int | str) -> dict | None:
|
|
35
|
+
return await asyncio.to_thread(self.get, table, row_id)
|
|
36
|
+
|
|
37
|
+
async def aquery(
|
|
38
|
+
self, table: str, where: str = "", params: tuple = (),
|
|
39
|
+
order_by: str = "", limit: int = 100,
|
|
40
|
+
) -> list[dict]:
|
|
41
|
+
return await asyncio.to_thread(self.query, table, where, params, order_by, limit)
|
|
42
|
+
|
|
43
|
+
async def araw_query(self, sql: str, params: tuple = ()) -> list[dict]:
|
|
44
|
+
return await asyncio.to_thread(self.raw_query, sql, params)
|
|
45
|
+
|
|
46
|
+
async def aread_query(self, sql: str, params: tuple = ()) -> list[dict]:
|
|
47
|
+
return await asyncio.to_thread(self.read_query, sql, params)
|
|
48
|
+
|
|
49
|
+
async def acount(self, table: str, where: str = "", params: tuple = ()) -> int:
|
|
50
|
+
return await asyncio.to_thread(self.count, table, where, params)
|
|
51
|
+
|
|
52
|
+
async def asearch(self, table: str, column: str, query: str | None = None, **kwargs: Any) -> list[dict]:
|
|
53
|
+
return await asyncio.to_thread(self.search, table, column, query, **kwargs)
|
|
54
|
+
|
|
55
|
+
async def asearch_all(self, table: str, query: str, **kwargs: Any) -> list[dict]:
|
|
56
|
+
return await asyncio.to_thread(self.search_all, table, query, **kwargs)
|
|
57
|
+
|
|
58
|
+
async def ahealth(self, table: str) -> dict:
|
|
59
|
+
return await asyncio.to_thread(self.health, table)
|
|
60
|
+
|
|
61
|
+
async def areconcile(self, table: str) -> dict:
|
|
62
|
+
return await asyncio.to_thread(self.reconcile, table)
|
|
63
|
+
|
|
64
|
+
async def aprocess_journal(self, limit: int = 5000) -> int:
|
|
65
|
+
return await asyncio.to_thread(self.process_journal, limit)
|
|
66
|
+
|
|
67
|
+
async def aclose(self) -> None:
|
|
68
|
+
await asyncio.to_thread(self.close)
|
hybriddb/crud.py
ADDED
|
@@ -0,0 +1,275 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
import logging
|
|
5
|
+
import shutil
|
|
6
|
+
import sqlite3
|
|
7
|
+
import struct
|
|
8
|
+
import tempfile
|
|
9
|
+
import uuid
|
|
10
|
+
from collections import defaultdict
|
|
11
|
+
from datetime import UTC, datetime
|
|
12
|
+
from pathlib import Path
|
|
13
|
+
from typing import Any
|
|
14
|
+
|
|
15
|
+
from hybriddb.embedding import EMBEDDING_DIM
|
|
16
|
+
from hybriddb.types import Column, SearchMode
|
|
17
|
+
from hybriddb.utils import (
|
|
18
|
+
CHROMA_BATCH,
|
|
19
|
+
JOURNAL_CAP,
|
|
20
|
+
RRF_K,
|
|
21
|
+
_CHROMA_INDEX_MAX_ELEMENTS,
|
|
22
|
+
_CHROMA_INDEX_MAX_M0,
|
|
23
|
+
_CHROMA_INDEX_WARN_FACTOR,
|
|
24
|
+
_CHROMA_REBUILD_BATCH,
|
|
25
|
+
_SKIP_SEARCH_COLUMNS,
|
|
26
|
+
_SYSTEM_TABLES,
|
|
27
|
+
_coerce_search_mode,
|
|
28
|
+
_column_spec,
|
|
29
|
+
_is_safe_identifier,
|
|
30
|
+
_now_iso,
|
|
31
|
+
_sanitize_fts_query,
|
|
32
|
+
_validate_identifier,
|
|
33
|
+
_validate_order_by,
|
|
34
|
+
)
|
|
35
|
+
|
|
36
|
+
logger = logging.getLogger("hybriddb")
|
|
37
|
+
|
|
38
|
+
class CrudMixin:
|
|
39
|
+
def _row_to_metadata(self, table: str, row: dict[str, Any]) -> dict[str, Any]:
|
|
40
|
+
meta = self._table_meta(table)
|
|
41
|
+
if not meta:
|
|
42
|
+
return {}
|
|
43
|
+
result = {}
|
|
44
|
+
for col, ctype in meta["columns"].items():
|
|
45
|
+
if ctype in ("LONGTEXT", "JSON"):
|
|
46
|
+
continue
|
|
47
|
+
val = row.get(col)
|
|
48
|
+
if val is None:
|
|
49
|
+
continue
|
|
50
|
+
base = ctype.replace("_PK", "")
|
|
51
|
+
if base == "BOOLEAN":
|
|
52
|
+
result[col] = bool(val)
|
|
53
|
+
else:
|
|
54
|
+
result[col] = val
|
|
55
|
+
return result
|
|
56
|
+
|
|
57
|
+
def insert(
|
|
58
|
+
self, table: str, data: dict, sync: bool = True,
|
|
59
|
+
skip_journal_columns: set[str] | None = None,
|
|
60
|
+
) -> int | str:
|
|
61
|
+
_validate_identifier(table, "table")
|
|
62
|
+
meta = self._table_meta(table)
|
|
63
|
+
if not meta:
|
|
64
|
+
raise ValueError(f"Table '{table}' not found")
|
|
65
|
+
filtered = {k: v for k, v in data.items() if k in meta["columns"]}
|
|
66
|
+
columns = list(filtered.keys())
|
|
67
|
+
placeholders = ", ".join("?" * len(columns))
|
|
68
|
+
values = list(filtered.values())
|
|
69
|
+
|
|
70
|
+
with self._connect() as cur:
|
|
71
|
+
cur.execute(f"INSERT INTO {table} ({', '.join(columns)}) VALUES ({placeholders})", values)
|
|
72
|
+
internal_rowid = cur.lastrowid
|
|
73
|
+
has_auto_id = self._has_autoincrement_id(table)
|
|
74
|
+
if has_auto_id:
|
|
75
|
+
user_pk = internal_rowid
|
|
76
|
+
row = dict(cur.execute(f"SELECT * FROM {table} WHERE id = ?", (user_pk,)).fetchone())
|
|
77
|
+
elif "id" in filtered:
|
|
78
|
+
user_pk = filtered["id"]
|
|
79
|
+
row = dict(cur.execute(f"SELECT * FROM {table} WHERE id = ?", (user_pk,)).fetchone())
|
|
80
|
+
else:
|
|
81
|
+
user_pk = internal_rowid
|
|
82
|
+
row = dict(cur.execute(f"SELECT * FROM {table} WHERE rowid = ?", (internal_rowid,)).fetchone())
|
|
83
|
+
|
|
84
|
+
metadata = self._row_to_metadata(table, row)
|
|
85
|
+
now = _now_iso()
|
|
86
|
+
for col in self._get_longtext_columns(table):
|
|
87
|
+
if skip_journal_columns and col in skip_journal_columns:
|
|
88
|
+
continue
|
|
89
|
+
cur.execute(
|
|
90
|
+
"INSERT INTO _journal (app_table, row_id, column_name, op, data, metadata, created_at) "
|
|
91
|
+
"VALUES (?, ?, ?, 'add', ?, ?, ?)",
|
|
92
|
+
(table, internal_rowid, col, row.get(col, ""), json.dumps(metadata), now),
|
|
93
|
+
)
|
|
94
|
+
cur.execute(
|
|
95
|
+
"INSERT INTO _journal (app_table, row_id, op, data, created_at) "
|
|
96
|
+
"VALUES (?, ?, 'row_add', ?, ?)",
|
|
97
|
+
(table, internal_rowid, json.dumps(dict(row), default=str), now),
|
|
98
|
+
)
|
|
99
|
+
if sync:
|
|
100
|
+
self._process_journal()
|
|
101
|
+
return user_pk or 0
|
|
102
|
+
|
|
103
|
+
def row_to_metadata(self, table: str, row: dict[str, Any]) -> dict[str, Any]:
|
|
104
|
+
_validate_identifier(table, "table")
|
|
105
|
+
return self._row_to_metadata(table, row)
|
|
106
|
+
|
|
107
|
+
def vector_upsert(
|
|
108
|
+
self, collection_name: str, row_id: int | str, document: str,
|
|
109
|
+
embedding: list[float], metadata: dict[str, Any] | None = None,
|
|
110
|
+
) -> bool:
|
|
111
|
+
_validate_identifier(collection_name, "collection")
|
|
112
|
+
collection = self._get_collection(collection_name)
|
|
113
|
+
if collection is None:
|
|
114
|
+
return False
|
|
115
|
+
collection.upsert(
|
|
116
|
+
ids=[str(row_id)], embeddings=[embedding], documents=[document],
|
|
117
|
+
metadatas=[metadata or {}],
|
|
118
|
+
)
|
|
119
|
+
return True
|
|
120
|
+
|
|
121
|
+
def insert_batch(self, table: str, rows: list[dict], sync: bool = True) -> list[int | str]:
|
|
122
|
+
_validate_identifier(table, "table")
|
|
123
|
+
if len(rows) > JOURNAL_CAP:
|
|
124
|
+
logger.warning("insert_batch.large_batch table=%s rows=%d limit=%d", table, len(rows), JOURNAL_CAP)
|
|
125
|
+
meta = self._table_meta(table)
|
|
126
|
+
if not meta:
|
|
127
|
+
raise ValueError(f"Table '{table}' not found")
|
|
128
|
+
ids: list[int | str] = []
|
|
129
|
+
with self._connect() as cur:
|
|
130
|
+
now = _now_iso()
|
|
131
|
+
for data in rows:
|
|
132
|
+
filtered = {k: v for k, v in data.items() if k in meta["columns"]}
|
|
133
|
+
columns = list(filtered.keys())
|
|
134
|
+
placeholders = ", ".join("?" * len(columns))
|
|
135
|
+
values = list(filtered.values())
|
|
136
|
+
cur.execute(f"INSERT INTO {table} ({', '.join(columns)}) VALUES ({placeholders})", values)
|
|
137
|
+
internal_rowid = cur.lastrowid
|
|
138
|
+
has_auto_id = self._has_autoincrement_id(table)
|
|
139
|
+
if has_auto_id:
|
|
140
|
+
user_pk = internal_rowid
|
|
141
|
+
row = dict(cur.execute(f"SELECT * FROM {table} WHERE id = ?", (user_pk,)).fetchone())
|
|
142
|
+
elif "id" in filtered:
|
|
143
|
+
user_pk = filtered["id"]
|
|
144
|
+
row = dict(cur.execute(f"SELECT * FROM {table} WHERE id = ?", (user_pk,)).fetchone())
|
|
145
|
+
else:
|
|
146
|
+
assert internal_rowid is not None
|
|
147
|
+
user_pk = internal_rowid
|
|
148
|
+
row = dict(cur.execute(f"SELECT * FROM {table} WHERE rowid = ?", (internal_rowid,)).fetchone())
|
|
149
|
+
ids.append(user_pk)
|
|
150
|
+
metadata = self._row_to_metadata(table, row)
|
|
151
|
+
for col in self._get_longtext_columns(table):
|
|
152
|
+
cur.execute(
|
|
153
|
+
"INSERT INTO _journal (app_table, row_id, column_name, op, data, metadata, created_at) "
|
|
154
|
+
"VALUES (?, ?, ?, 'add', ?, ?, ?)",
|
|
155
|
+
(table, internal_rowid, col, row.get(col, ""), json.dumps(metadata), now),
|
|
156
|
+
)
|
|
157
|
+
cur.execute(
|
|
158
|
+
"INSERT INTO _journal (app_table, row_id, op, data, created_at) "
|
|
159
|
+
"VALUES (?, ?, 'row_add', ?, ?)",
|
|
160
|
+
(table, internal_rowid, json.dumps(dict(row), default=str), now),
|
|
161
|
+
)
|
|
162
|
+
if sync:
|
|
163
|
+
self._process_journal()
|
|
164
|
+
return ids
|
|
165
|
+
|
|
166
|
+
def update(self, table: str, row_id: int | str, data: dict, sync: bool = True) -> bool:
|
|
167
|
+
_validate_identifier(table, "table")
|
|
168
|
+
meta = self._table_meta(table)
|
|
169
|
+
if not meta:
|
|
170
|
+
raise ValueError(f"Table '{table}' not found")
|
|
171
|
+
filtered = {k: v for k, v in data.items() if k in meta["columns"]}
|
|
172
|
+
if not filtered:
|
|
173
|
+
return False
|
|
174
|
+
|
|
175
|
+
with self._connect() as cur:
|
|
176
|
+
internal_rowid = self._resolve_internal_rowid(cur, table, row_id)
|
|
177
|
+
if internal_rowid is None:
|
|
178
|
+
return False
|
|
179
|
+
set_clause = ", ".join(f"{k} = ?" for k in filtered.keys())
|
|
180
|
+
cur.execute(f"UPDATE {table} SET {set_clause} WHERE id = ?", list(filtered.values()) + [row_id])
|
|
181
|
+
if cur.rowcount == 0:
|
|
182
|
+
return False
|
|
183
|
+
row = dict(cur.execute(f"SELECT * FROM {table} WHERE id = ?", (row_id,)).fetchone())
|
|
184
|
+
metadata = self._row_to_metadata(table, row)
|
|
185
|
+
for col in self._get_longtext_columns(table):
|
|
186
|
+
now = _now_iso()
|
|
187
|
+
cur.execute(
|
|
188
|
+
"INSERT INTO _journal (app_table, row_id, column_name, op, data, metadata, created_at) "
|
|
189
|
+
"VALUES (?, ?, ?, 'update', ?, ?, ?)",
|
|
190
|
+
(table, internal_rowid, col, row.get(col, ""), json.dumps(metadata), now),
|
|
191
|
+
)
|
|
192
|
+
now = _now_iso()
|
|
193
|
+
cur.execute(
|
|
194
|
+
"INSERT INTO _journal (app_table, row_id, op, data, created_at) "
|
|
195
|
+
"VALUES (?, ?, 'row_update', ?, ?)",
|
|
196
|
+
(table, internal_rowid, json.dumps(dict(row), default=str), now),
|
|
197
|
+
)
|
|
198
|
+
if sync:
|
|
199
|
+
self._process_journal()
|
|
200
|
+
return True
|
|
201
|
+
|
|
202
|
+
def delete(self, table: str, row_id: int | str, sync: bool = True) -> bool:
|
|
203
|
+
_validate_identifier(table, "table")
|
|
204
|
+
with self._connect() as cur:
|
|
205
|
+
internal_rowid = self._resolve_internal_rowid(cur, table, row_id)
|
|
206
|
+
if internal_rowid is None:
|
|
207
|
+
return False
|
|
208
|
+
cur.execute(f"DELETE FROM {table} WHERE id = ?", (row_id,))
|
|
209
|
+
if cur.rowcount == 0:
|
|
210
|
+
return False
|
|
211
|
+
for col in self._get_longtext_columns(table):
|
|
212
|
+
now = _now_iso()
|
|
213
|
+
cur.execute(
|
|
214
|
+
"INSERT INTO _journal (app_table, row_id, column_name, op, created_at) "
|
|
215
|
+
"VALUES (?, ?, ?, 'delete', ?)",
|
|
216
|
+
(table, internal_rowid, col, now),
|
|
217
|
+
)
|
|
218
|
+
now = _now_iso()
|
|
219
|
+
cur.execute(
|
|
220
|
+
"INSERT INTO _journal (app_table, row_id, op, created_at) "
|
|
221
|
+
"VALUES (?, ?, 'row_delete', ?)",
|
|
222
|
+
(table, internal_rowid, now),
|
|
223
|
+
)
|
|
224
|
+
if sync:
|
|
225
|
+
self._process_journal()
|
|
226
|
+
return True
|
|
227
|
+
|
|
228
|
+
def get(self, table: str, row_id: int | str) -> dict | None:
|
|
229
|
+
_validate_identifier(table, "table")
|
|
230
|
+
with self._connect() as cur:
|
|
231
|
+
cur.execute(f"SELECT * FROM {table} WHERE id = ?", (row_id,))
|
|
232
|
+
row = cur.fetchone()
|
|
233
|
+
if not row:
|
|
234
|
+
return None
|
|
235
|
+
return dict(row)
|
|
236
|
+
|
|
237
|
+
def query(
|
|
238
|
+
self, table: str, where: str = "", params: tuple = (),
|
|
239
|
+
order_by: str = "", limit: int = 100,
|
|
240
|
+
) -> list[dict]:
|
|
241
|
+
_validate_identifier(table, "table")
|
|
242
|
+
_validate_order_by(order_by)
|
|
243
|
+
sql = f"SELECT * FROM {table}"
|
|
244
|
+
if where:
|
|
245
|
+
sql += f" WHERE {where}"
|
|
246
|
+
if order_by:
|
|
247
|
+
sql += f" ORDER BY {order_by}"
|
|
248
|
+
sql += f" LIMIT {limit}"
|
|
249
|
+
with self._connect() as cur:
|
|
250
|
+
cur.execute(sql, params)
|
|
251
|
+
rows = cur.fetchall()
|
|
252
|
+
return [dict(r) for r in rows]
|
|
253
|
+
|
|
254
|
+
def raw_query(self, sql: str, params: tuple = ()) -> list[dict]:
|
|
255
|
+
with self._connect() as cur:
|
|
256
|
+
cur.execute(sql, params)
|
|
257
|
+
rows = cur.fetchall()
|
|
258
|
+
return [dict(r) for r in rows]
|
|
259
|
+
|
|
260
|
+
def read_query(self, sql: str, params: tuple = ()) -> list[dict]:
|
|
261
|
+
"""Run a read-only SQL query against SQLite."""
|
|
262
|
+
first = sql.lstrip().split(None, 1)[0].upper() if sql.strip() else ""
|
|
263
|
+
if first not in {"SELECT", "WITH", "PRAGMA", "EXPLAIN"}:
|
|
264
|
+
raise ValueError("read_query only accepts read-only SQL")
|
|
265
|
+
return self.raw_query(sql, params)
|
|
266
|
+
|
|
267
|
+
def count(self, table: str, where: str = "", params: tuple = ()) -> int:
|
|
268
|
+
_validate_identifier(table, "table")
|
|
269
|
+
sql = f"SELECT COUNT(*) FROM {table}"
|
|
270
|
+
if where:
|
|
271
|
+
sql += f" WHERE {where}"
|
|
272
|
+
with self._connect() as cur:
|
|
273
|
+
result = cur.execute(sql, params).fetchone()
|
|
274
|
+
return result[0] if result else 0
|
|
275
|
+
|