bkndb 0.2.0__py3-none-win_amd64.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- bkndb/__init__.py +139 -0
- bkndb/_common.py +63 -0
- bkndb/_io.py +168 -0
- bkndb/_native/.gitignore +6 -0
- bkndb/_native/__init__.py +1 -0
- bkndb/_native/bkndb_ffi.dll +0 -0
- bkndb/_native/bkndb_ffi.py +8175 -0
- bkndb/_ops_core.py +328 -0
- bkndb/_ops_extra.py +226 -0
- bkndb/_ops_io.py +187 -0
- bkndb/_ops_search.py +101 -0
- bkndb/_records.py +176 -0
- bkndb/_tables.py +328 -0
- bkndb/_values.py +103 -0
- bkndb/database.py +100 -0
- bkndb/errors.py +160 -0
- bkndb/py.typed +0 -0
- bkndb/query.py +268 -0
- bkndb/types.py +108 -0
- bkndb-0.2.0.dist-info/METADATA +243 -0
- bkndb-0.2.0.dist-info/RECORD +23 -0
- bkndb-0.2.0.dist-info/WHEEL +4 -0
- bkndb-0.2.0.dist-info/sboms/bkndb-ffi.cyclonedx.json +3325 -0
bkndb/__init__.py
ADDED
|
@@ -0,0 +1,139 @@
|
|
|
1
|
+
"""bkndb — embedded hybrid graph + relational database engine for Python.
|
|
2
|
+
|
|
3
|
+
Public API. Only import from here — :mod:`bkndb._native` is the raw
|
|
4
|
+
UniFFI-generated layer and is an implementation detail that may change
|
|
5
|
+
shape between releases without notice.
|
|
6
|
+
|
|
7
|
+
import bkndb
|
|
8
|
+
from bkndb import col, Column, TableSchema
|
|
9
|
+
|
|
10
|
+
with bkndb.open("my.bkndb") as db:
|
|
11
|
+
doc = db.create_node("Document", {"title": "..."})
|
|
12
|
+
db.create_table(TableSchema("chunks", [Column("id", int), Column("doc", int), Column("text", str)],
|
|
13
|
+
primary_key="id", auto_increment=True, indexes=["doc"]))
|
|
14
|
+
with db.transaction() as tx:
|
|
15
|
+
tx.insert("chunks", {"doc": doc, "text": "..."})
|
|
16
|
+
rows = db.select("chunks", col("doc") == doc)
|
|
17
|
+
"""
|
|
18
|
+
from __future__ import annotations
|
|
19
|
+
|
|
20
|
+
from importlib import metadata as _metadata
|
|
21
|
+
|
|
22
|
+
from .database import Database, Table, Transaction
|
|
23
|
+
from .errors import (
|
|
24
|
+
BackendError,
|
|
25
|
+
BknDbError,
|
|
26
|
+
ConstraintViolationError,
|
|
27
|
+
CorruptionError,
|
|
28
|
+
QueryError,
|
|
29
|
+
DatabaseClosedError,
|
|
30
|
+
DatabaseLockedError,
|
|
31
|
+
DuplicateKeyError,
|
|
32
|
+
EncodingError,
|
|
33
|
+
InvalidArgumentError,
|
|
34
|
+
NotFoundError,
|
|
35
|
+
ReservedTableNameError,
|
|
36
|
+
SchemaMismatchError,
|
|
37
|
+
TableNotFoundError,
|
|
38
|
+
TransactionAbortedError,
|
|
39
|
+
TransactionClosedError,
|
|
40
|
+
TransactionError,
|
|
41
|
+
TransactionInProgressError,
|
|
42
|
+
)
|
|
43
|
+
from .query import Agg, Col, Expr, col
|
|
44
|
+
from .types import (
|
|
45
|
+
AggregateRow,
|
|
46
|
+
Column,
|
|
47
|
+
DbStats,
|
|
48
|
+
Direction,
|
|
49
|
+
Edge,
|
|
50
|
+
Hub,
|
|
51
|
+
IntegrityReport,
|
|
52
|
+
Neighbor,
|
|
53
|
+
NewNode,
|
|
54
|
+
Node,
|
|
55
|
+
Path,
|
|
56
|
+
PathStep,
|
|
57
|
+
Properties,
|
|
58
|
+
PropertiesLike,
|
|
59
|
+
PropertyValue,
|
|
60
|
+
QueryResult,
|
|
61
|
+
Row,
|
|
62
|
+
ScoredRow,
|
|
63
|
+
StorageStats,
|
|
64
|
+
SyncResult,
|
|
65
|
+
TableSchema,
|
|
66
|
+
VectorIndexInfo,
|
|
67
|
+
TraversalHit,
|
|
68
|
+
TypedNeighbor,
|
|
69
|
+
WeightedPath,
|
|
70
|
+
pack_vector,
|
|
71
|
+
)
|
|
72
|
+
|
|
73
|
+
# `bkndb.open(...)` works, but `open` is left out of `__all__` so that
|
|
74
|
+
# `from bkndb import *` can't shadow the builtin.
|
|
75
|
+
open = Database.open
|
|
76
|
+
in_memory = Database.in_memory
|
|
77
|
+
|
|
78
|
+
try:
|
|
79
|
+
__version__ = _metadata.version("bkndb")
|
|
80
|
+
except _metadata.PackageNotFoundError: # running from a source tree without an install
|
|
81
|
+
__version__ = "0.0.0+unknown"
|
|
82
|
+
|
|
83
|
+
__all__ = [
|
|
84
|
+
"Database",
|
|
85
|
+
"Transaction",
|
|
86
|
+
"Table",
|
|
87
|
+
"in_memory",
|
|
88
|
+
# queries
|
|
89
|
+
"col",
|
|
90
|
+
"Col",
|
|
91
|
+
"Expr",
|
|
92
|
+
"Agg",
|
|
93
|
+
# schemas & results
|
|
94
|
+
"Column",
|
|
95
|
+
"TableSchema",
|
|
96
|
+
"Row",
|
|
97
|
+
"AggregateRow",
|
|
98
|
+
"Direction",
|
|
99
|
+
"Node",
|
|
100
|
+
"Edge",
|
|
101
|
+
"Neighbor",
|
|
102
|
+
"NewNode",
|
|
103
|
+
"WeightedPath",
|
|
104
|
+
"TypedNeighbor",
|
|
105
|
+
"TraversalHit",
|
|
106
|
+
"Path",
|
|
107
|
+
"PathStep",
|
|
108
|
+
"Hub",
|
|
109
|
+
"SyncResult",
|
|
110
|
+
"DbStats",
|
|
111
|
+
"StorageStats",
|
|
112
|
+
"IntegrityReport",
|
|
113
|
+
"QueryResult",
|
|
114
|
+
"ScoredRow",
|
|
115
|
+
"VectorIndexInfo",
|
|
116
|
+
"pack_vector",
|
|
117
|
+
"Properties",
|
|
118
|
+
"PropertiesLike",
|
|
119
|
+
"PropertyValue",
|
|
120
|
+
# errors
|
|
121
|
+
"BknDbError",
|
|
122
|
+
"BackendError",
|
|
123
|
+
"TableNotFoundError",
|
|
124
|
+
"NotFoundError",
|
|
125
|
+
"EncodingError",
|
|
126
|
+
"ReservedTableNameError",
|
|
127
|
+
"DatabaseLockedError",
|
|
128
|
+
"DatabaseClosedError",
|
|
129
|
+
"DuplicateKeyError",
|
|
130
|
+
"SchemaMismatchError",
|
|
131
|
+
"ConstraintViolationError",
|
|
132
|
+
"TransactionError",
|
|
133
|
+
"TransactionInProgressError",
|
|
134
|
+
"TransactionClosedError",
|
|
135
|
+
"TransactionAbortedError",
|
|
136
|
+
"InvalidArgumentError",
|
|
137
|
+
"CorruptionError",
|
|
138
|
+
"QueryError",
|
|
139
|
+
]
|
bkndb/_common.py
ADDED
|
@@ -0,0 +1,63 @@
|
|
|
1
|
+
"""Type aliases and small helpers shared by the Database / Transaction wrappers."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
import typing
|
|
5
|
+
|
|
6
|
+
from . import errors, types
|
|
7
|
+
from ._native.bkndb_ffi import FfiNodeRef, FfiVectorMetric
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
T = typing.TypeVar("T")
|
|
11
|
+
|
|
12
|
+
NodeSpec = typing.Tuple[str, types.PropertiesLike]
|
|
13
|
+
EdgeSpec = typing.Tuple[int, str, int, types.PropertiesLike]
|
|
14
|
+
NodeRefLike = typing.Union[int, types.NewNode]
|
|
15
|
+
SyncEdgeSpec = typing.Tuple[NodeRefLike, str, NodeRefLike, types.PropertiesLike]
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def _call(fn: typing.Callable[[], T]) -> T:
|
|
19
|
+
"""Runs `fn`, translating anything the native layer raises into a public exception."""
|
|
20
|
+
try:
|
|
21
|
+
return fn()
|
|
22
|
+
except BaseException as exc:
|
|
23
|
+
mapped = errors.wrap(exc)
|
|
24
|
+
if mapped is exc:
|
|
25
|
+
raise
|
|
26
|
+
raise mapped from exc
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def _node_ref(r: "NodeRefLike") -> typing.Any:
|
|
30
|
+
if isinstance(r, types.NewNode):
|
|
31
|
+
return FfiNodeRef.NEW(r.index)
|
|
32
|
+
return FfiNodeRef.EXISTING(r)
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def _props(p: typing.Optional[typing.Mapping[str, types.PropertyValue]]) -> dict:
|
|
36
|
+
return types.to_ffi_properties(p or {})
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
Params = typing.Union[typing.Sequence[types.PropertyValue], typing.Mapping[str, types.PropertyValue], None]
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def _params(params: Params) -> typing.Tuple[list, typing.Optional[dict]]:
|
|
43
|
+
if params is None:
|
|
44
|
+
return [], None
|
|
45
|
+
if isinstance(params, typing.Mapping):
|
|
46
|
+
return [], {str(k): types.to_ffi_value(v) for k, v in params.items()}
|
|
47
|
+
if isinstance(params, (str, bytes)):
|
|
48
|
+
raise TypeError("params must be a list/tuple or a dict, not a single value")
|
|
49
|
+
return [types.to_ffi_value(v) for v in params], None
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def _row_or_none(record) -> typing.Optional[types.Row]:
|
|
53
|
+
return types.Row._from_ffi(record) if record is not None else None
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
_METRICS = {"cosine": FfiVectorMetric.COSINE, "dot": FfiVectorMetric.DOT, "euclidean": FfiVectorMetric.EUCLIDEAN}
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def _metric(name: str) -> FfiVectorMetric:
|
|
60
|
+
try:
|
|
61
|
+
return _METRICS[name]
|
|
62
|
+
except KeyError:
|
|
63
|
+
raise ValueError(f"metric must be one of {sorted(_METRICS)}") from None
|
bkndb/_io.py
ADDED
|
@@ -0,0 +1,168 @@
|
|
|
1
|
+
"""Row (de)serialization for :meth:`bkndb.Database.export_csv` and friends.
|
|
2
|
+
|
|
3
|
+
Formats:
|
|
4
|
+
|
|
5
|
+
* **JSONL** — one JSON object per line, ``{column: value}`` including the
|
|
6
|
+
primary key. Lists and dicts are written as JSON arrays/objects; the
|
|
7
|
+
types JSON lacks are tagged one-key objects: ``{"$bytes": "<base64>"}``,
|
|
8
|
+
``{"$datetime": "<ISO 8601, UTC>"}``, ``{"$uuid": "<hex>"}``. Types
|
|
9
|
+
round-trip exactly (except ``float`` NaN/infinity, which Python's
|
|
10
|
+
``json`` writes as ``NaN``/``Infinity``).
|
|
11
|
+
* **CSV** — a header row of column names (schema order), then one row per
|
|
12
|
+
record. An empty cell means "not given": on import the column is left
|
|
13
|
+
out, so its DEFAULT (or NULL) applies. Booleans are ``true``/``false``,
|
|
14
|
+
bytes are base64, datetimes ISO 8601, UUIDs hex, lists/dicts JSON text.
|
|
15
|
+
On import each cell is converted to its column's type.
|
|
16
|
+
"""
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
import base64
|
|
20
|
+
import datetime
|
|
21
|
+
import json
|
|
22
|
+
import os
|
|
23
|
+
import typing
|
|
24
|
+
import uuid
|
|
25
|
+
|
|
26
|
+
from . import errors, types
|
|
27
|
+
|
|
28
|
+
PathLike = typing.Union[str, "os.PathLike[str]"]
|
|
29
|
+
|
|
30
|
+
_BYTES_KEY = "$bytes"
|
|
31
|
+
_DATETIME_KEY = "$datetime"
|
|
32
|
+
_UUID_KEY = "$uuid"
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def _iso(dt: datetime.datetime) -> str:
|
|
36
|
+
micros = types.datetime_to_micros(dt)
|
|
37
|
+
return types.micros_to_datetime(micros).isoformat().replace("+00:00", "Z")
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def _parse_iso(text: str) -> datetime.datetime:
|
|
41
|
+
# `fromisoformat` only accepts a trailing "Z" from Python 3.11 on.
|
|
42
|
+
if text.endswith(("Z", "z")):
|
|
43
|
+
text = text[:-1] + "+00:00"
|
|
44
|
+
return datetime.datetime.fromisoformat(text)
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def to_json_value(v: types.PropertyValue) -> typing.Any:
|
|
48
|
+
if isinstance(v, (bytes, bytearray)):
|
|
49
|
+
return {_BYTES_KEY: base64.b64encode(bytes(v)).decode("ascii")}
|
|
50
|
+
if isinstance(v, datetime.datetime):
|
|
51
|
+
return {_DATETIME_KEY: _iso(v)}
|
|
52
|
+
if isinstance(v, uuid.UUID):
|
|
53
|
+
return {_UUID_KEY: str(v)}
|
|
54
|
+
if isinstance(v, (list, tuple)):
|
|
55
|
+
return [to_json_value(x) for x in v]
|
|
56
|
+
if isinstance(v, dict):
|
|
57
|
+
return {k: to_json_value(x) for k, x in v.items()}
|
|
58
|
+
return v
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def from_json_value(v: typing.Any) -> types.PropertyValue:
|
|
62
|
+
if isinstance(v, dict):
|
|
63
|
+
if len(v) == 1:
|
|
64
|
+
((key, inner),) = v.items()
|
|
65
|
+
try:
|
|
66
|
+
if key == _BYTES_KEY:
|
|
67
|
+
return base64.b64decode(inner, validate=True)
|
|
68
|
+
if key == _DATETIME_KEY:
|
|
69
|
+
return _parse_iso(inner)
|
|
70
|
+
if key == _UUID_KEY:
|
|
71
|
+
return uuid.UUID(inner)
|
|
72
|
+
except (ValueError, TypeError) as exc:
|
|
73
|
+
raise errors.InvalidArgumentError(f"bad {key} value {inner!r}: {exc}") from None
|
|
74
|
+
return {k: from_json_value(x) for k, x in v.items()}
|
|
75
|
+
if isinstance(v, list):
|
|
76
|
+
return [from_json_value(x) for x in v]
|
|
77
|
+
return v
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def row_to_json(schema: types.TableSchema, row: types.Row) -> typing.Dict[str, typing.Any]:
|
|
81
|
+
out: typing.Dict[str, typing.Any] = {schema.primary_key: to_json_value(row.pk)}
|
|
82
|
+
for k, v in row.values.items():
|
|
83
|
+
out[k] = to_json_value(v)
|
|
84
|
+
return out
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def _kind(column: types.Column) -> str:
|
|
88
|
+
t = column.type
|
|
89
|
+
return t if isinstance(t, str) else t.__name__
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
def to_csv_cell(v: types.PropertyValue) -> str:
|
|
93
|
+
if v is None:
|
|
94
|
+
return ""
|
|
95
|
+
if isinstance(v, bool):
|
|
96
|
+
return "true" if v else "false"
|
|
97
|
+
if isinstance(v, (bytes, bytearray)):
|
|
98
|
+
return base64.b64encode(bytes(v)).decode("ascii")
|
|
99
|
+
if isinstance(v, float):
|
|
100
|
+
return repr(v)
|
|
101
|
+
if isinstance(v, datetime.datetime):
|
|
102
|
+
return _iso(v)
|
|
103
|
+
if isinstance(v, (list, tuple, dict)):
|
|
104
|
+
return json.dumps(to_json_value(v), ensure_ascii=False)
|
|
105
|
+
return str(v)
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
_TRUE = {"true", "t", "1", "yes", "y"}
|
|
109
|
+
_FALSE = {"false", "f", "0", "no", "n"}
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
def from_csv_cell(column: types.Column, cell: str) -> types.PropertyValue:
|
|
113
|
+
kind = _kind(column)
|
|
114
|
+
try:
|
|
115
|
+
if kind == "str":
|
|
116
|
+
return cell
|
|
117
|
+
if kind == "int":
|
|
118
|
+
return int(cell)
|
|
119
|
+
if kind == "float":
|
|
120
|
+
return float(cell)
|
|
121
|
+
if kind == "bool":
|
|
122
|
+
low = cell.strip().lower()
|
|
123
|
+
if low in _TRUE:
|
|
124
|
+
return True
|
|
125
|
+
if low in _FALSE:
|
|
126
|
+
return False
|
|
127
|
+
raise ValueError(f"not a boolean: {cell!r}")
|
|
128
|
+
if kind == "bytes":
|
|
129
|
+
return base64.b64decode(cell, validate=True)
|
|
130
|
+
if kind in ("timestamp", "datetime"):
|
|
131
|
+
return _parse_iso(cell)
|
|
132
|
+
if kind in ("uuid", "UUID"):
|
|
133
|
+
return uuid.UUID(cell)
|
|
134
|
+
if kind in ("list", "map", "dict"):
|
|
135
|
+
value = from_json_value(json.loads(cell))
|
|
136
|
+
if not isinstance(value, list if kind == "list" else dict):
|
|
137
|
+
raise ValueError(f"expected a JSON {'array' if kind == 'list' else 'object'}")
|
|
138
|
+
return value
|
|
139
|
+
except ValueError as exc:
|
|
140
|
+
raise errors.InvalidArgumentError(f"column '{column.name}': cannot read {cell!r} as {kind}: {exc}") from None
|
|
141
|
+
raise errors.InvalidArgumentError(f"column '{column.name}' has unsupported type {kind!r}")
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
def csv_header(schema: types.TableSchema) -> typing.List[str]:
|
|
145
|
+
names = [c.name for c in schema.columns]
|
|
146
|
+
if schema.primary_key in names:
|
|
147
|
+
names.remove(schema.primary_key)
|
|
148
|
+
return [schema.primary_key, *names]
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
def row_to_csv(header: typing.Sequence[str], schema: types.TableSchema, row: types.Row) -> typing.List[str]:
|
|
152
|
+
return [to_csv_cell(row.pk if name == schema.primary_key else row.values.get(name)) for name in header]
|
|
153
|
+
|
|
154
|
+
|
|
155
|
+
def csv_record_to_row(
|
|
156
|
+
schema: types.TableSchema, header: typing.Sequence[str], record: typing.Sequence[str], line: int
|
|
157
|
+
) -> typing.Dict[str, types.PropertyValue]:
|
|
158
|
+
if len(record) != len(header):
|
|
159
|
+
raise errors.InvalidArgumentError(f"CSV line {line}: expected {len(header)} fields, got {len(record)}")
|
|
160
|
+
row: typing.Dict[str, types.PropertyValue] = {}
|
|
161
|
+
for name, cell in zip(header, record):
|
|
162
|
+
if cell == "":
|
|
163
|
+
continue
|
|
164
|
+
column = schema.column(name)
|
|
165
|
+
if column is None:
|
|
166
|
+
raise errors.SchemaMismatchError(schema.name, f"CSV column '{name}' is not in the table")
|
|
167
|
+
row[name] = from_csv_cell(column, cell)
|
|
168
|
+
return row
|
bkndb/_native/.gitignore
ADDED
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
# Everything in this directory is generated by maturin (`maturin develop` /
|
|
2
|
+
# `maturin build`) from crates/bkndb-ffi: the UniFFI Python bindings, their
|
|
3
|
+
# `__init__.py`, and the compiled library they are checksum-matched with.
|
|
4
|
+
# Never commit or hand-edit these.
|
|
5
|
+
*
|
|
6
|
+
!.gitignore
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
from .bkndb_ffi import * # NOQA
|
|
Binary file
|