astrocyte-sqlite 0.16.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- astrocyte_sqlite-0.16.0/.gitignore +60 -0
- astrocyte_sqlite-0.16.0/PKG-INFO +85 -0
- astrocyte_sqlite-0.16.0/README.md +69 -0
- astrocyte_sqlite-0.16.0/astrocyte_sqlite/__init__.py +11 -0
- astrocyte_sqlite-0.16.0/astrocyte_sqlite/store.py +799 -0
- astrocyte_sqlite-0.16.0/pyproject.toml +59 -0
- astrocyte_sqlite-0.16.0/tests/test_parity_postgres.py +291 -0
- astrocyte_sqlite-0.16.0/tests/test_pipeline_integration.py +85 -0
- astrocyte_sqlite-0.16.0/tests/test_sqlite_store.py +356 -0
- astrocyte_sqlite-0.16.0/uv.lock +2440 -0
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
# Node (Starlight in docs/)
|
|
2
|
+
node_modules/
|
|
3
|
+
.pnpm-store/
|
|
4
|
+
docs/.astro/
|
|
5
|
+
|
|
6
|
+
# Generated by docs/scripts/fetch-pypi-pins.mjs (recreated on every pnpm dev/build)
|
|
7
|
+
docs/src/data/pypi-install-pins.json
|
|
8
|
+
|
|
9
|
+
# Generated by docs/scripts/sync-docs.mjs (recreated on every pnpm dev/build)
|
|
10
|
+
docs/src/content/docs/introduction.md
|
|
11
|
+
docs/src/content/docs/design/
|
|
12
|
+
docs/src/content/docs/plugins/
|
|
13
|
+
docs/src/content/docs/end-user/
|
|
14
|
+
docs/src/content/docs/tutorials/
|
|
15
|
+
|
|
16
|
+
# Claude Code
|
|
17
|
+
.claude/
|
|
18
|
+
|
|
19
|
+
# OS
|
|
20
|
+
.DS_Store
|
|
21
|
+
|
|
22
|
+
# Local reference material (not tracked)
|
|
23
|
+
references/
|
|
24
|
+
|
|
25
|
+
# nWave feature artifacts (per-feature discover/deliver, roadmaps, execution logs).
|
|
26
|
+
# Intended to be a separate private worktree or nested git repo; not part of public Astrocyte.
|
|
27
|
+
docs/feature/
|
|
28
|
+
|
|
29
|
+
# Python
|
|
30
|
+
__pycache__/
|
|
31
|
+
*.py[cod]
|
|
32
|
+
*$py.class
|
|
33
|
+
.venv/
|
|
34
|
+
venv/
|
|
35
|
+
.env
|
|
36
|
+
.env.*
|
|
37
|
+
!.env.example
|
|
38
|
+
*.egg-info/
|
|
39
|
+
.eggs/
|
|
40
|
+
dist/
|
|
41
|
+
build/
|
|
42
|
+
.pytest_cache/
|
|
43
|
+
.mypy_cache/
|
|
44
|
+
.ruff_cache/
|
|
45
|
+
.coverage
|
|
46
|
+
htmlcov/
|
|
47
|
+
benchmark-results/
|
|
48
|
+
datasets/
|
|
49
|
+
|
|
50
|
+
# Rust
|
|
51
|
+
target/
|
|
52
|
+
|
|
53
|
+
# IDE (optional)
|
|
54
|
+
.idea/
|
|
55
|
+
astrocyte-py/logs/
|
|
56
|
+
astrocyte-py/logs/ # if not already there
|
|
57
|
+
docs/src/content/docs/reference-archive/
|
|
58
|
+
docs/public/openapi.json
|
|
59
|
+
astrocyte-services-py/astrocyte-aml-py/.venv/
|
|
60
|
+
astrocyte-services-py/astrocyte-aml-py/runs/
|
|
@@ -0,0 +1,85 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: astrocyte-sqlite
|
|
3
|
+
Version: 0.16.0
|
|
4
|
+
Summary: SQLite adapter for Astrocyte (vector + document stores in one local file; zero infrastructure)
|
|
5
|
+
License-Expression: Apache-2.0
|
|
6
|
+
Requires-Python: >=3.11
|
|
7
|
+
Requires-Dist: astrocyte<2,>=0.15.0
|
|
8
|
+
Requires-Dist: numpy>=1.26
|
|
9
|
+
Provides-Extra: dev
|
|
10
|
+
Requires-Dist: astrocyte-postgres; extra == 'dev'
|
|
11
|
+
Requires-Dist: astrocyte[mcp]; extra == 'dev'
|
|
12
|
+
Requires-Dist: pytest-asyncio>=0.23; extra == 'dev'
|
|
13
|
+
Requires-Dist: pytest-timeout>=2.2; extra == 'dev'
|
|
14
|
+
Requires-Dist: pytest>=8.0; extra == 'dev'
|
|
15
|
+
Description-Content-Type: text/markdown
|
|
16
|
+
|
|
17
|
+
# astrocyte-sqlite
|
|
18
|
+
|
|
19
|
+
Zero-infrastructure storage for Astrocyte: every memory in one SQLite file.
|
|
20
|
+
No server, no Docker, no native extensions — `pip install` and retain.
|
|
21
|
+
|
|
22
|
+
```yaml
|
|
23
|
+
# astrocyte.yaml
|
|
24
|
+
provider_tier: storage
|
|
25
|
+
vector_store: sqlite
|
|
26
|
+
vector_store_config:
|
|
27
|
+
path: ~/.local/share/astrocyte/astrocyte.db # default; or ASTROCYTE_SQLITE_PATH
|
|
28
|
+
```
|
|
29
|
+
|
|
30
|
+
`SqliteStore` satisfies both the `VectorStore` and `DocumentStore` protocols,
|
|
31
|
+
so the pipeline's keyword retrieval leg turns on automatically (the same
|
|
32
|
+
auto-wire `PostgresStore` gets).
|
|
33
|
+
|
|
34
|
+
## Behaves like Postgres, deliberately
|
|
35
|
+
|
|
36
|
+
Semantics match `astrocyte-postgres`, not the in-memory test store, because
|
|
37
|
+
Postgres is what users deploy and what every benchmark measures:
|
|
38
|
+
|
|
39
|
+
| Behaviour | `SqliteStore` / `PostgresStore` |
|
|
40
|
+
|---|---|
|
|
41
|
+
| `delete` | soft delete (`forgotten_at`); `as_of` time travel still sees it |
|
|
42
|
+
| re-storing a deleted id | resurrects it |
|
|
43
|
+
| `tags` / `fact_types` filters | items without tags / type are excluded |
|
|
44
|
+
| similarity score | cosine, clamped to [0, 1] |
|
|
45
|
+
| `search_similar` + `time_range` | not applied (`list_recent_vectors` applies it) |
|
|
46
|
+
| batch with a wrong-length vector | whole batch rejected |
|
|
47
|
+
| keyword query | English stopwords dropped, remaining terms ANDed, stemmed |
|
|
48
|
+
|
|
49
|
+
`tests/test_parity_postgres.py` runs identical operations against both
|
|
50
|
+
backends and fails on any divergence. Known, accepted differences: keyword
|
|
51
|
+
*ranking* (FTS5 `bm25` vs `ts_rank_cd` — the pipeline fuses by rank, and the
|
|
52
|
+
match set is identical), stemmer edge cases (Porter vs Snowball), and zero
|
|
53
|
+
vectors (score 0.0 here; NaN in pgvector).
|
|
54
|
+
|
|
55
|
+
## Design
|
|
56
|
+
|
|
57
|
+
- **Exact cosine in numpy over float32 BLOBs.** pgvector stores float4 too.
|
|
58
|
+
`sqlite-vec` was rejected: it needs `enable_load_extension`, which some Python
|
|
59
|
+
builds (including macOS system Python) omit.
|
|
60
|
+
- **FTS5 with a LIKE fallback** for SQLite builds compiled without it.
|
|
61
|
+
`health()` reports which is active.
|
|
62
|
+
- **Safe across processes.** An MCP server and short-lived agent-hook
|
|
63
|
+
processes write the same file: WAL journal, `BEGIN IMMEDIATE` writes, busy
|
|
64
|
+
timeout, a connection per call.
|
|
65
|
+
- **Embedding dimension is pinned on first write.** Switching embedding models
|
|
66
|
+
against an existing file fails loudly instead of mixing incomparable vectors.
|
|
67
|
+
- **Scale.** Exact search is linear in a bank's live memories. Measured on an
|
|
68
|
+
Apple Silicon laptop, median of 15 queries, `limit=50`:
|
|
69
|
+
|
|
70
|
+
| memories | dim | vector query | keyword query | file |
|
|
71
|
+
|---:|---:|---:|---:|---:|
|
|
72
|
+
| 1,000 | 384 | 3.7 ms | 3.5 ms | 2 MB |
|
|
73
|
+
| 10,000 | 384 | 23 ms | 19 ms | 21 MB |
|
|
74
|
+
| 50,000 | 384 | 116 ms | 95 ms | 107 MB |
|
|
75
|
+
| 10,000 | 1536 | 68 ms | 27 ms | 83 MB |
|
|
76
|
+
| 50,000 | 1536 | 420 ms | 143 ms | 414 MB |
|
|
77
|
+
|
|
78
|
+
Use the embedding model's native width: `local_embeddings` with
|
|
79
|
+
`pad_to: null` gives bge-small's 384 dims. Zero-padding to 1536 (needed only
|
|
80
|
+
for Postgres's fixed `vector(1536)` column) costs ~4x in time and disk and
|
|
81
|
+
changes no similarity. Keyword figures are pessimistic: the synthetic corpus
|
|
82
|
+
puts the query terms in nearly every document.
|
|
83
|
+
- **Identifiers sort bytewise.** `list_vectors` orders by `id` in byte order.
|
|
84
|
+
Postgres orders by the database collation; the two agree for the lowercase
|
|
85
|
+
UUIDs Astrocyte generates, but could differ for arbitrary mixed-case ids.
|
|
@@ -0,0 +1,69 @@
|
|
|
1
|
+
# astrocyte-sqlite
|
|
2
|
+
|
|
3
|
+
Zero-infrastructure storage for Astrocyte: every memory in one SQLite file.
|
|
4
|
+
No server, no Docker, no native extensions — `pip install` and retain.
|
|
5
|
+
|
|
6
|
+
```yaml
|
|
7
|
+
# astrocyte.yaml
|
|
8
|
+
provider_tier: storage
|
|
9
|
+
vector_store: sqlite
|
|
10
|
+
vector_store_config:
|
|
11
|
+
path: ~/.local/share/astrocyte/astrocyte.db # default; or ASTROCYTE_SQLITE_PATH
|
|
12
|
+
```
|
|
13
|
+
|
|
14
|
+
`SqliteStore` satisfies both the `VectorStore` and `DocumentStore` protocols,
|
|
15
|
+
so the pipeline's keyword retrieval leg turns on automatically (the same
|
|
16
|
+
auto-wire `PostgresStore` gets).
|
|
17
|
+
|
|
18
|
+
## Behaves like Postgres, deliberately
|
|
19
|
+
|
|
20
|
+
Semantics match `astrocyte-postgres`, not the in-memory test store, because
|
|
21
|
+
Postgres is what users deploy and what every benchmark measures:
|
|
22
|
+
|
|
23
|
+
| Behaviour | `SqliteStore` / `PostgresStore` |
|
|
24
|
+
|---|---|
|
|
25
|
+
| `delete` | soft delete (`forgotten_at`); `as_of` time travel still sees it |
|
|
26
|
+
| re-storing a deleted id | resurrects it |
|
|
27
|
+
| `tags` / `fact_types` filters | items without tags / type are excluded |
|
|
28
|
+
| similarity score | cosine, clamped to [0, 1] |
|
|
29
|
+
| `search_similar` + `time_range` | not applied (`list_recent_vectors` applies it) |
|
|
30
|
+
| batch with a wrong-length vector | whole batch rejected |
|
|
31
|
+
| keyword query | English stopwords dropped, remaining terms ANDed, stemmed |
|
|
32
|
+
|
|
33
|
+
`tests/test_parity_postgres.py` runs identical operations against both
|
|
34
|
+
backends and fails on any divergence. Known, accepted differences: keyword
|
|
35
|
+
*ranking* (FTS5 `bm25` vs `ts_rank_cd` — the pipeline fuses by rank, and the
|
|
36
|
+
match set is identical), stemmer edge cases (Porter vs Snowball), and zero
|
|
37
|
+
vectors (score 0.0 here; NaN in pgvector).
|
|
38
|
+
|
|
39
|
+
## Design
|
|
40
|
+
|
|
41
|
+
- **Exact cosine in numpy over float32 BLOBs.** pgvector stores float4 too.
|
|
42
|
+
`sqlite-vec` was rejected: it needs `enable_load_extension`, which some Python
|
|
43
|
+
builds (including macOS system Python) omit.
|
|
44
|
+
- **FTS5 with a LIKE fallback** for SQLite builds compiled without it.
|
|
45
|
+
`health()` reports which is active.
|
|
46
|
+
- **Safe across processes.** An MCP server and short-lived agent-hook
|
|
47
|
+
processes write the same file: WAL journal, `BEGIN IMMEDIATE` writes, busy
|
|
48
|
+
timeout, a connection per call.
|
|
49
|
+
- **Embedding dimension is pinned on first write.** Switching embedding models
|
|
50
|
+
against an existing file fails loudly instead of mixing incomparable vectors.
|
|
51
|
+
- **Scale.** Exact search is linear in a bank's live memories. Measured on an
|
|
52
|
+
Apple Silicon laptop, median of 15 queries, `limit=50`:
|
|
53
|
+
|
|
54
|
+
| memories | dim | vector query | keyword query | file |
|
|
55
|
+
|---:|---:|---:|---:|---:|
|
|
56
|
+
| 1,000 | 384 | 3.7 ms | 3.5 ms | 2 MB |
|
|
57
|
+
| 10,000 | 384 | 23 ms | 19 ms | 21 MB |
|
|
58
|
+
| 50,000 | 384 | 116 ms | 95 ms | 107 MB |
|
|
59
|
+
| 10,000 | 1536 | 68 ms | 27 ms | 83 MB |
|
|
60
|
+
| 50,000 | 1536 | 420 ms | 143 ms | 414 MB |
|
|
61
|
+
|
|
62
|
+
Use the embedding model's native width: `local_embeddings` with
|
|
63
|
+
`pad_to: null` gives bge-small's 384 dims. Zero-padding to 1536 (needed only
|
|
64
|
+
for Postgres's fixed `vector(1536)` column) costs ~4x in time and disk and
|
|
65
|
+
changes no similarity. Keyword figures are pessimistic: the synthetic corpus
|
|
66
|
+
puts the query terms in nearly every document.
|
|
67
|
+
- **Identifiers sort bytewise.** `list_vectors` orders by `id` in byte order.
|
|
68
|
+
Postgres orders by the database collation; the two agree for the lowercase
|
|
69
|
+
UUIDs Astrocyte generates, but could differ for arbitrary mixed-case ids.
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
"""SQLite adapter for Astrocyte — zero-infrastructure local storage.
|
|
2
|
+
|
|
3
|
+
One file on disk, no server, no native extensions. Provides a
|
|
4
|
+
:class:`SqliteStore` satisfying both the VectorStore and DocumentStore
|
|
5
|
+
protocols, with semantics matched to :class:`astrocyte_postgres.PostgresStore`
|
|
6
|
+
so recall behaves the same on a laptop as on the benched production backend.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from astrocyte_sqlite.store import SqliteStore
|
|
10
|
+
|
|
11
|
+
__all__ = ["SqliteStore"]
|