citadeldb-haystack 2.1.0__tar.gz → 2.2.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -5,6 +5,7 @@
5
5
  *~
6
6
  .DS_Store
7
7
  site/public/
8
+ site/data/release.json
8
9
  site/static/wasm/*.wasm
9
10
  site/static/wasm/*.js
10
11
  /notes/
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: citadeldb-haystack
3
- Version: 2.1.0
3
+ Version: 2.2.0
4
4
  Summary: Haystack document store backed by Citadel: encrypted at rest, with deletes that destroy the key
5
5
  Project-URL: Homepage, https://citadeldb.dev
6
6
  Project-URL: Repository, https://github.com/yp3y5akh0v/citadel
@@ -13,7 +13,7 @@ Classifier: Programming Language :: Python :: 3
13
13
  Classifier: Topic :: Database
14
14
  Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
15
15
  Requires-Python: >=3.10
16
- Requires-Dist: citadeldb<3,>=2.0
16
+ Requires-Dist: citadeldb<3,>=2.2
17
17
  Requires-Dist: haystack-ai<4,>=2.9
18
18
  Provides-Extra: test
19
19
  Requires-Dist: pytest-asyncio>=0.23; extra == 'test'
@@ -26,19 +26,20 @@ A [Haystack](https://github.com/deepset-ai/haystack) `DocumentStore` backed by
26
26
  [Citadel](https://citadeldb.dev). Encrypted at rest, embedded in your process, and deletes
27
27
  that destroy the key, not just the row.
28
28
 
29
- Passes deepset's own `DocumentStoreBaseTests` conformance suite.
30
-
31
29
  ```
32
- pip install citadeldb-haystack
30
+ pip install citadeldb-haystack sentence-transformers
33
31
  ```
34
32
 
33
+ Requires `citadeldb>=2.2,<3` and `haystack-ai>=2.9,<4`. Set `CITADEL_KEY` before
34
+ running the example. The embedding model downloads on first use and runs locally.
35
+
35
36
  ```python
36
37
  from haystack import Document
37
38
  from haystack.components.embedders import SentenceTransformersTextEmbedder
38
39
  from haystack.utils import Secret
39
40
  from citadeldb_haystack import CitadelDocumentStore
40
41
 
41
- embedder = SentenceTransformersTextEmbedder()
42
+ embedder = SentenceTransformersTextEmbedder(model="sentence-transformers/all-mpnet-base-v2")
42
43
  store = CitadelDocumentStore(
43
44
  "corpus.cdl",
44
45
  Secret.from_env_var("CITADEL_KEY"),
@@ -46,9 +47,10 @@ store = CitadelDocumentStore(
46
47
  dim=768,
47
48
  embedding_similarity_function="cosine",
48
49
  )
49
- # CITADEL_KEY must be set: an env-var secret is what lets a pipeline serialize.
50
-
51
- store.write_documents([Document(id="d1", content="...", meta={"chapter": "intro"})])
50
+ store.write_documents([
51
+ Document(id="d1", content="The deployment failed because the disk was full.",
52
+ meta={"chapter": "intro"}),
53
+ ])
52
54
  store.filter_documents({"field": "meta.chapter", "operator": "==", "value": "intro"})
53
55
  ```
54
56
 
@@ -60,10 +62,10 @@ Embedders exposing `model_id`, `model`, or `model_name` record that identity aut
60
62
  in that order. For a custom component without any of those attributes, pass a stable
61
63
  `model_id=` explicitly; Citadel refuses to guess from the Python class name.
62
64
 
63
- ## The passphrase never lands in a pipeline file
65
+ ## Pipeline serialization
64
66
 
65
- The passphrase is a Haystack `Secret`. Pipelines are serialized to disk, and a literal
66
- token refuses to serialize, so a passphrase cannot be written into a pipeline by accident:
67
+ Use a Haystack environment-variable `Secret` for pipeline serialization. Literal
68
+ passphrases cannot be serialized:
67
69
 
68
70
  ```python
69
71
  CitadelDocumentStore(
@@ -72,7 +74,8 @@ CitadelDocumentStore(
72
74
  # ValueError: Cannot serialize token-based secret.
73
75
 
74
76
  CitadelDocumentStore(
75
- "corpus.cdl", Secret.from_env_var("CITADEL_KEY"), embedder=embedder, dim=768
77
+ "corpus.cdl", Secret.from_env_var("CITADEL_KEY"), embedder=embedder, dim=768,
78
+ embedding_similarity_function="cosine",
76
79
  ).to_dict()
77
80
  # {... "key": {"type": "env_var", "env_vars": ["CITADEL_KEY"], ...}}
78
81
  ```
@@ -81,22 +84,22 @@ Use `Secret.from_env_var` for any store that goes into a saved pipeline.
81
84
 
82
85
  ## Deletes destroy the key
83
86
 
84
- Every document is sealed under its own key. Deleting destroys that key and then removes the
85
- row, so any ciphertext surviving elsewhere stays unreadable.
87
+ Every document is sealed under its own key. Deleting destroys that key and removes the
88
+ row. Pre-erasure backups or snapshots containing keys, and exported plaintext, are outside
89
+ that erasure.
86
90
 
87
91
  ```python
88
92
  store.delete_documents(["d1"])
89
93
  store.delete_all() # returns the number erased
90
94
  ```
91
95
 
92
- `DuplicatePolicy.NONE` falls back to `FAIL`, as `InMemoryDocumentStore` does, so an
93
- accidental re-write is reported rather than silently replacing a document whose key would
94
- then be destroyed.
96
+ `DuplicatePolicy.NONE` is treated as `FAIL`: duplicate ids are rejected.
95
97
 
96
98
  ## Retrieval
97
99
 
98
100
  ```python
99
- query_embedding = [0.0] * 768 # from your Haystack text embedder, `dim` wide
101
+ embedder.warm_up()
102
+ query_embedding = embedder.run(text="Why did the release break?")["embedding"]
100
103
 
101
104
  store.embedding_retrieval(
102
105
  query_embedding,
@@ -107,23 +110,18 @@ store.embedding_retrieval(
107
110
  )
108
111
  ```
109
112
 
110
- Filtering uses Haystack's own evaluator, so the whole filter language, date comparisons
111
- included, matches `InMemoryDocumentStore` operator for operator. `top_k` is `top_k`: a
112
- filter matching only distant documents still returns them, however many others outrank
113
- them.
114
-
115
- A top-level `AND` of string equality conditions is pushed into the scan, including nested
116
- paths like `meta.person.name`. Everything else is evaluated afterwards, so the two agree:
117
- nothing is pushed under `OR` or `NOT`, and numbers are not pushed either, because `==` here
118
- is Python's (`1 == 1.0`) where the stored comparison is JSON-type exact.
113
+ Filters use Haystack's evaluator.
114
+ Top-level `AND` string equalities, including nested paths such as `meta.person.name`,
115
+ are passed to Citadel as payload filters. Other predicates filter ranked candidates; the
116
+ search window expands until `top_k` matches survive or the region is exhausted.
119
117
 
120
118
  ## Notes
121
119
 
122
120
  The store requires a Haystack text embedder. Documents that arrive without a vector are
123
121
  embedded with it, while vectors already supplied by the pipeline are stored as-is. Pass the
124
122
  same model to the pipeline and store so both paths remain in one vector space. The store warms
125
- the embedder lazily before its first model call and serializes it, so pipeline round-trips retain
126
- the model instead of reopening with a hidden fallback.
123
+ the embedder lazily before its first model call and includes its configuration in pipeline
124
+ serialization.
127
125
 
128
126
  Citadel is embedded and one process owns the file. A path already open on this thread,
129
127
  under the same passphrase, is shared, so this can sit on the same database as another
@@ -4,19 +4,20 @@ A [Haystack](https://github.com/deepset-ai/haystack) `DocumentStore` backed by
4
4
  [Citadel](https://citadeldb.dev). Encrypted at rest, embedded in your process, and deletes
5
5
  that destroy the key, not just the row.
6
6
 
7
- Passes deepset's own `DocumentStoreBaseTests` conformance suite.
8
-
9
7
  ```
10
- pip install citadeldb-haystack
8
+ pip install citadeldb-haystack sentence-transformers
11
9
  ```
12
10
 
11
+ Requires `citadeldb>=2.2,<3` and `haystack-ai>=2.9,<4`. Set `CITADEL_KEY` before
12
+ running the example. The embedding model downloads on first use and runs locally.
13
+
13
14
  ```python
14
15
  from haystack import Document
15
16
  from haystack.components.embedders import SentenceTransformersTextEmbedder
16
17
  from haystack.utils import Secret
17
18
  from citadeldb_haystack import CitadelDocumentStore
18
19
 
19
- embedder = SentenceTransformersTextEmbedder()
20
+ embedder = SentenceTransformersTextEmbedder(model="sentence-transformers/all-mpnet-base-v2")
20
21
  store = CitadelDocumentStore(
21
22
  "corpus.cdl",
22
23
  Secret.from_env_var("CITADEL_KEY"),
@@ -24,9 +25,10 @@ store = CitadelDocumentStore(
24
25
  dim=768,
25
26
  embedding_similarity_function="cosine",
26
27
  )
27
- # CITADEL_KEY must be set: an env-var secret is what lets a pipeline serialize.
28
-
29
- store.write_documents([Document(id="d1", content="...", meta={"chapter": "intro"})])
28
+ store.write_documents([
29
+ Document(id="d1", content="The deployment failed because the disk was full.",
30
+ meta={"chapter": "intro"}),
31
+ ])
30
32
  store.filter_documents({"field": "meta.chapter", "operator": "==", "value": "intro"})
31
33
  ```
32
34
 
@@ -38,10 +40,10 @@ Embedders exposing `model_id`, `model`, or `model_name` record that identity aut
38
40
  in that order. For a custom component without any of those attributes, pass a stable
39
41
  `model_id=` explicitly; Citadel refuses to guess from the Python class name.
40
42
 
41
- ## The passphrase never lands in a pipeline file
43
+ ## Pipeline serialization
42
44
 
43
- The passphrase is a Haystack `Secret`. Pipelines are serialized to disk, and a literal
44
- token refuses to serialize, so a passphrase cannot be written into a pipeline by accident:
45
+ Use a Haystack environment-variable `Secret` for pipeline serialization. Literal
46
+ passphrases cannot be serialized:
45
47
 
46
48
  ```python
47
49
  CitadelDocumentStore(
@@ -50,7 +52,8 @@ CitadelDocumentStore(
50
52
  # ValueError: Cannot serialize token-based secret.
51
53
 
52
54
  CitadelDocumentStore(
53
- "corpus.cdl", Secret.from_env_var("CITADEL_KEY"), embedder=embedder, dim=768
55
+ "corpus.cdl", Secret.from_env_var("CITADEL_KEY"), embedder=embedder, dim=768,
56
+ embedding_similarity_function="cosine",
54
57
  ).to_dict()
55
58
  # {... "key": {"type": "env_var", "env_vars": ["CITADEL_KEY"], ...}}
56
59
  ```
@@ -59,22 +62,22 @@ Use `Secret.from_env_var` for any store that goes into a saved pipeline.
59
62
 
60
63
  ## Deletes destroy the key
61
64
 
62
- Every document is sealed under its own key. Deleting destroys that key and then removes the
63
- row, so any ciphertext surviving elsewhere stays unreadable.
65
+ Every document is sealed under its own key. Deleting destroys that key and removes the
66
+ row. Pre-erasure backups or snapshots containing keys, and exported plaintext, are outside
67
+ that erasure.
64
68
 
65
69
  ```python
66
70
  store.delete_documents(["d1"])
67
71
  store.delete_all() # returns the number erased
68
72
  ```
69
73
 
70
- `DuplicatePolicy.NONE` falls back to `FAIL`, as `InMemoryDocumentStore` does, so an
71
- accidental re-write is reported rather than silently replacing a document whose key would
72
- then be destroyed.
74
+ `DuplicatePolicy.NONE` is treated as `FAIL`: duplicate ids are rejected.
73
75
 
74
76
  ## Retrieval
75
77
 
76
78
  ```python
77
- query_embedding = [0.0] * 768 # from your Haystack text embedder, `dim` wide
79
+ embedder.warm_up()
80
+ query_embedding = embedder.run(text="Why did the release break?")["embedding"]
78
81
 
79
82
  store.embedding_retrieval(
80
83
  query_embedding,
@@ -85,23 +88,18 @@ store.embedding_retrieval(
85
88
  )
86
89
  ```
87
90
 
88
- Filtering uses Haystack's own evaluator, so the whole filter language, date comparisons
89
- included, matches `InMemoryDocumentStore` operator for operator. `top_k` is `top_k`: a
90
- filter matching only distant documents still returns them, however many others outrank
91
- them.
92
-
93
- A top-level `AND` of string equality conditions is pushed into the scan, including nested
94
- paths like `meta.person.name`. Everything else is evaluated afterwards, so the two agree:
95
- nothing is pushed under `OR` or `NOT`, and numbers are not pushed either, because `==` here
96
- is Python's (`1 == 1.0`) where the stored comparison is JSON-type exact.
91
+ Filters use Haystack's evaluator.
92
+ Top-level `AND` string equalities, including nested paths such as `meta.person.name`,
93
+ are passed to Citadel as payload filters. Other predicates filter ranked candidates; the
94
+ search window expands until `top_k` matches survive or the region is exhausted.
97
95
 
98
96
  ## Notes
99
97
 
100
98
  The store requires a Haystack text embedder. Documents that arrive without a vector are
101
99
  embedded with it, while vectors already supplied by the pipeline are stored as-is. Pass the
102
100
  same model to the pipeline and store so both paths remain in one vector space. The store warms
103
- the embedder lazily before its first model call and serializes it, so pipeline round-trips retain
104
- the model instead of reopening with a hidden fallback.
101
+ the embedder lazily before its first model call and includes its configuration in pipeline
102
+ serialization.
105
103
 
106
104
  Citadel is embedded and one process owns the file. A path already open on this thread,
107
105
  under the same passphrase, is shared, so this can sit on the same database as another
@@ -18,8 +18,8 @@ classifiers = [
18
18
  "Topic :: Database",
19
19
  "Topic :: Scientific/Engineering :: Artificial Intelligence",
20
20
  ]
21
- # The precomputed vector needs the `embedding` field added in citadeldb 2.0.
22
- dependencies = ["citadeldb>=2.0,<3", "haystack-ai>=2.9,<4"]
21
+ # Reads the expanded AtomHit scoring fields added in citadeldb 2.2.
22
+ dependencies = ["citadeldb>=2.2,<3", "haystack-ai>=2.9,<4"]
23
23
 
24
24
  [project.optional-dependencies]
25
25
  test = ["pytest>=8", "pytest-asyncio>=0.23"]
@@ -119,10 +119,27 @@ class _HaystackEmbedder:
119
119
  return vector
120
120
 
121
121
  def embed(self, texts: list[str]) -> list[list[float]]:
122
- return [self._one(text) for text in texts]
122
+ return self.embed_with_cancel(texts, None)
123
123
 
124
124
  def embed_queries(self, texts: list[str]) -> list[list[float]]:
125
- return self.embed(texts)
125
+ return self.embed_queries_with_cancel(texts, None)
126
+
127
+ def embed_with_cancel(
128
+ self, texts: list[str], cancel_token: Any | None
129
+ ) -> list[list[float]]:
130
+ vectors: list[list[float]] = []
131
+ for text in texts:
132
+ if cancel_token is not None:
133
+ cancel_token.check()
134
+ vectors.append(self._one(text))
135
+ if cancel_token is not None:
136
+ cancel_token.check()
137
+ return vectors
138
+
139
+ def embed_queries_with_cancel(
140
+ self, texts: list[str], cancel_token: Any | None
141
+ ) -> list[list[float]]:
142
+ return self.embed_with_cancel(texts, cancel_token)
126
143
 
127
144
 
128
145
  def _fetch(mem: Any, region: str, criterion: dict[str, Any] | None) -> list[Any]:
@@ -449,7 +466,7 @@ class CitadelDocumentStore:
449
466
  if filters and not document_matches_filter(filters, doc):
450
467
  continue
451
468
  if h.distance is None:
452
- score = h.score
469
+ score = h.relevance if h.relevance is not None else 0.0
453
470
  elif self.embedding_similarity_function == "dot_product":
454
471
  score = -h.distance
455
472
  else: