summa-client-python 2.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,305 @@
1
+ """Type definitions for Summa client.
2
+
3
+ All search-related types mirror the proto API structure exactly.
4
+ Query is a dict with exactly one key matching the proto Query oneof variant.
5
+ """
6
+
7
+ from dataclasses import dataclass, field
8
+ from typing import Any, Literal, Required, TypedDict
9
+
10
+ # =============================================================================
11
+ # Multi-value score combiner (mirrors proto MultiValueCombiner)
12
+ # =============================================================================
13
+
14
+ Combiner = Literal["log_sum_exp", "max", "avg", "sum", "weighted_top_k"]
15
+
16
+ # =============================================================================
17
+ # Query types (mirrors proto Query oneof)
18
+ # =============================================================================
19
+
20
+
21
+ class TermQuery(TypedDict, total=False):
22
+ field: Required[str]
23
+ term: Required[str]
24
+ tokenizer_hint: str
25
+
26
+
27
+ class MatchQuery(TypedDict, total=False):
28
+ field: Required[str]
29
+ text: Required[str]
30
+ # Passed to the field's tokenizer; a dynamic stemmer reads it as a
31
+ # comma-separated language list ("ru,en"). Static tokenizers ignore it.
32
+ tokenizer_hint: str
33
+
34
+
35
+ class PhraseQuery(TypedDict, total=False):
36
+ field: Required[str]
37
+ text: Required[str]
38
+ slop: int
39
+ tokenizer_hint: str
40
+
41
+
42
+ class BooleanQuery(TypedDict, total=False):
43
+ must: list["Query"]
44
+ should: list["Query"]
45
+ must_not: list["Query"]
46
+
47
+
48
+ class BoostQuery(TypedDict):
49
+ query: "Query"
50
+ boost: float
51
+
52
+
53
+ class AllQuery(TypedDict):
54
+ pass
55
+
56
+
57
+ class SparseVectorQuery(TypedDict, total=False):
58
+ field: str # required but total=False for optional fields
59
+ indices: list[int]
60
+ values: list[float]
61
+ text: str
62
+ combiner: Combiner
63
+ heap_factor: float
64
+ lsp_gamma: int
65
+ combiner_temperature: float
66
+ combiner_top_k: int
67
+ combiner_decay: float
68
+ weight_threshold: float
69
+ max_query_dims: int
70
+ pruning: float
71
+ seismic_cut: int
72
+ seismic_factor: float
73
+ exhaustive: bool
74
+
75
+
76
+ class DenseVectorQuery(TypedDict, total=False):
77
+ field: str # required but total=False for optional fields
78
+ vector: list[float]
79
+ nprobe: int
80
+ combiner: Combiner
81
+ combiner_temperature: float
82
+ combiner_top_k: int
83
+ combiner_decay: float
84
+
85
+
86
+ class BinaryDenseVectorQuery(TypedDict, total=False):
87
+ field: str # required but total=False for optional fields
88
+ vector: bytes # packed-bit query vector (ceil(dim/8) bytes)
89
+ combiner: Combiner
90
+ combiner_temperature: float
91
+ combiner_top_k: int
92
+ combiner_decay: float
93
+
94
+
95
+ class RangeQuery(TypedDict, total=False):
96
+ field: str # required but total=False for optional fields
97
+ min_u64: int
98
+ max_u64: int
99
+ min_i64: int
100
+ max_i64: int
101
+ min_f64: float
102
+ max_f64: float
103
+
104
+
105
+ # Query is a dict with exactly one key matching a protobuf Query variant,
106
+ # including text, Boolean, vector, range/prefix, all, and fusion queries.
107
+ Query = dict[str, Any]
108
+
109
+ # =============================================================================
110
+ # Reranker (mirrors proto Reranker)
111
+ # =============================================================================
112
+
113
+
114
+ class Reranker(TypedDict, total=False):
115
+ field: str
116
+ vector: list[float]
117
+ combiner: Combiner
118
+ combiner_temperature: float
119
+ combiner_top_k: int
120
+ combiner_decay: float
121
+ matryoshka_dims: int
122
+ binary_vector: bytes # packed-bit query vector (for binary dense fields)
123
+ rrf_k: float # Reciprocal Rank Fusion k (0 = disabled, typical: 60)
124
+
125
+
126
+ # =============================================================================
127
+ # Response types
128
+ # =============================================================================
129
+
130
+
131
+ @dataclass
132
+ class Document:
133
+ """A document with field values."""
134
+
135
+ fields: dict[str, Any] = field(default_factory=dict)
136
+
137
+ def __getitem__(self, key: str) -> Any:
138
+ return self.fields[key]
139
+
140
+ def __setitem__(self, key: str, value: Any) -> None:
141
+ self.fields[key] = value
142
+
143
+ def get(self, key: str, default: Any = None) -> Any:
144
+ return self.fields.get(key, default)
145
+
146
+
147
+ @dataclass
148
+ class DocAddress:
149
+ """Unique document address: segment + local doc_id."""
150
+
151
+ segment_id: str
152
+ doc_id: int
153
+
154
+
155
+ @dataclass
156
+ class OrdinalScore:
157
+ """Score contribution from a specific ordinal in a multi-valued field."""
158
+
159
+ ordinal: int
160
+ score: float
161
+
162
+
163
+ @dataclass
164
+ class PassageScores:
165
+ ordinal: int
166
+ scores: dict[str, float]
167
+ l1_score: float | None = None
168
+
169
+
170
+ @dataclass
171
+ class CandidateScores:
172
+ document: dict[str, float]
173
+ passages: list[PassageScores]
174
+ scored_passages: int
175
+
176
+
177
+ @dataclass
178
+ class RrfContribution:
179
+ query_index: int
180
+ query_name: str
181
+ rank: int
182
+ score: float
183
+ ordinal: int | None = None
184
+
185
+
186
+ @dataclass
187
+ class SearchHit:
188
+ """A single search result."""
189
+
190
+ address: DocAddress
191
+ score: float
192
+ fields: dict[str, Any] = field(default_factory=dict)
193
+ ordinal_scores: list[OrdinalScore] = field(default_factory=list)
194
+ candidate_scores: CandidateScores | None = None
195
+ rrf_score: float | None = None
196
+ rrf_contributions: list[RrfContribution] = field(default_factory=list)
197
+
198
+
199
+ @dataclass
200
+ class SearchTimings:
201
+ """Detailed timing breakdown for search phases (all values in microseconds)."""
202
+
203
+ search_us: int
204
+ rerank_us: int
205
+ load_us: int
206
+ total_us: int
207
+ candidate_scoring_us: int = 0
208
+
209
+
210
+ @dataclass
211
+ class FusionCandidate:
212
+ address: DocAddress
213
+ score: float
214
+ ordinal_scores: list[OrdinalScore] = field(default_factory=list)
215
+
216
+
217
+ @dataclass
218
+ class FusionCandidateList:
219
+ query_index: int
220
+ candidates: list[FusionCandidate] = field(default_factory=list)
221
+
222
+
223
+ @dataclass
224
+ class QueryTrace:
225
+ query_index: int
226
+ query_name: str
227
+ query: dict[str, Any]
228
+ scope: int
229
+ score_only: bool
230
+ candidate_depth: int
231
+ total_seen: int
232
+ candidates: list[FusionCandidate] = field(default_factory=list)
233
+
234
+
235
+ @dataclass
236
+ class ShardSearchTrace:
237
+ shard_id: str
238
+ backend_id: str
239
+ index_name: str
240
+ queries: list[QueryTrace]
241
+ selected: list[FusionCandidate]
242
+ ranking_method: str
243
+ truncated: bool
244
+ filters: list[dict[str, Any]] = field(default_factory=list)
245
+
246
+
247
+ @dataclass
248
+ class SearchTrace:
249
+ shards: list[ShardSearchTrace] = field(default_factory=list)
250
+
251
+
252
+ @dataclass
253
+ class SearchResponse:
254
+ """Search response with hits and metadata."""
255
+
256
+ hits: list[SearchHit]
257
+ total_hits: int
258
+ took_ms: int
259
+ timings: SearchTimings | None = None
260
+ ranking_method: str = ""
261
+ seeded_document_passages: bool = False
262
+ truncated: bool = False
263
+ fusion_candidates: list[FusionCandidateList] = field(default_factory=list)
264
+ trace: SearchTrace | None = None
265
+
266
+
267
+ @dataclass
268
+ class VectorFieldStats:
269
+ """Per-field vector statistics."""
270
+
271
+ field_name: str
272
+ vector_type: str # "dense" or "sparse"
273
+ total_vectors: int
274
+ dimension: int
275
+
276
+
277
+ @dataclass
278
+ class IndexInfo:
279
+ """Information about an index."""
280
+
281
+ index_name: str
282
+ num_docs: int
283
+ num_segments: int
284
+ schema: str
285
+ vector_stats: list[VectorFieldStats] = field(default_factory=list)
286
+ physical_num_docs: int = 0
287
+ num_deleted_docs: int = 0
288
+ deleted_ratio: float = 0.0
289
+ candidate_scoring_version: int = 0
290
+ unprepared_candidate_fields: list[str] = field(default_factory=list)
291
+
292
+
293
+ class DocumentMutationError(TypedDict):
294
+ """Rejected operation at a zero-based position in the input batch."""
295
+
296
+ index: int
297
+ error: str
298
+
299
+
300
+ @dataclass
301
+ class DocumentMutationResult:
302
+ """Staged operations (not affected rows); commit publishes accepted work."""
303
+
304
+ accepted_count: int
305
+ errors: list[DocumentMutationError] = field(default_factory=list)
@@ -0,0 +1,193 @@
1
+ Metadata-Version: 2.5
2
+ Name: summa-client-python
3
+ Version: 2.0.0
4
+ Summary: Async Python client for Summa search server
5
+ Project-URL: Homepage, https://github.com/SpaceFrontiers/summa
6
+ Project-URL: Repository, https://github.com/SpaceFrontiers/summa
7
+ Author: izihawa
8
+ License-Expression: MIT
9
+ Keywords: async,full-text-search,grpc,search
10
+ Classifier: Development Status :: 4 - Beta
11
+ Classifier: Intended Audience :: Developers
12
+ Classifier: License :: OSI Approved :: MIT License
13
+ Classifier: Programming Language :: Python :: 3
14
+ Classifier: Programming Language :: Python :: 3.10
15
+ Classifier: Programming Language :: Python :: 3.11
16
+ Classifier: Programming Language :: Python :: 3.12
17
+ Classifier: Programming Language :: Python :: 3.13
18
+ Classifier: Topic :: Database :: Database Engines/Servers
19
+ Classifier: Topic :: Text Processing :: Indexing
20
+ Requires-Python: >=3.10
21
+ Requires-Dist: grpcio>=1.76.0
22
+ Requires-Dist: protobuf>=6.33.4
23
+ Description-Content-Type: text/markdown
24
+
25
+ # Summa Python client
26
+
27
+ Async Python client for the
28
+ [Summa](https://github.com/SpaceFrontiers/summa) gRPC search server.
29
+
30
+ ## Installation
31
+
32
+ ```bash
33
+ pip install summa-client-python
34
+ ```
35
+
36
+ Python 3.10 or newer is required.
37
+
38
+ ## Quick start
39
+
40
+ ```python
41
+ import asyncio
42
+
43
+ from summa_client_python import SummaClient
44
+
45
+
46
+ async def main():
47
+ async with SummaClient("localhost:50051") as client:
48
+ await client.create_index(
49
+ "articles",
50
+ """
51
+ index articles {
52
+ field id: text<raw> [primary, stored]
53
+ field title: text<simple> [indexed, stored]
54
+ field body: text<simple> [indexed, stored]
55
+ }
56
+ """,
57
+ )
58
+
59
+ indexed, error_count, errors = await client.index_documents(
60
+ "articles",
61
+ [
62
+ {"id": "1", "title": "Hello World", "body": "First article"},
63
+ {"id": "2", "title": "Summa Search", "body": "Fast retrieval"},
64
+ ],
65
+ )
66
+ if error_count:
67
+ raise RuntimeError(errors)
68
+ print(f"Indexed {indexed} documents")
69
+
70
+ await client.commit("articles")
71
+
72
+ results = await client.search(
73
+ "articles",
74
+ query={"match": {"field": "title", "text": "hello"}},
75
+ fields_to_load=["title", "body"],
76
+ )
77
+ for hit in results.hits:
78
+ print(hit.address, hit.score, hit.fields)
79
+
80
+ if results.hits:
81
+ document = await client.get_document("articles", results.hits[0].address)
82
+ print(document.fields if document else "document not found")
83
+
84
+
85
+ asyncio.run(main())
86
+ ```
87
+
88
+ The context manager connects and closes the channel. Manual callers use
89
+ `await client.connect()` / `await client.close()`.
90
+
91
+ ## Index management
92
+
93
+ ```python
94
+ await client.list_indexes()
95
+ info = await client.get_index_info("articles")
96
+ await client.reorder("articles")
97
+ await client.retrain_vector_index("articles")
98
+ ```
99
+
100
+ `index_documents` returns `(indexed_count, error_count, errors)`;
101
+ `index_documents_stream` takes an async iterable and returns counts. Inspect
102
+ errors, then `commit()` to publish accepted work. Repeated values are lists;
103
+ flat numeric lists are dense vectors, `(dimension, weight)` pairs are sparse
104
+ vectors. See [client types](src/summa_client_python/types.py).
105
+
106
+ ### Delete and upsert documents
107
+
108
+ With a text primary key, delete by exact key or supply a complete replacement:
109
+
110
+ ```python
111
+ await client.delete_document("articles", "2")
112
+ await client.upsert_document("articles", {"id": "1", "title": "Updated title"})
113
+ await client.commit("articles")
114
+ ```
115
+
116
+ `delete_documents` / `upsert_documents` return `DocumentMutationResult` with
117
+ `accepted_count` and `errors: [{index, error}]`. Single-item helpers raise on
118
+ rejection. Missing deletes succeed; upserts insert missing keys. Staged rows can
119
+ be replaced/deleted again before commit; the latest accepted version wins.
120
+ Optional [content hashes](../docs/content-deduplication.md) skip unchanged writes.
121
+
122
+ Limits: 100,000 deletion keys / 8 MiB key bytes; 1,000 replacements / 32 MiB
123
+ encoded protobuf, or one replacement / 200 MiB including the request envelope.
124
+ Broker commits are atomic per partition. See [mutation semantics](../docs/row-deletion.md).
125
+
126
+ ### Compact deleted rows
127
+
128
+ ```python
129
+ await client.force_merge("articles") # Retain tombstones.
130
+ await client.force_merge("articles", compact=True) # Physically remove deleted rows.
131
+ ```
132
+
133
+ Compaction handles singleton segments and may change addresses and BM25 scores.
134
+ Index info exposes `num_docs`, `physical_num_docs`, `num_deleted_docs`, and
135
+ `deleted_ratio`. Use primary keys for durable identity.
136
+
137
+ ## Searching
138
+
139
+ `search(index_name, query=..., limit=10, fields_to_load=[...])` accepts one query
140
+ variant: `term`, `match`, `phrase`, `boolean`, `sparse_vector`, `dense_vector`,
141
+ `binary_dense_vector`, `boost`, `range`, `prefix`, `all`, or `fusion`.
142
+ See [query types](src/summa_client_python/types.py) and the
143
+ [wire contract](../summa-proto/summa.proto) for options.
144
+
145
+ ```python
146
+ results = await client.search(
147
+ "articles",
148
+ query={
149
+ "boolean": {
150
+ "must": [{"match": {"field": "title", "text": "search"}}],
151
+ "must_not": [{"term": {"field": "title", "term": "draft"}}],
152
+ }
153
+ },
154
+ fields_to_load=["title"],
155
+ )
156
+ ```
157
+
158
+ `get_document(index_name, hit.address)` uses the full segment/document address
159
+ and returns `None` on `NOT_FOUND`.
160
+
161
+ ## Ranking diagnostics and recall traces
162
+
163
+ Search options `include_rrf_scores=True` and `tracing=True` default to false.
164
+ RRF diagnostics describe organic branch nominations; traces retain bounded
165
+ candidates and query trees, including hits outside the final page. Neither
166
+ changes retrieval depth or ranking. Oversized exports fail explicitly.
167
+
168
+ For named branches, use `l1={"formula": "0.2 * title + 0.8 * body + 3 * rrf"}`.
169
+ The formula is the only L1 scoring interface; old coefficient fields are removed.
170
+ See [candidate scoring](../docs/candidate-rescoring.md) for backfill, passage
171
+ selection, expression limits, capability versions, and distributed behavior.
172
+
173
+ ## Deadlines and errors
174
+
175
+ RPCs accept `timeout` in seconds, overriding the constructor's `default_timeout`.
176
+ gRPC failures raise `grpc.aio.AioRpcError`, except document `NOT_FOUND` as above.
177
+ An expired mutation may already be staged, and an accepted commit continues
178
+ after disconnection. Resolve the outcome before retrying replacements.
179
+
180
+ ## Development
181
+
182
+ From this directory:
183
+
184
+ ```bash
185
+ uv sync --group dev --group test
186
+ uv run ruff check .
187
+ uv run ruff format --check .
188
+ uv run pytest tests/test_client_unit.py
189
+ uv run --group dev python generate_proto.py
190
+ ```
191
+
192
+ Integration tests require `target/debug/summa-server`. Regenerate bindings after
193
+ [protocol changes](../summa-proto/README.md#regeneration-and-validation).
@@ -0,0 +1,8 @@
1
+ summa_client_python/__init__.py,sha256=kfG49vFn87L6nIxeJ25QI2jum9cBhZ1pIVEtfHHFe-k,1561
2
+ summa_client_python/client.py,sha256=YUjdGJMvfXmqGw_hNKW1wJnbNuWvqW6WqtBO_5ci6Mg,42956
3
+ summa_client_python/summa_pb2.py,sha256=1ZScK1c7ewzeNe93DyaRq-nUqhyUtZlBrN0d3LUOk0M,30881
4
+ summa_client_python/summa_pb2_grpc.py,sha256=DBvzr2NKYukpdQhF7rW58Xnd5zy9TfuanoGedy_9lEQ,28792
5
+ summa_client_python/types.py,sha256=FPCDsWYl8usGM7IAXeQURSEl3GaRMgi4KJ794rN__gY,7588
6
+ summa_client_python-2.0.0.dist-info/METADATA,sha256=Xp827Dg8mZOU5R49uwNn1kVQU9mEsNRmiWx5hTFmA0Y,6719
7
+ summa_client_python-2.0.0.dist-info/WHEEL,sha256=W3fkpkm7-wf9vBI5Z-7s0eWkeM-spu78I8Neb98DeEg,87
8
+ summa_client_python-2.0.0.dist-info/RECORD,,
@@ -0,0 +1,4 @@
1
+ Wheel-Version: 1.0
2
+ Generator: hatchling 1.32.4
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any