docforge-sdk 0.2.0__tar.gz → 0.4.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/PKG-INFO +24 -2
- {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/README.md +23 -1
- {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/docforge_sdk/__init__.py +15 -0
- {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/docforge_sdk/_version.py +1 -1
- {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/docforge_sdk/models/__init__.py +15 -0
- {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/docforge_sdk/models/collections.py +30 -0
- docforge_sdk-0.4.0/docforge_sdk/models/storage.py +149 -0
- {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/docforge_sdk/resources/collections.py +39 -0
- docforge_sdk-0.4.0/tests/openapi_snapshot.json +6807 -0
- {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/tests/parity_map.py +13 -0
- {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/tests/unit/test_collections.py +69 -0
- docforge_sdk-0.2.0/tests/openapi_snapshot.json +0 -1
- {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/.gitignore +0 -0
- {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/.python-version +0 -0
- {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/LICENSE +0 -0
- {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/docforge_sdk/_exceptions.py +0 -0
- {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/docforge_sdk/_requestspec.py +0 -0
- {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/docforge_sdk/_transport_async.py +0 -0
- {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/docforge_sdk/_transport_base.py +0 -0
- {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/docforge_sdk/_transport_sync.py +0 -0
- {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/docforge_sdk/client.py +0 -0
- {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/docforge_sdk/models/_shared.py +0 -0
- {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/docforge_sdk/models/auth.py +0 -0
- {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/docforge_sdk/models/blobs.py +0 -0
- {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/docforge_sdk/models/documents.py +0 -0
- {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/docforge_sdk/models/explorer.py +0 -0
- {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/docforge_sdk/models/health.py +0 -0
- {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/docforge_sdk/models/ir.py +0 -0
- {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/docforge_sdk/models/jobs.py +0 -0
- {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/docforge_sdk/models/pipelines.py +0 -0
- {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/docforge_sdk/models/search.py +0 -0
- {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/docforge_sdk/py.typed +0 -0
- {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/docforge_sdk/resources/__init__.py +0 -0
- {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/docforge_sdk/resources/_base.py +0 -0
- {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/docforge_sdk/resources/auth.py +0 -0
- {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/docforge_sdk/resources/blobs.py +0 -0
- {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/docforge_sdk/resources/documents.py +0 -0
- {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/docforge_sdk/resources/explorer.py +0 -0
- {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/docforge_sdk/resources/health.py +0 -0
- {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/docforge_sdk/resources/jobs.py +0 -0
- {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/docforge_sdk/resources/pipelines.py +0 -0
- {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/docforge_sdk/resources/search.py +0 -0
- {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/pyproject.toml +0 -0
- {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/tests/__init__.py +0 -0
- {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/tests/check_schema_drift.py +0 -0
- {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/tests/live/__init__.py +0 -0
- {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/tests/live/test_openapi_parity_live.py +0 -0
- {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/tests/unit/__init__.py +0 -0
- {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/tests/unit/test_auth.py +0 -0
- {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/tests/unit/test_blobs.py +0 -0
- {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/tests/unit/test_documents.py +0 -0
- {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/tests/unit/test_explorer.py +0 -0
- {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/tests/unit/test_health.py +0 -0
- {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/tests/unit/test_jobs.py +0 -0
- {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/tests/unit/test_models_offline_parity.py +0 -0
- {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/tests/unit/test_pipelines.py +0 -0
- {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/tests/unit/test_resource_parity.py +0 -0
- {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/tests/unit/test_search.py +0 -0
- {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/tests/unit/test_transport.py +0 -0
- {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/uv.lock +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: docforge-sdk
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.4.0
|
|
4
4
|
Summary: Typed async + sync Python client for the DocForge REST API (standalone, zero server deps).
|
|
5
5
|
Project-URL: Homepage, https://github.com/Florian-BARRE/docforge
|
|
6
6
|
Project-URL: Repository, https://github.com/Florian-BARRE/docforge
|
|
@@ -189,6 +189,7 @@ are identical.
|
|
|
189
189
|
| `create(CreateCollectionRequest)` | `CollectionModel` | Create a collection (contract). |
|
|
190
190
|
| `update(collection_id, UpdateCollectionRequest)` | `CollectionModel` | Patch name / formats / fields / pipelines. |
|
|
191
191
|
| `delete(collection_id)` | `None` | Delete a collection. |
|
|
192
|
+
| `storage(collection_id)` | `CollectionStorageResponse` | Material storage footprint per store (S3 exact, Postgres/Qdrant estimated) + per-document breakdown. |
|
|
192
193
|
|
|
193
194
|
### `documents`
|
|
194
195
|
| Method | Returns | Purpose |
|
|
@@ -271,6 +272,12 @@ with Client("http://localhost:10040", api_token="df_...") as client:
|
|
|
271
272
|
print(collection.id)
|
|
272
273
|
```
|
|
273
274
|
|
|
275
|
+
Key `CreateCollectionRequest` fields beyond the schema: `max_file_size_bytes` (bytes) and
|
|
276
|
+
`job_timeout_seconds` (`float | None`, seconds) — the whole-ingest-job wall-clock budget for that
|
|
277
|
+
collection; `None` (the default) inherits the worker's global job-timeout default. Same field, same
|
|
278
|
+
semantics on `CollectionModel` (read) and `UpdateCollectionRequest` (write; there, omitting it
|
|
279
|
+
leaves the current value unchanged, a set value overrides it).
|
|
280
|
+
|
|
274
281
|
### Upload a document and wait for ingestion
|
|
275
282
|
|
|
276
283
|
`upload` returns immediately with a `job_id`; ingestion runs asynchronously on the worker. Poll the
|
|
@@ -338,6 +345,20 @@ with Client("http://localhost:10040", api_token="df_...") as client:
|
|
|
338
345
|
client.explorer.set_chunk_enabled(chunks[0].chunk_id, enabled=False)
|
|
339
346
|
```
|
|
340
347
|
|
|
348
|
+
### Measure a collection's storage footprint
|
|
349
|
+
|
|
350
|
+
`storage` reports the material footprint per store — S3 bytes are exact (deduped), Postgres/Qdrant
|
|
351
|
+
bytes are estimates (each section flags this via its own `estimated`) — plus a per-document
|
|
352
|
+
breakdown sorted heaviest first.
|
|
353
|
+
|
|
354
|
+
```python
|
|
355
|
+
with Client("http://localhost:10040", api_token="df_...") as client:
|
|
356
|
+
footprint = client.collections.storage(collection_id)
|
|
357
|
+
print(footprint.grand_total_bytes, footprint.s3.physical_unique_bytes)
|
|
358
|
+
for doc in footprint.documents[:5]:
|
|
359
|
+
print(doc.filename, doc.total_bytes)
|
|
360
|
+
```
|
|
361
|
+
|
|
341
362
|
### Manage API keys
|
|
342
363
|
|
|
343
364
|
Mint a scoped, expiring key (the plaintext is shown **once**, on creation):
|
|
@@ -424,6 +445,7 @@ from docforge_sdk import (
|
|
|
424
445
|
UpdateCollectionRequest,
|
|
425
446
|
FieldSpec,
|
|
426
447
|
FieldType,
|
|
448
|
+
CollectionStorageResponse,
|
|
427
449
|
# documents / explorer
|
|
428
450
|
UploadAccepted,
|
|
429
451
|
DocumentDetail,
|
|
@@ -465,7 +487,7 @@ server's OpenAPI on every change, so a published version is coherent with the AP
|
|
|
465
487
|
version in production:
|
|
466
488
|
|
|
467
489
|
```bash
|
|
468
|
-
pip install "docforge-sdk==0.
|
|
490
|
+
pip install "docforge-sdk==0.3.0"
|
|
469
491
|
```
|
|
470
492
|
|
|
471
493
|
## License
|
|
@@ -166,6 +166,7 @@ are identical.
|
|
|
166
166
|
| `create(CreateCollectionRequest)` | `CollectionModel` | Create a collection (contract). |
|
|
167
167
|
| `update(collection_id, UpdateCollectionRequest)` | `CollectionModel` | Patch name / formats / fields / pipelines. |
|
|
168
168
|
| `delete(collection_id)` | `None` | Delete a collection. |
|
|
169
|
+
| `storage(collection_id)` | `CollectionStorageResponse` | Material storage footprint per store (S3 exact, Postgres/Qdrant estimated) + per-document breakdown. |
|
|
169
170
|
|
|
170
171
|
### `documents`
|
|
171
172
|
| Method | Returns | Purpose |
|
|
@@ -248,6 +249,12 @@ with Client("http://localhost:10040", api_token="df_...") as client:
|
|
|
248
249
|
print(collection.id)
|
|
249
250
|
```
|
|
250
251
|
|
|
252
|
+
Key `CreateCollectionRequest` fields beyond the schema: `max_file_size_bytes` (bytes) and
|
|
253
|
+
`job_timeout_seconds` (`float | None`, seconds) — the whole-ingest-job wall-clock budget for that
|
|
254
|
+
collection; `None` (the default) inherits the worker's global job-timeout default. Same field, same
|
|
255
|
+
semantics on `CollectionModel` (read) and `UpdateCollectionRequest` (write; there, omitting it
|
|
256
|
+
leaves the current value unchanged, a set value overrides it).
|
|
257
|
+
|
|
251
258
|
### Upload a document and wait for ingestion
|
|
252
259
|
|
|
253
260
|
`upload` returns immediately with a `job_id`; ingestion runs asynchronously on the worker. Poll the
|
|
@@ -315,6 +322,20 @@ with Client("http://localhost:10040", api_token="df_...") as client:
|
|
|
315
322
|
client.explorer.set_chunk_enabled(chunks[0].chunk_id, enabled=False)
|
|
316
323
|
```
|
|
317
324
|
|
|
325
|
+
### Measure a collection's storage footprint
|
|
326
|
+
|
|
327
|
+
`storage` reports the material footprint per store — S3 bytes are exact (deduped), Postgres/Qdrant
|
|
328
|
+
bytes are estimates (each section flags this via its own `estimated`) — plus a per-document
|
|
329
|
+
breakdown sorted heaviest first.
|
|
330
|
+
|
|
331
|
+
```python
|
|
332
|
+
with Client("http://localhost:10040", api_token="df_...") as client:
|
|
333
|
+
footprint = client.collections.storage(collection_id)
|
|
334
|
+
print(footprint.grand_total_bytes, footprint.s3.physical_unique_bytes)
|
|
335
|
+
for doc in footprint.documents[:5]:
|
|
336
|
+
print(doc.filename, doc.total_bytes)
|
|
337
|
+
```
|
|
338
|
+
|
|
318
339
|
### Manage API keys
|
|
319
340
|
|
|
320
341
|
Mint a scoped, expiring key (the plaintext is shown **once**, on creation):
|
|
@@ -401,6 +422,7 @@ from docforge_sdk import (
|
|
|
401
422
|
UpdateCollectionRequest,
|
|
402
423
|
FieldSpec,
|
|
403
424
|
FieldType,
|
|
425
|
+
CollectionStorageResponse,
|
|
404
426
|
# documents / explorer
|
|
405
427
|
UploadAccepted,
|
|
406
428
|
DocumentDetail,
|
|
@@ -442,7 +464,7 @@ server's OpenAPI on every change, so a published version is coherent with the AP
|
|
|
442
464
|
version in production:
|
|
443
465
|
|
|
444
466
|
```bash
|
|
445
|
-
pip install "docforge-sdk==0.
|
|
467
|
+
pip install "docforge-sdk==0.3.0"
|
|
446
468
|
```
|
|
447
469
|
|
|
448
470
|
## License
|
|
@@ -80,6 +80,15 @@ from .models.pipelines import (
|
|
|
80
80
|
# ------------------- Search models ------------------- #
|
|
81
81
|
from .models.search import BlockLocation, SearchHit, SearchRequest, SearchResponse, SearchTarget
|
|
82
82
|
|
|
83
|
+
# ------------------- Storage models ------------------- #
|
|
84
|
+
from .models.storage import (
|
|
85
|
+
CollectionStorageResponse,
|
|
86
|
+
DocumentStorageModel,
|
|
87
|
+
PostgresFootprintModel,
|
|
88
|
+
QdrantFootprintModel,
|
|
89
|
+
S3FootprintModel,
|
|
90
|
+
)
|
|
91
|
+
|
|
83
92
|
# ------------------- Public API ------------------- #
|
|
84
93
|
__all__ = [
|
|
85
94
|
"__version__",
|
|
@@ -150,6 +159,12 @@ __all__ = [
|
|
|
150
159
|
"EditResponse",
|
|
151
160
|
"StageViewResponse",
|
|
152
161
|
"StageApplyResponse",
|
|
162
|
+
# Storage
|
|
163
|
+
"S3FootprintModel",
|
|
164
|
+
"PostgresFootprintModel",
|
|
165
|
+
"QdrantFootprintModel",
|
|
166
|
+
"DocumentStorageModel",
|
|
167
|
+
"CollectionStorageResponse",
|
|
153
168
|
# Exceptions
|
|
154
169
|
"DocForgeError",
|
|
155
170
|
"APIConnectionError",
|
|
@@ -64,6 +64,15 @@ from .pipelines import (
|
|
|
64
64
|
# ------------------- Search models ------------------- #
|
|
65
65
|
from .search import BlockLocation, SearchHit, SearchRequest, SearchResponse, SearchTarget
|
|
66
66
|
|
|
67
|
+
# ------------------- Storage models ------------------- #
|
|
68
|
+
from .storage import (
|
|
69
|
+
CollectionStorageResponse,
|
|
70
|
+
DocumentStorageModel,
|
|
71
|
+
PostgresFootprintModel,
|
|
72
|
+
QdrantFootprintModel,
|
|
73
|
+
S3FootprintModel,
|
|
74
|
+
)
|
|
75
|
+
|
|
67
76
|
# ------------------- Public API ------------------- #
|
|
68
77
|
__all__ = [
|
|
69
78
|
# Shared vocabulary
|
|
@@ -130,4 +139,10 @@ __all__ = [
|
|
|
130
139
|
"EditResponse",
|
|
131
140
|
"StageViewResponse",
|
|
132
141
|
"StageApplyResponse",
|
|
142
|
+
# Storage
|
|
143
|
+
"S3FootprintModel",
|
|
144
|
+
"PostgresFootprintModel",
|
|
145
|
+
"QdrantFootprintModel",
|
|
146
|
+
"DocumentStorageModel",
|
|
147
|
+
"CollectionStorageResponse",
|
|
133
148
|
]
|
|
@@ -56,6 +56,8 @@ class CollectionModel(BaseModel):
|
|
|
56
56
|
name (str): Unique human name.
|
|
57
57
|
supported_formats (list[str]): Accepted upload extensions (e.g. pdf).
|
|
58
58
|
max_file_size_bytes (int): Upload size ceiling, bytes.
|
|
59
|
+
job_timeout_seconds (float | None): Per-collection whole-ingest-job wall-clock budget,
|
|
60
|
+
seconds. None = inherit the worker's global WORKER_JOB_TIMEOUT_SECONDS default.
|
|
59
61
|
needs_reindex (bool): True when a config change requires reindexing.
|
|
60
62
|
created_at (datetime | None): Creation timestamp.
|
|
61
63
|
pipeline (dict[str, Any]): The ingestion pipeline blob (the graph).
|
|
@@ -67,6 +69,14 @@ class CollectionModel(BaseModel):
|
|
|
67
69
|
name: str = Field(description="Unique human name.")
|
|
68
70
|
supported_formats: list[str] = Field(description="Accepted upload extensions (e.g. pdf).")
|
|
69
71
|
max_file_size_bytes: int = Field(description="Upload size ceiling, bytes.")
|
|
72
|
+
job_timeout_seconds: float | None = Field(
|
|
73
|
+
default=None,
|
|
74
|
+
gt=0,
|
|
75
|
+
description=(
|
|
76
|
+
"Per-collection whole-ingest-job wall-clock budget, seconds. None = inherit the "
|
|
77
|
+
"worker's global WORKER_JOB_TIMEOUT_SECONDS default."
|
|
78
|
+
),
|
|
79
|
+
)
|
|
70
80
|
needs_reindex: bool = Field(description="True when a config change requires reindexing.")
|
|
71
81
|
created_at: datetime | None = Field(default=None, description="Creation timestamp.")
|
|
72
82
|
pipeline: dict[str, Any] = Field(description="The ingestion pipeline blob (the graph).")
|
|
@@ -84,6 +94,8 @@ class CreateCollectionRequest(BaseModel):
|
|
|
84
94
|
name (str): Unique human name.
|
|
85
95
|
supported_formats (list[str]): Accepted upload extensions (e.g. pdf).
|
|
86
96
|
max_file_size_bytes (int): Upload size ceiling, bytes.
|
|
97
|
+
job_timeout_seconds (float | None): Per-collection whole-ingest-job wall-clock budget,
|
|
98
|
+
seconds. None = inherit the worker's global WORKER_JOB_TIMEOUT_SECONDS default.
|
|
87
99
|
fields (list[FieldSpec]): The FULL schema, declared up front (vector space is fixed).
|
|
88
100
|
pipeline (dict[str, Any] | None): The pipeline blob; omitted → the product default.
|
|
89
101
|
"""
|
|
@@ -91,6 +103,14 @@ class CreateCollectionRequest(BaseModel):
|
|
|
91
103
|
name: str = Field(description="Unique human name.")
|
|
92
104
|
supported_formats: list[str] = Field(description="Accepted upload extensions (e.g. pdf).")
|
|
93
105
|
max_file_size_bytes: int = Field(description="Upload size ceiling, bytes.")
|
|
106
|
+
job_timeout_seconds: float | None = Field(
|
|
107
|
+
default=None,
|
|
108
|
+
gt=0,
|
|
109
|
+
description=(
|
|
110
|
+
"Per-collection whole-ingest-job wall-clock budget, seconds. None = inherit the "
|
|
111
|
+
"worker's global WORKER_JOB_TIMEOUT_SECONDS default."
|
|
112
|
+
),
|
|
113
|
+
)
|
|
94
114
|
fields: list[FieldSpec] = Field(
|
|
95
115
|
default_factory=list,
|
|
96
116
|
description="The FULL schema, declared up front (vector space is fixed at creation).",
|
|
@@ -114,6 +134,8 @@ class UpdateCollectionRequest(BaseModel):
|
|
|
114
134
|
name (str | None): New unique name.
|
|
115
135
|
supported_formats (list[str] | None): New accepted upload extensions.
|
|
116
136
|
max_file_size_bytes (int | None): New size ceiling, bytes.
|
|
137
|
+
job_timeout_seconds (float | None): New per-collection whole-ingest-job wall-clock
|
|
138
|
+
budget, seconds. Omitted = leave the current value unchanged.
|
|
117
139
|
fields (list[FieldSpec] | None): The TARGET schema (diffed by field name).
|
|
118
140
|
pipeline (dict[str, Any] | None): New pipeline blob (validated before storage).
|
|
119
141
|
search (dict[str, Any] | None): New search graph blob ({} = stock default).
|
|
@@ -125,6 +147,14 @@ class UpdateCollectionRequest(BaseModel):
|
|
|
125
147
|
default=None, description="New accepted upload extensions."
|
|
126
148
|
)
|
|
127
149
|
max_file_size_bytes: int | None = Field(default=None, description="New size ceiling, bytes.")
|
|
150
|
+
job_timeout_seconds: float | None = Field(
|
|
151
|
+
default=None,
|
|
152
|
+
gt=0,
|
|
153
|
+
description=(
|
|
154
|
+
"New per-collection whole-ingest-job wall-clock budget, seconds. Omitted = leave the "
|
|
155
|
+
"current value unchanged; a set value overrides the global WORKER_JOB_TIMEOUT_SECONDS."
|
|
156
|
+
),
|
|
157
|
+
)
|
|
128
158
|
fields: list[FieldSpec] | None = Field(
|
|
129
159
|
default=None,
|
|
130
160
|
description="The TARGET schema (diffed by field name; omitted fields are removed).",
|
|
@@ -0,0 +1,149 @@
|
|
|
1
|
+
# ====== Code Summary ======
|
|
2
|
+
# Response models for the collection storage-footprint endpoint, mirrored field-for-field from the
|
|
3
|
+
# DocForge backend router models (S3FootprintModel / PostgresFootprintModel / QdrantFootprintModel /
|
|
4
|
+
# DocumentStorageModel / CollectionStorageResponse). S3 bytes are EXACT; Postgres and Qdrant bytes are
|
|
5
|
+
# ESTIMATES — each section carries its own ``estimated`` flag.
|
|
6
|
+
|
|
7
|
+
# ====== Third-Party Library Imports ======
|
|
8
|
+
from pydantic import BaseModel, Field
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
class S3FootprintModel(BaseModel):
|
|
12
|
+
"""
|
|
13
|
+
EXACT S3 bytes from the content-addressed blob registry (``estimated`` is always false).
|
|
14
|
+
|
|
15
|
+
Attributes:
|
|
16
|
+
original_bytes (int): Uploaded source file bytes.
|
|
17
|
+
rendered_bytes (int): Derived-blob bytes (canonical PDF, page renders, crops).
|
|
18
|
+
total_bytes (int): Logical bytes (original + rendered).
|
|
19
|
+
physical_unique_bytes (int): Deduped disk cost — a blob shared across documents counts
|
|
20
|
+
once (<= total).
|
|
21
|
+
estimated (bool): Always false: S3 bytes are measured exactly.
|
|
22
|
+
"""
|
|
23
|
+
|
|
24
|
+
original_bytes: int = Field(description="Uploaded source file bytes.")
|
|
25
|
+
rendered_bytes: int = Field(
|
|
26
|
+
description="Derived-blob bytes (canonical PDF, page renders, crops)."
|
|
27
|
+
)
|
|
28
|
+
total_bytes: int = Field(description="Logical bytes (original + rendered).")
|
|
29
|
+
physical_unique_bytes: int = Field(
|
|
30
|
+
description="Deduped disk cost — a blob shared across documents counts once (<= total)."
|
|
31
|
+
)
|
|
32
|
+
estimated: bool = Field(description="Always false: S3 bytes are measured exactly.")
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
class PostgresFootprintModel(BaseModel):
|
|
36
|
+
"""
|
|
37
|
+
ESTIMATED Postgres row bytes via ``pg_column_size`` (excludes index/TOAST/bloat).
|
|
38
|
+
|
|
39
|
+
Attributes:
|
|
40
|
+
documents_bytes (int): ``document`` + ``page`` rows.
|
|
41
|
+
ir_blocks_bytes (int): ``block`` + ``block_table`` + ``block_figure`` rows.
|
|
42
|
+
enrichment_bytes (int): ``block_enrichment`` + ``enrichment_attempt`` rows.
|
|
43
|
+
chunks_bytes (int): ``chunk`` + ``chunk_block`` + ``chunk_metadata`` + ``entity_mention`` rows.
|
|
44
|
+
metadata_bytes (int): ``document_metadata`` rows.
|
|
45
|
+
observability_bytes (int): ``job`` + ``job_stage_event`` rows.
|
|
46
|
+
total_bytes (int): Sum of every bucket.
|
|
47
|
+
estimated (bool): Always true: real row bytes, no index/TOAST/bloat.
|
|
48
|
+
"""
|
|
49
|
+
|
|
50
|
+
documents_bytes: int = Field(description="``document`` + ``page`` rows.")
|
|
51
|
+
ir_blocks_bytes: int = Field(description="``block`` + ``block_table`` + ``block_figure`` rows.")
|
|
52
|
+
enrichment_bytes: int = Field(description="``block_enrichment`` + ``enrichment_attempt`` rows.")
|
|
53
|
+
chunks_bytes: int = Field(
|
|
54
|
+
description="``chunk`` + ``chunk_block`` + ``chunk_metadata`` + ``entity_mention`` rows."
|
|
55
|
+
)
|
|
56
|
+
metadata_bytes: int = Field(description="``document_metadata`` rows.")
|
|
57
|
+
observability_bytes: int = Field(description="``job`` + ``job_stage_event`` rows.")
|
|
58
|
+
total_bytes: int = Field(description="Sum of every bucket.")
|
|
59
|
+
estimated: bool = Field(description="Always true: real row bytes, no index/TOAST/bloat.")
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
class QdrantFootprintModel(BaseModel):
|
|
63
|
+
"""
|
|
64
|
+
ESTIMATED vector-store bytes (points x declared shape — excludes HNSW index overhead).
|
|
65
|
+
|
|
66
|
+
Attributes:
|
|
67
|
+
points (int): Point count (collection total, or a document's points).
|
|
68
|
+
dense_bytes (int): On-disk float32 dense bytes, summed per named vector weighted by its
|
|
69
|
+
carrier count. Excludes the int8 quantized RAM-resident copy — this is disk, not RAM.
|
|
70
|
+
sparse_bytes (int): ``points * avg_sparse_entries * 8`` (int32 index + float32 value).
|
|
71
|
+
payload_bytes (int): ``points * avg_payload_json_bytes``.
|
|
72
|
+
total_bytes (int): Sum of dense + sparse + payload.
|
|
73
|
+
estimated (bool): Always true: count-based, excludes index overhead.
|
|
74
|
+
"""
|
|
75
|
+
|
|
76
|
+
points: int = Field(description="Point count (collection total, or a document's points).")
|
|
77
|
+
dense_bytes: int = Field(
|
|
78
|
+
description=(
|
|
79
|
+
"On-disk float32 dense bytes, summed per named vector weighted by its carrier count "
|
|
80
|
+
"(content_dense on every point, each meta vector only on its field's documents). "
|
|
81
|
+
"Excludes the int8 quantized RAM-resident copy — this is disk, not RAM."
|
|
82
|
+
)
|
|
83
|
+
)
|
|
84
|
+
sparse_bytes: int = Field(
|
|
85
|
+
description="``points x avg_sparse_entries x 8`` (int32 index + float32 value)."
|
|
86
|
+
)
|
|
87
|
+
payload_bytes: int = Field(description="``points x avg_payload_json_bytes``.")
|
|
88
|
+
total_bytes: int = Field(description="Sum of dense + sparse + payload.")
|
|
89
|
+
estimated: bool = Field(description="Always true: count-based, excludes index overhead.")
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
class DocumentStorageModel(BaseModel):
|
|
93
|
+
"""
|
|
94
|
+
One document's footprint across the three stores.
|
|
95
|
+
|
|
96
|
+
Attributes:
|
|
97
|
+
document_id (str): The document's UUID.
|
|
98
|
+
filename (str): The document's display name.
|
|
99
|
+
s3 (S3FootprintModel): EXACT S3 bytes.
|
|
100
|
+
postgres (PostgresFootprintModel): ESTIMATED Postgres row bytes.
|
|
101
|
+
qdrant (QdrantFootprintModel): ESTIMATED vector-store bytes.
|
|
102
|
+
total_bytes (int): S3 (logical) + Postgres + Qdrant.
|
|
103
|
+
"""
|
|
104
|
+
|
|
105
|
+
document_id: str = Field(description="The document's UUID.")
|
|
106
|
+
filename: str = Field(description="The document's display name.")
|
|
107
|
+
s3: S3FootprintModel = Field(description="EXACT S3 bytes.")
|
|
108
|
+
postgres: PostgresFootprintModel = Field(description="ESTIMATED Postgres row bytes.")
|
|
109
|
+
qdrant: QdrantFootprintModel = Field(description="ESTIMATED vector-store bytes.")
|
|
110
|
+
total_bytes: int = Field(description="S3 (logical) + Postgres + Qdrant.")
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
class CollectionStorageResponse(BaseModel):
|
|
114
|
+
"""
|
|
115
|
+
A collection's material footprint per store, plus the per-document breakdown (heaviest first).
|
|
116
|
+
|
|
117
|
+
S3 bytes are EXACT; Postgres and Qdrant bytes are ESTIMATES (each section flags this via its own
|
|
118
|
+
``estimated``). ``grand_total_bytes`` uses the DEDUPED S3 disk cost (``physical_unique_bytes``),
|
|
119
|
+
so it reflects real hardware rather than the logical per-document sum.
|
|
120
|
+
|
|
121
|
+
Attributes:
|
|
122
|
+
collection_id (str): The measured collection's UUID.
|
|
123
|
+
s3 (S3FootprintModel): EXACT S3 totals (logical + deduped physical).
|
|
124
|
+
postgres (PostgresFootprintModel): ESTIMATED Postgres row bytes.
|
|
125
|
+
qdrant (QdrantFootprintModel): ESTIMATED vector-store bytes.
|
|
126
|
+
grand_total_bytes (int): Material footprint — S3 physical_unique + Postgres + Qdrant.
|
|
127
|
+
documents (list[DocumentStorageModel]): Per-document breakdown, sorted by total bytes
|
|
128
|
+
descending (doubles as top-N).
|
|
129
|
+
"""
|
|
130
|
+
|
|
131
|
+
collection_id: str = Field(description="The measured collection's UUID.")
|
|
132
|
+
s3: S3FootprintModel = Field(description="EXACT S3 totals (logical + deduped physical).")
|
|
133
|
+
postgres: PostgresFootprintModel = Field(description="ESTIMATED Postgres row bytes.")
|
|
134
|
+
qdrant: QdrantFootprintModel = Field(description="ESTIMATED vector-store bytes.")
|
|
135
|
+
grand_total_bytes: int = Field(
|
|
136
|
+
description="Material footprint — S3 physical_unique + Postgres + Qdrant."
|
|
137
|
+
)
|
|
138
|
+
documents: list[DocumentStorageModel] = Field(
|
|
139
|
+
description="Per-document breakdown, sorted by total bytes descending (doubles as top-N)."
|
|
140
|
+
)
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
__all__ = [
|
|
144
|
+
"S3FootprintModel",
|
|
145
|
+
"PostgresFootprintModel",
|
|
146
|
+
"QdrantFootprintModel",
|
|
147
|
+
"DocumentStorageModel",
|
|
148
|
+
"CollectionStorageResponse",
|
|
149
|
+
]
|
|
@@ -10,6 +10,7 @@ from ..models.collections import (
|
|
|
10
10
|
CreateCollectionRequest,
|
|
11
11
|
UpdateCollectionRequest,
|
|
12
12
|
)
|
|
13
|
+
from ..models.storage import CollectionStorageResponse
|
|
13
14
|
from ._base import AsyncResource, SyncResource, _ResourceMixin
|
|
14
15
|
|
|
15
16
|
|
|
@@ -80,6 +81,18 @@ class _CollectionsSpecs(_ResourceMixin):
|
|
|
80
81
|
"""
|
|
81
82
|
return RequestSpec("DELETE", f"{self._COLLECTIONS_PATH}/{collection_id}")
|
|
82
83
|
|
|
84
|
+
def _storage_spec(self, collection_id: str) -> RequestSpec:
|
|
85
|
+
"""
|
|
86
|
+
Build the spec for measuring a collection's material storage footprint.
|
|
87
|
+
|
|
88
|
+
Args:
|
|
89
|
+
collection_id (str): The collection's UUID.
|
|
90
|
+
|
|
91
|
+
Returns:
|
|
92
|
+
RequestSpec: A GET on the collection's storage sub-resource.
|
|
93
|
+
"""
|
|
94
|
+
return RequestSpec("GET", f"{self._COLLECTIONS_PATH}/{collection_id}/storage")
|
|
95
|
+
|
|
83
96
|
|
|
84
97
|
class AsyncCollections(AsyncResource, _CollectionsSpecs):
|
|
85
98
|
"""Asynchronous collection management (list / get / create / update / delete)."""
|
|
@@ -141,6 +154,20 @@ class AsyncCollections(AsyncResource, _CollectionsSpecs):
|
|
|
141
154
|
"""
|
|
142
155
|
return await self._transport.request(self._delete_spec(collection_id), type(None))
|
|
143
156
|
|
|
157
|
+
async def storage(self, collection_id: str) -> CollectionStorageResponse:
|
|
158
|
+
"""
|
|
159
|
+
Measure a collection's material footprint per store (S3 exact, Postgres/Qdrant estimated).
|
|
160
|
+
|
|
161
|
+
Args:
|
|
162
|
+
collection_id (str): The collection's UUID.
|
|
163
|
+
|
|
164
|
+
Returns:
|
|
165
|
+
CollectionStorageResponse: Per-store totals + the per-document breakdown, heaviest first.
|
|
166
|
+
"""
|
|
167
|
+
return await self._transport.request(
|
|
168
|
+
self._storage_spec(collection_id), CollectionStorageResponse
|
|
169
|
+
)
|
|
170
|
+
|
|
144
171
|
|
|
145
172
|
class SyncCollections(SyncResource, _CollectionsSpecs):
|
|
146
173
|
"""Synchronous collection management (list / get / create / update / delete)."""
|
|
@@ -200,5 +227,17 @@ class SyncCollections(SyncResource, _CollectionsSpecs):
|
|
|
200
227
|
"""
|
|
201
228
|
return self._transport.request(self._delete_spec(collection_id), type(None))
|
|
202
229
|
|
|
230
|
+
def storage(self, collection_id: str) -> CollectionStorageResponse:
|
|
231
|
+
"""
|
|
232
|
+
Measure a collection's material footprint per store (S3 exact, Postgres/Qdrant estimated).
|
|
233
|
+
|
|
234
|
+
Args:
|
|
235
|
+
collection_id (str): The collection's UUID.
|
|
236
|
+
|
|
237
|
+
Returns:
|
|
238
|
+
CollectionStorageResponse: Per-store totals + the per-document breakdown, heaviest first.
|
|
239
|
+
"""
|
|
240
|
+
return self._transport.request(self._storage_spec(collection_id), CollectionStorageResponse)
|
|
241
|
+
|
|
203
242
|
|
|
204
243
|
__all__ = ["AsyncCollections", "SyncCollections"]
|