docforge-sdk 0.2.0__tar.gz → 0.4.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (60) hide show
  1. {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/PKG-INFO +24 -2
  2. {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/README.md +23 -1
  3. {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/docforge_sdk/__init__.py +15 -0
  4. {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/docforge_sdk/_version.py +1 -1
  5. {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/docforge_sdk/models/__init__.py +15 -0
  6. {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/docforge_sdk/models/collections.py +30 -0
  7. docforge_sdk-0.4.0/docforge_sdk/models/storage.py +149 -0
  8. {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/docforge_sdk/resources/collections.py +39 -0
  9. docforge_sdk-0.4.0/tests/openapi_snapshot.json +6807 -0
  10. {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/tests/parity_map.py +13 -0
  11. {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/tests/unit/test_collections.py +69 -0
  12. docforge_sdk-0.2.0/tests/openapi_snapshot.json +0 -1
  13. {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/.gitignore +0 -0
  14. {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/.python-version +0 -0
  15. {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/LICENSE +0 -0
  16. {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/docforge_sdk/_exceptions.py +0 -0
  17. {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/docforge_sdk/_requestspec.py +0 -0
  18. {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/docforge_sdk/_transport_async.py +0 -0
  19. {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/docforge_sdk/_transport_base.py +0 -0
  20. {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/docforge_sdk/_transport_sync.py +0 -0
  21. {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/docforge_sdk/client.py +0 -0
  22. {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/docforge_sdk/models/_shared.py +0 -0
  23. {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/docforge_sdk/models/auth.py +0 -0
  24. {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/docforge_sdk/models/blobs.py +0 -0
  25. {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/docforge_sdk/models/documents.py +0 -0
  26. {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/docforge_sdk/models/explorer.py +0 -0
  27. {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/docforge_sdk/models/health.py +0 -0
  28. {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/docforge_sdk/models/ir.py +0 -0
  29. {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/docforge_sdk/models/jobs.py +0 -0
  30. {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/docforge_sdk/models/pipelines.py +0 -0
  31. {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/docforge_sdk/models/search.py +0 -0
  32. {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/docforge_sdk/py.typed +0 -0
  33. {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/docforge_sdk/resources/__init__.py +0 -0
  34. {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/docforge_sdk/resources/_base.py +0 -0
  35. {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/docforge_sdk/resources/auth.py +0 -0
  36. {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/docforge_sdk/resources/blobs.py +0 -0
  37. {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/docforge_sdk/resources/documents.py +0 -0
  38. {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/docforge_sdk/resources/explorer.py +0 -0
  39. {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/docforge_sdk/resources/health.py +0 -0
  40. {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/docforge_sdk/resources/jobs.py +0 -0
  41. {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/docforge_sdk/resources/pipelines.py +0 -0
  42. {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/docforge_sdk/resources/search.py +0 -0
  43. {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/pyproject.toml +0 -0
  44. {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/tests/__init__.py +0 -0
  45. {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/tests/check_schema_drift.py +0 -0
  46. {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/tests/live/__init__.py +0 -0
  47. {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/tests/live/test_openapi_parity_live.py +0 -0
  48. {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/tests/unit/__init__.py +0 -0
  49. {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/tests/unit/test_auth.py +0 -0
  50. {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/tests/unit/test_blobs.py +0 -0
  51. {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/tests/unit/test_documents.py +0 -0
  52. {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/tests/unit/test_explorer.py +0 -0
  53. {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/tests/unit/test_health.py +0 -0
  54. {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/tests/unit/test_jobs.py +0 -0
  55. {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/tests/unit/test_models_offline_parity.py +0 -0
  56. {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/tests/unit/test_pipelines.py +0 -0
  57. {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/tests/unit/test_resource_parity.py +0 -0
  58. {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/tests/unit/test_search.py +0 -0
  59. {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/tests/unit/test_transport.py +0 -0
  60. {docforge_sdk-0.2.0 → docforge_sdk-0.4.0}/uv.lock +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: docforge-sdk
3
- Version: 0.2.0
3
+ Version: 0.4.0
4
4
  Summary: Typed async + sync Python client for the DocForge REST API (standalone, zero server deps).
5
5
  Project-URL: Homepage, https://github.com/Florian-BARRE/docforge
6
6
  Project-URL: Repository, https://github.com/Florian-BARRE/docforge
@@ -189,6 +189,7 @@ are identical.
189
189
  | `create(CreateCollectionRequest)` | `CollectionModel` | Create a collection (contract). |
190
190
  | `update(collection_id, UpdateCollectionRequest)` | `CollectionModel` | Patch name / formats / fields / pipelines. |
191
191
  | `delete(collection_id)` | `None` | Delete a collection. |
192
+ | `storage(collection_id)` | `CollectionStorageResponse` | Material storage footprint per store (S3 exact, Postgres/Qdrant estimated) + per-document breakdown. |
192
193
 
193
194
  ### `documents`
194
195
  | Method | Returns | Purpose |
@@ -271,6 +272,12 @@ with Client("http://localhost:10040", api_token="df_...") as client:
271
272
  print(collection.id)
272
273
  ```
273
274
 
275
+ Key `CreateCollectionRequest` fields beyond the schema: `max_file_size_bytes` (bytes) and
276
+ `job_timeout_seconds` (`float | None`, seconds) — the whole-ingest-job wall-clock budget for that
277
+ collection; `None` (the default) inherits the worker's global job-timeout default. Same field, same
278
+ semantics on `CollectionModel` (read) and `UpdateCollectionRequest` (write; there, omitting it
279
+ leaves the current value unchanged, a set value overrides it).
280
+
274
281
  ### Upload a document and wait for ingestion
275
282
 
276
283
  `upload` returns immediately with a `job_id`; ingestion runs asynchronously on the worker. Poll the
@@ -338,6 +345,20 @@ with Client("http://localhost:10040", api_token="df_...") as client:
338
345
  client.explorer.set_chunk_enabled(chunks[0].chunk_id, enabled=False)
339
346
  ```
340
347
 
348
+ ### Measure a collection's storage footprint
349
+
350
+ `storage` reports the material footprint per store — S3 bytes are exact (deduped), Postgres/Qdrant
351
+ bytes are estimates (each section flags this via its own `estimated`) — plus a per-document
352
+ breakdown sorted heaviest first.
353
+
354
+ ```python
355
+ with Client("http://localhost:10040", api_token="df_...") as client:
356
+ footprint = client.collections.storage(collection_id)
357
+ print(footprint.grand_total_bytes, footprint.s3.physical_unique_bytes)
358
+ for doc in footprint.documents[:5]:
359
+ print(doc.filename, doc.total_bytes)
360
+ ```
361
+
341
362
  ### Manage API keys
342
363
 
343
364
  Mint a scoped, expiring key (the plaintext is shown **once**, on creation):
@@ -424,6 +445,7 @@ from docforge_sdk import (
424
445
  UpdateCollectionRequest,
425
446
  FieldSpec,
426
447
  FieldType,
448
+ CollectionStorageResponse,
427
449
  # documents / explorer
428
450
  UploadAccepted,
429
451
  DocumentDetail,
@@ -465,7 +487,7 @@ server's OpenAPI on every change, so a published version is coherent with the AP
465
487
  version in production:
466
488
 
467
489
  ```bash
468
- pip install "docforge-sdk==0.1.0"
490
+ pip install "docforge-sdk==0.3.0"
469
491
  ```
470
492
 
471
493
  ## License
@@ -166,6 +166,7 @@ are identical.
166
166
  | `create(CreateCollectionRequest)` | `CollectionModel` | Create a collection (contract). |
167
167
  | `update(collection_id, UpdateCollectionRequest)` | `CollectionModel` | Patch name / formats / fields / pipelines. |
168
168
  | `delete(collection_id)` | `None` | Delete a collection. |
169
+ | `storage(collection_id)` | `CollectionStorageResponse` | Material storage footprint per store (S3 exact, Postgres/Qdrant estimated) + per-document breakdown. |
169
170
 
170
171
  ### `documents`
171
172
  | Method | Returns | Purpose |
@@ -248,6 +249,12 @@ with Client("http://localhost:10040", api_token="df_...") as client:
248
249
  print(collection.id)
249
250
  ```
250
251
 
252
+ Key `CreateCollectionRequest` fields beyond the schema: `max_file_size_bytes` (bytes) and
253
+ `job_timeout_seconds` (`float | None`, seconds) — the whole-ingest-job wall-clock budget for that
254
+ collection; `None` (the default) inherits the worker's global job-timeout default. Same field, same
255
+ semantics on `CollectionModel` (read) and `UpdateCollectionRequest` (write; there, omitting it
256
+ leaves the current value unchanged, a set value overrides it).
257
+
251
258
  ### Upload a document and wait for ingestion
252
259
 
253
260
  `upload` returns immediately with a `job_id`; ingestion runs asynchronously on the worker. Poll the
@@ -315,6 +322,20 @@ with Client("http://localhost:10040", api_token="df_...") as client:
315
322
  client.explorer.set_chunk_enabled(chunks[0].chunk_id, enabled=False)
316
323
  ```
317
324
 
325
+ ### Measure a collection's storage footprint
326
+
327
+ `storage` reports the material footprint per store — S3 bytes are exact (deduped), Postgres/Qdrant
328
+ bytes are estimates (each section flags this via its own `estimated`) — plus a per-document
329
+ breakdown sorted heaviest first.
330
+
331
+ ```python
332
+ with Client("http://localhost:10040", api_token="df_...") as client:
333
+ footprint = client.collections.storage(collection_id)
334
+ print(footprint.grand_total_bytes, footprint.s3.physical_unique_bytes)
335
+ for doc in footprint.documents[:5]:
336
+ print(doc.filename, doc.total_bytes)
337
+ ```
338
+
318
339
  ### Manage API keys
319
340
 
320
341
  Mint a scoped, expiring key (the plaintext is shown **once**, on creation):
@@ -401,6 +422,7 @@ from docforge_sdk import (
401
422
  UpdateCollectionRequest,
402
423
  FieldSpec,
403
424
  FieldType,
425
+ CollectionStorageResponse,
404
426
  # documents / explorer
405
427
  UploadAccepted,
406
428
  DocumentDetail,
@@ -442,7 +464,7 @@ server's OpenAPI on every change, so a published version is coherent with the AP
442
464
  version in production:
443
465
 
444
466
  ```bash
445
- pip install "docforge-sdk==0.1.0"
467
+ pip install "docforge-sdk==0.3.0"
446
468
  ```
447
469
 
448
470
  ## License
@@ -80,6 +80,15 @@ from .models.pipelines import (
80
80
  # ------------------- Search models ------------------- #
81
81
  from .models.search import BlockLocation, SearchHit, SearchRequest, SearchResponse, SearchTarget
82
82
 
83
+ # ------------------- Storage models ------------------- #
84
+ from .models.storage import (
85
+ CollectionStorageResponse,
86
+ DocumentStorageModel,
87
+ PostgresFootprintModel,
88
+ QdrantFootprintModel,
89
+ S3FootprintModel,
90
+ )
91
+
83
92
  # ------------------- Public API ------------------- #
84
93
  __all__ = [
85
94
  "__version__",
@@ -150,6 +159,12 @@ __all__ = [
150
159
  "EditResponse",
151
160
  "StageViewResponse",
152
161
  "StageApplyResponse",
162
+ # Storage
163
+ "S3FootprintModel",
164
+ "PostgresFootprintModel",
165
+ "QdrantFootprintModel",
166
+ "DocumentStorageModel",
167
+ "CollectionStorageResponse",
153
168
  # Exceptions
154
169
  "DocForgeError",
155
170
  "APIConnectionError",
@@ -2,4 +2,4 @@
2
2
  # Single source of truth for the package version. Hatchling reads ``__version__`` from this file to
3
3
  # populate the distribution metadata (see ``[tool.hatch.version]`` in pyproject.toml).
4
4
 
5
- __version__ = "0.2.0"
5
+ __version__ = "0.4.0"
@@ -64,6 +64,15 @@ from .pipelines import (
64
64
  # ------------------- Search models ------------------- #
65
65
  from .search import BlockLocation, SearchHit, SearchRequest, SearchResponse, SearchTarget
66
66
 
67
+ # ------------------- Storage models ------------------- #
68
+ from .storage import (
69
+ CollectionStorageResponse,
70
+ DocumentStorageModel,
71
+ PostgresFootprintModel,
72
+ QdrantFootprintModel,
73
+ S3FootprintModel,
74
+ )
75
+
67
76
  # ------------------- Public API ------------------- #
68
77
  __all__ = [
69
78
  # Shared vocabulary
@@ -130,4 +139,10 @@ __all__ = [
130
139
  "EditResponse",
131
140
  "StageViewResponse",
132
141
  "StageApplyResponse",
142
+ # Storage
143
+ "S3FootprintModel",
144
+ "PostgresFootprintModel",
145
+ "QdrantFootprintModel",
146
+ "DocumentStorageModel",
147
+ "CollectionStorageResponse",
133
148
  ]
@@ -56,6 +56,8 @@ class CollectionModel(BaseModel):
56
56
  name (str): Unique human name.
57
57
  supported_formats (list[str]): Accepted upload extensions (e.g. pdf).
58
58
  max_file_size_bytes (int): Upload size ceiling, bytes.
59
+ job_timeout_seconds (float | None): Per-collection whole-ingest-job wall-clock budget,
60
+ seconds. None = inherit the worker's global WORKER_JOB_TIMEOUT_SECONDS default.
59
61
  needs_reindex (bool): True when a config change requires reindexing.
60
62
  created_at (datetime | None): Creation timestamp.
61
63
  pipeline (dict[str, Any]): The ingestion pipeline blob (the graph).
@@ -67,6 +69,14 @@ class CollectionModel(BaseModel):
67
69
  name: str = Field(description="Unique human name.")
68
70
  supported_formats: list[str] = Field(description="Accepted upload extensions (e.g. pdf).")
69
71
  max_file_size_bytes: int = Field(description="Upload size ceiling, bytes.")
72
+ job_timeout_seconds: float | None = Field(
73
+ default=None,
74
+ gt=0,
75
+ description=(
76
+ "Per-collection whole-ingest-job wall-clock budget, seconds. None = inherit the "
77
+ "worker's global WORKER_JOB_TIMEOUT_SECONDS default."
78
+ ),
79
+ )
70
80
  needs_reindex: bool = Field(description="True when a config change requires reindexing.")
71
81
  created_at: datetime | None = Field(default=None, description="Creation timestamp.")
72
82
  pipeline: dict[str, Any] = Field(description="The ingestion pipeline blob (the graph).")
@@ -84,6 +94,8 @@ class CreateCollectionRequest(BaseModel):
84
94
  name (str): Unique human name.
85
95
  supported_formats (list[str]): Accepted upload extensions (e.g. pdf).
86
96
  max_file_size_bytes (int): Upload size ceiling, bytes.
97
+ job_timeout_seconds (float | None): Per-collection whole-ingest-job wall-clock budget,
98
+ seconds. None = inherit the worker's global WORKER_JOB_TIMEOUT_SECONDS default.
87
99
  fields (list[FieldSpec]): The FULL schema, declared up front (vector space is fixed).
88
100
  pipeline (dict[str, Any] | None): The pipeline blob; omitted → the product default.
89
101
  """
@@ -91,6 +103,14 @@ class CreateCollectionRequest(BaseModel):
91
103
  name: str = Field(description="Unique human name.")
92
104
  supported_formats: list[str] = Field(description="Accepted upload extensions (e.g. pdf).")
93
105
  max_file_size_bytes: int = Field(description="Upload size ceiling, bytes.")
106
+ job_timeout_seconds: float | None = Field(
107
+ default=None,
108
+ gt=0,
109
+ description=(
110
+ "Per-collection whole-ingest-job wall-clock budget, seconds. None = inherit the "
111
+ "worker's global WORKER_JOB_TIMEOUT_SECONDS default."
112
+ ),
113
+ )
94
114
  fields: list[FieldSpec] = Field(
95
115
  default_factory=list,
96
116
  description="The FULL schema, declared up front (vector space is fixed at creation).",
@@ -114,6 +134,8 @@ class UpdateCollectionRequest(BaseModel):
114
134
  name (str | None): New unique name.
115
135
  supported_formats (list[str] | None): New accepted upload extensions.
116
136
  max_file_size_bytes (int | None): New size ceiling, bytes.
137
+ job_timeout_seconds (float | None): New per-collection whole-ingest-job wall-clock
138
+ budget, seconds. Omitted = leave the current value unchanged.
117
139
  fields (list[FieldSpec] | None): The TARGET schema (diffed by field name).
118
140
  pipeline (dict[str, Any] | None): New pipeline blob (validated before storage).
119
141
  search (dict[str, Any] | None): New search graph blob ({} = stock default).
@@ -125,6 +147,14 @@ class UpdateCollectionRequest(BaseModel):
125
147
  default=None, description="New accepted upload extensions."
126
148
  )
127
149
  max_file_size_bytes: int | None = Field(default=None, description="New size ceiling, bytes.")
150
+ job_timeout_seconds: float | None = Field(
151
+ default=None,
152
+ gt=0,
153
+ description=(
154
+ "New per-collection whole-ingest-job wall-clock budget, seconds. Omitted = leave the "
155
+ "current value unchanged; a set value overrides the global WORKER_JOB_TIMEOUT_SECONDS."
156
+ ),
157
+ )
128
158
  fields: list[FieldSpec] | None = Field(
129
159
  default=None,
130
160
  description="The TARGET schema (diffed by field name; omitted fields are removed).",
@@ -0,0 +1,149 @@
1
+ # ====== Code Summary ======
2
+ # Response models for the collection storage-footprint endpoint, mirrored field-for-field from the
3
+ # DocForge backend router models (S3FootprintModel / PostgresFootprintModel / QdrantFootprintModel /
4
+ # DocumentStorageModel / CollectionStorageResponse). S3 bytes are EXACT; Postgres and Qdrant bytes are
5
+ # ESTIMATES — each section carries its own ``estimated`` flag.
6
+
7
+ # ====== Third-Party Library Imports ======
8
+ from pydantic import BaseModel, Field
9
+
10
+
11
+ class S3FootprintModel(BaseModel):
12
+ """
13
+ EXACT S3 bytes from the content-addressed blob registry (``estimated`` is always false).
14
+
15
+ Attributes:
16
+ original_bytes (int): Uploaded source file bytes.
17
+ rendered_bytes (int): Derived-blob bytes (canonical PDF, page renders, crops).
18
+ total_bytes (int): Logical bytes (original + rendered).
19
+ physical_unique_bytes (int): Deduped disk cost — a blob shared across documents counts
20
+ once (<= total).
21
+ estimated (bool): Always false: S3 bytes are measured exactly.
22
+ """
23
+
24
+ original_bytes: int = Field(description="Uploaded source file bytes.")
25
+ rendered_bytes: int = Field(
26
+ description="Derived-blob bytes (canonical PDF, page renders, crops)."
27
+ )
28
+ total_bytes: int = Field(description="Logical bytes (original + rendered).")
29
+ physical_unique_bytes: int = Field(
30
+ description="Deduped disk cost — a blob shared across documents counts once (<= total)."
31
+ )
32
+ estimated: bool = Field(description="Always false: S3 bytes are measured exactly.")
33
+
34
+
35
+ class PostgresFootprintModel(BaseModel):
36
+ """
37
+ ESTIMATED Postgres row bytes via ``pg_column_size`` (excludes index/TOAST/bloat).
38
+
39
+ Attributes:
40
+ documents_bytes (int): ``document`` + ``page`` rows.
41
+ ir_blocks_bytes (int): ``block`` + ``block_table`` + ``block_figure`` rows.
42
+ enrichment_bytes (int): ``block_enrichment`` + ``enrichment_attempt`` rows.
43
+ chunks_bytes (int): ``chunk`` + ``chunk_block`` + ``chunk_metadata`` + ``entity_mention`` rows.
44
+ metadata_bytes (int): ``document_metadata`` rows.
45
+ observability_bytes (int): ``job`` + ``job_stage_event`` rows.
46
+ total_bytes (int): Sum of every bucket.
47
+ estimated (bool): Always true: real row bytes, no index/TOAST/bloat.
48
+ """
49
+
50
+ documents_bytes: int = Field(description="``document`` + ``page`` rows.")
51
+ ir_blocks_bytes: int = Field(description="``block`` + ``block_table`` + ``block_figure`` rows.")
52
+ enrichment_bytes: int = Field(description="``block_enrichment`` + ``enrichment_attempt`` rows.")
53
+ chunks_bytes: int = Field(
54
+ description="``chunk`` + ``chunk_block`` + ``chunk_metadata`` + ``entity_mention`` rows."
55
+ )
56
+ metadata_bytes: int = Field(description="``document_metadata`` rows.")
57
+ observability_bytes: int = Field(description="``job`` + ``job_stage_event`` rows.")
58
+ total_bytes: int = Field(description="Sum of every bucket.")
59
+ estimated: bool = Field(description="Always true: real row bytes, no index/TOAST/bloat.")
60
+
61
+
62
+ class QdrantFootprintModel(BaseModel):
63
+ """
64
+ ESTIMATED vector-store bytes (points x declared shape — excludes HNSW index overhead).
65
+
66
+ Attributes:
67
+ points (int): Point count (collection total, or a document's points).
68
+ dense_bytes (int): On-disk float32 dense bytes, summed per named vector weighted by its
69
+ carrier count. Excludes the int8 quantized RAM-resident copy — this is disk, not RAM.
70
+ sparse_bytes (int): ``points * avg_sparse_entries * 8`` (int32 index + float32 value).
71
+ payload_bytes (int): ``points * avg_payload_json_bytes``.
72
+ total_bytes (int): Sum of dense + sparse + payload.
73
+ estimated (bool): Always true: count-based, excludes index overhead.
74
+ """
75
+
76
+ points: int = Field(description="Point count (collection total, or a document's points).")
77
+ dense_bytes: int = Field(
78
+ description=(
79
+ "On-disk float32 dense bytes, summed per named vector weighted by its carrier count "
80
+ "(content_dense on every point, each meta vector only on its field's documents). "
81
+ "Excludes the int8 quantized RAM-resident copy — this is disk, not RAM."
82
+ )
83
+ )
84
+ sparse_bytes: int = Field(
85
+ description="``points x avg_sparse_entries x 8`` (int32 index + float32 value)."
86
+ )
87
+ payload_bytes: int = Field(description="``points x avg_payload_json_bytes``.")
88
+ total_bytes: int = Field(description="Sum of dense + sparse + payload.")
89
+ estimated: bool = Field(description="Always true: count-based, excludes index overhead.")
90
+
91
+
92
+ class DocumentStorageModel(BaseModel):
93
+ """
94
+ One document's footprint across the three stores.
95
+
96
+ Attributes:
97
+ document_id (str): The document's UUID.
98
+ filename (str): The document's display name.
99
+ s3 (S3FootprintModel): EXACT S3 bytes.
100
+ postgres (PostgresFootprintModel): ESTIMATED Postgres row bytes.
101
+ qdrant (QdrantFootprintModel): ESTIMATED vector-store bytes.
102
+ total_bytes (int): S3 (logical) + Postgres + Qdrant.
103
+ """
104
+
105
+ document_id: str = Field(description="The document's UUID.")
106
+ filename: str = Field(description="The document's display name.")
107
+ s3: S3FootprintModel = Field(description="EXACT S3 bytes.")
108
+ postgres: PostgresFootprintModel = Field(description="ESTIMATED Postgres row bytes.")
109
+ qdrant: QdrantFootprintModel = Field(description="ESTIMATED vector-store bytes.")
110
+ total_bytes: int = Field(description="S3 (logical) + Postgres + Qdrant.")
111
+
112
+
113
+ class CollectionStorageResponse(BaseModel):
114
+ """
115
+ A collection's material footprint per store, plus the per-document breakdown (heaviest first).
116
+
117
+ S3 bytes are EXACT; Postgres and Qdrant bytes are ESTIMATES (each section flags this via its own
118
+ ``estimated``). ``grand_total_bytes`` uses the DEDUPED S3 disk cost (``physical_unique_bytes``),
119
+ so it reflects real hardware rather than the logical per-document sum.
120
+
121
+ Attributes:
122
+ collection_id (str): The measured collection's UUID.
123
+ s3 (S3FootprintModel): EXACT S3 totals (logical + deduped physical).
124
+ postgres (PostgresFootprintModel): ESTIMATED Postgres row bytes.
125
+ qdrant (QdrantFootprintModel): ESTIMATED vector-store bytes.
126
+ grand_total_bytes (int): Material footprint — S3 physical_unique + Postgres + Qdrant.
127
+ documents (list[DocumentStorageModel]): Per-document breakdown, sorted by total bytes
128
+ descending (doubles as top-N).
129
+ """
130
+
131
+ collection_id: str = Field(description="The measured collection's UUID.")
132
+ s3: S3FootprintModel = Field(description="EXACT S3 totals (logical + deduped physical).")
133
+ postgres: PostgresFootprintModel = Field(description="ESTIMATED Postgres row bytes.")
134
+ qdrant: QdrantFootprintModel = Field(description="ESTIMATED vector-store bytes.")
135
+ grand_total_bytes: int = Field(
136
+ description="Material footprint — S3 physical_unique + Postgres + Qdrant."
137
+ )
138
+ documents: list[DocumentStorageModel] = Field(
139
+ description="Per-document breakdown, sorted by total bytes descending (doubles as top-N)."
140
+ )
141
+
142
+
143
+ __all__ = [
144
+ "S3FootprintModel",
145
+ "PostgresFootprintModel",
146
+ "QdrantFootprintModel",
147
+ "DocumentStorageModel",
148
+ "CollectionStorageResponse",
149
+ ]
@@ -10,6 +10,7 @@ from ..models.collections import (
10
10
  CreateCollectionRequest,
11
11
  UpdateCollectionRequest,
12
12
  )
13
+ from ..models.storage import CollectionStorageResponse
13
14
  from ._base import AsyncResource, SyncResource, _ResourceMixin
14
15
 
15
16
 
@@ -80,6 +81,18 @@ class _CollectionsSpecs(_ResourceMixin):
80
81
  """
81
82
  return RequestSpec("DELETE", f"{self._COLLECTIONS_PATH}/{collection_id}")
82
83
 
84
+ def _storage_spec(self, collection_id: str) -> RequestSpec:
85
+ """
86
+ Build the spec for measuring a collection's material storage footprint.
87
+
88
+ Args:
89
+ collection_id (str): The collection's UUID.
90
+
91
+ Returns:
92
+ RequestSpec: A GET on the collection's storage sub-resource.
93
+ """
94
+ return RequestSpec("GET", f"{self._COLLECTIONS_PATH}/{collection_id}/storage")
95
+
83
96
 
84
97
  class AsyncCollections(AsyncResource, _CollectionsSpecs):
85
98
  """Asynchronous collection management (list / get / create / update / delete)."""
@@ -141,6 +154,20 @@ class AsyncCollections(AsyncResource, _CollectionsSpecs):
141
154
  """
142
155
  return await self._transport.request(self._delete_spec(collection_id), type(None))
143
156
 
157
+ async def storage(self, collection_id: str) -> CollectionStorageResponse:
158
+ """
159
+ Measure a collection's material footprint per store (S3 exact, Postgres/Qdrant estimated).
160
+
161
+ Args:
162
+ collection_id (str): The collection's UUID.
163
+
164
+ Returns:
165
+ CollectionStorageResponse: Per-store totals + the per-document breakdown, heaviest first.
166
+ """
167
+ return await self._transport.request(
168
+ self._storage_spec(collection_id), CollectionStorageResponse
169
+ )
170
+
144
171
 
145
172
  class SyncCollections(SyncResource, _CollectionsSpecs):
146
173
  """Synchronous collection management (list / get / create / update / delete)."""
@@ -200,5 +227,17 @@ class SyncCollections(SyncResource, _CollectionsSpecs):
200
227
  """
201
228
  return self._transport.request(self._delete_spec(collection_id), type(None))
202
229
 
230
+ def storage(self, collection_id: str) -> CollectionStorageResponse:
231
+ """
232
+ Measure a collection's material footprint per store (S3 exact, Postgres/Qdrant estimated).
233
+
234
+ Args:
235
+ collection_id (str): The collection's UUID.
236
+
237
+ Returns:
238
+ CollectionStorageResponse: Per-store totals + the per-document breakdown, heaviest first.
239
+ """
240
+ return self._transport.request(self._storage_spec(collection_id), CollectionStorageResponse)
241
+
203
242
 
204
243
  __all__ = ["AsyncCollections", "SyncCollections"]