docforge-sdk 0.1.2__tar.gz → 0.3.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (60) hide show
  1. {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/.gitignore +8 -0
  2. {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/PKG-INFO +2 -3
  3. {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/README.md +1 -2
  4. {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/docforge_sdk/__init__.py +2 -1
  5. {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/docforge_sdk/_version.py +1 -1
  6. {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/docforge_sdk/models/__init__.py +2 -1
  7. {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/docforge_sdk/models/collections.py +36 -1
  8. {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/docforge_sdk/models/explorer.py +10 -0
  9. docforge_sdk-0.3.0/docforge_sdk/models/jobs.py +165 -0
  10. {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/docforge_sdk/models/search.py +59 -12
  11. docforge_sdk-0.3.0/tests/openapi_snapshot.json +1 -0
  12. {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/tests/parity_map.py +8 -1
  13. {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/tests/unit/test_jobs.py +5 -0
  14. docforge_sdk-0.1.2/docforge_sdk/models/jobs.py +0 -100
  15. docforge_sdk-0.1.2/tests/openapi_snapshot.json +0 -1
  16. {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/.python-version +0 -0
  17. {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/LICENSE +0 -0
  18. {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/docforge_sdk/_exceptions.py +0 -0
  19. {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/docforge_sdk/_requestspec.py +0 -0
  20. {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/docforge_sdk/_transport_async.py +0 -0
  21. {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/docforge_sdk/_transport_base.py +0 -0
  22. {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/docforge_sdk/_transport_sync.py +0 -0
  23. {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/docforge_sdk/client.py +0 -0
  24. {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/docforge_sdk/models/_shared.py +0 -0
  25. {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/docforge_sdk/models/auth.py +0 -0
  26. {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/docforge_sdk/models/blobs.py +0 -0
  27. {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/docforge_sdk/models/documents.py +0 -0
  28. {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/docforge_sdk/models/health.py +0 -0
  29. {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/docforge_sdk/models/ir.py +0 -0
  30. {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/docforge_sdk/models/pipelines.py +0 -0
  31. {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/docforge_sdk/py.typed +0 -0
  32. {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/docforge_sdk/resources/__init__.py +0 -0
  33. {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/docforge_sdk/resources/_base.py +0 -0
  34. {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/docforge_sdk/resources/auth.py +0 -0
  35. {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/docforge_sdk/resources/blobs.py +0 -0
  36. {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/docforge_sdk/resources/collections.py +0 -0
  37. {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/docforge_sdk/resources/documents.py +0 -0
  38. {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/docforge_sdk/resources/explorer.py +0 -0
  39. {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/docforge_sdk/resources/health.py +0 -0
  40. {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/docforge_sdk/resources/jobs.py +0 -0
  41. {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/docforge_sdk/resources/pipelines.py +0 -0
  42. {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/docforge_sdk/resources/search.py +0 -0
  43. {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/pyproject.toml +0 -0
  44. {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/tests/__init__.py +0 -0
  45. {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/tests/check_schema_drift.py +0 -0
  46. {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/tests/live/__init__.py +0 -0
  47. {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/tests/live/test_openapi_parity_live.py +0 -0
  48. {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/tests/unit/__init__.py +0 -0
  49. {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/tests/unit/test_auth.py +0 -0
  50. {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/tests/unit/test_blobs.py +0 -0
  51. {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/tests/unit/test_collections.py +0 -0
  52. {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/tests/unit/test_documents.py +0 -0
  53. {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/tests/unit/test_explorer.py +0 -0
  54. {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/tests/unit/test_health.py +0 -0
  55. {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/tests/unit/test_models_offline_parity.py +0 -0
  56. {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/tests/unit/test_pipelines.py +0 -0
  57. {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/tests/unit/test_resource_parity.py +0 -0
  58. {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/tests/unit/test_search.py +0 -0
  59. {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/tests/unit/test_transport.py +0 -0
  60. {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/uv.lock +0 -0
@@ -84,8 +84,16 @@ htmlcov/
84
84
  !src/docforge/tests/units/build/**
85
85
  src/docforge/app/frontend/scripts/node_modules/
86
86
 
87
+ # UI-screenshot dev tooling — internal, kept private like .claude/ (Playwright-via-Docker recipe).
88
+ src/docforge/app/frontend/scripts/ui-shot.mjs
89
+ src/docforge/app/frontend/scripts/*.png
90
+ src/docforge/app/frontend/UI-SCREENSHOT.md
91
+
87
92
  # Retired AI agent-memory snapshots — internal, not part of the public docs.
88
93
  docs/archive/agent-memory-legacy/
89
94
 
90
95
  # Root AI project instructions — local dev tooling, kept private (like .claude/).
91
96
  /CLAUDE.md
97
+
98
+ # Docker Compose interpolation vars (copied from .env.example) — may hold local pins
99
+ /.env
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: docforge-sdk
3
- Version: 0.1.2
3
+ Version: 0.3.0
4
4
  Summary: Typed async + sync Python client for the DocForge REST API (standalone, zero server deps).
5
5
  Project-URL: Homepage, https://github.com/Florian-BARRE/docforge
6
6
  Project-URL: Repository, https://github.com/Florian-BARRE/docforge
@@ -322,8 +322,7 @@ with Client("http://localhost:10040", api_token="df_...") as client:
322
322
  ```
323
323
 
324
324
  Key `SearchRequest` fields: `query` (required), `limit` (1–100, default 10), `filters`
325
- (`dict[str, Any]`), `search_in` (`list[SearchTarget]` to pick which vectors to query),
326
- `use_late_interaction`, `rescore_pool_size`.
325
+ (`dict[str, Any]`), `search_in` (`list[SearchTarget]` to pick which vectors to query).
327
326
 
328
327
  ### Explore a document (pages, IR, chunks)
329
328
 
@@ -299,8 +299,7 @@ with Client("http://localhost:10040", api_token="df_...") as client:
299
299
  ```
300
300
 
301
301
  Key `SearchRequest` fields: `query` (required), `limit` (1–100, default 10), `filters`
302
- (`dict[str, Any]`), `search_in` (`list[SearchTarget]` to pick which vectors to query),
303
- `use_late_interaction`, `rescore_pool_size`.
302
+ (`dict[str, Any]`), `search_in` (`list[SearchTarget]` to pick which vectors to query).
304
303
 
305
304
  ### Explore a document (pages, IR, chunks)
306
305
 
@@ -78,7 +78,7 @@ from .models.pipelines import (
78
78
  )
79
79
 
80
80
  # ------------------- Search models ------------------- #
81
- from .models.search import SearchHit, SearchRequest, SearchResponse, SearchTarget
81
+ from .models.search import BlockLocation, SearchHit, SearchRequest, SearchResponse, SearchTarget
82
82
 
83
83
  # ------------------- Public API ------------------- #
84
84
  __all__ = [
@@ -131,6 +131,7 @@ __all__ = [
131
131
  # Search
132
132
  "SearchTarget",
133
133
  "SearchRequest",
134
+ "BlockLocation",
134
135
  "SearchHit",
135
136
  "SearchResponse",
136
137
  # Jobs
@@ -2,4 +2,4 @@
2
2
  # Single source of truth for the package version. Hatchling reads ``__version__`` from this file to
3
3
  # populate the distribution metadata (see ``[tool.hatch.version]`` in pyproject.toml).
4
4
 
5
- __version__ = "0.1.2"
5
+ __version__ = "0.3.0"
@@ -62,7 +62,7 @@ from .pipelines import (
62
62
  )
63
63
 
64
64
  # ------------------- Search models ------------------- #
65
- from .search import SearchHit, SearchRequest, SearchResponse, SearchTarget
65
+ from .search import BlockLocation, SearchHit, SearchRequest, SearchResponse, SearchTarget
66
66
 
67
67
  # ------------------- Public API ------------------- #
68
68
  __all__ = [
@@ -111,6 +111,7 @@ __all__ = [
111
111
  # Search
112
112
  "SearchTarget",
113
113
  "SearchRequest",
114
+ "BlockLocation",
114
115
  "SearchHit",
115
116
  "SearchResponse",
116
117
  # Jobs
@@ -5,7 +5,7 @@
5
5
 
6
6
  # ====== Standard Library Imports ======
7
7
  from datetime import datetime
8
- from typing import Any
8
+ from typing import Any, Literal
9
9
 
10
10
  # ====== Third-Party Library Imports ======
11
11
  from pydantic import BaseModel, Field
@@ -56,6 +56,8 @@ class CollectionModel(BaseModel):
56
56
  name (str): Unique human name.
57
57
  supported_formats (list[str]): Accepted upload extensions (e.g. pdf).
58
58
  max_file_size_bytes (int): Upload size ceiling, bytes.
59
+ job_timeout_seconds (float | None): Per-collection whole-ingest-job wall-clock budget,
60
+ seconds. None = inherit the worker's global WORKER_JOB_TIMEOUT_SECONDS default.
59
61
  needs_reindex (bool): True when a config change requires reindexing.
60
62
  created_at (datetime | None): Creation timestamp.
61
63
  pipeline (dict[str, Any]): The ingestion pipeline blob (the graph).
@@ -67,6 +69,14 @@ class CollectionModel(BaseModel):
67
69
  name: str = Field(description="Unique human name.")
68
70
  supported_formats: list[str] = Field(description="Accepted upload extensions (e.g. pdf).")
69
71
  max_file_size_bytes: int = Field(description="Upload size ceiling, bytes.")
72
+ job_timeout_seconds: float | None = Field(
73
+ default=None,
74
+ gt=0,
75
+ description=(
76
+ "Per-collection whole-ingest-job wall-clock budget, seconds. None = inherit the "
77
+ "worker's global WORKER_JOB_TIMEOUT_SECONDS default."
78
+ ),
79
+ )
70
80
  needs_reindex: bool = Field(description="True when a config change requires reindexing.")
71
81
  created_at: datetime | None = Field(default=None, description="Creation timestamp.")
72
82
  pipeline: dict[str, Any] = Field(description="The ingestion pipeline blob (the graph).")
@@ -84,6 +94,8 @@ class CreateCollectionRequest(BaseModel):
84
94
  name (str): Unique human name.
85
95
  supported_formats (list[str]): Accepted upload extensions (e.g. pdf).
86
96
  max_file_size_bytes (int): Upload size ceiling, bytes.
97
+ job_timeout_seconds (float | None): Per-collection whole-ingest-job wall-clock budget,
98
+ seconds. None = inherit the worker's global WORKER_JOB_TIMEOUT_SECONDS default.
87
99
  fields (list[FieldSpec]): The FULL schema, declared up front (vector space is fixed).
88
100
  pipeline (dict[str, Any] | None): The pipeline blob; omitted → the product default.
89
101
  """
@@ -91,6 +103,14 @@ class CreateCollectionRequest(BaseModel):
91
103
  name: str = Field(description="Unique human name.")
92
104
  supported_formats: list[str] = Field(description="Accepted upload extensions (e.g. pdf).")
93
105
  max_file_size_bytes: int = Field(description="Upload size ceiling, bytes.")
106
+ job_timeout_seconds: float | None = Field(
107
+ default=None,
108
+ gt=0,
109
+ description=(
110
+ "Per-collection whole-ingest-job wall-clock budget, seconds. None = inherit the "
111
+ "worker's global WORKER_JOB_TIMEOUT_SECONDS default."
112
+ ),
113
+ )
94
114
  fields: list[FieldSpec] = Field(
95
115
  default_factory=list,
96
116
  description="The FULL schema, declared up front (vector space is fixed at creation).",
@@ -99,6 +119,11 @@ class CreateCollectionRequest(BaseModel):
99
119
  default=None,
100
120
  description="The pipeline blob; omitted → the product default (all stages wired).",
101
121
  )
122
+ preset: Literal["standard", "light"] | None = Field(
123
+ default=None,
124
+ description="Stock-blob selector (ignored when pipeline is set); 'light' = fast, "
125
+ "enrichment-free core.",
126
+ )
102
127
 
103
128
 
104
129
  class UpdateCollectionRequest(BaseModel):
@@ -109,6 +134,8 @@ class UpdateCollectionRequest(BaseModel):
109
134
  name (str | None): New unique name.
110
135
  supported_formats (list[str] | None): New accepted upload extensions.
111
136
  max_file_size_bytes (int | None): New size ceiling, bytes.
137
+ job_timeout_seconds (float | None): New per-collection whole-ingest-job wall-clock
138
+ budget, seconds. Omitted = leave the current value unchanged.
112
139
  fields (list[FieldSpec] | None): The TARGET schema (diffed by field name).
113
140
  pipeline (dict[str, Any] | None): New pipeline blob (validated before storage).
114
141
  search (dict[str, Any] | None): New search graph blob ({} = stock default).
@@ -120,6 +147,14 @@ class UpdateCollectionRequest(BaseModel):
120
147
  default=None, description="New accepted upload extensions."
121
148
  )
122
149
  max_file_size_bytes: int | None = Field(default=None, description="New size ceiling, bytes.")
150
+ job_timeout_seconds: float | None = Field(
151
+ default=None,
152
+ gt=0,
153
+ description=(
154
+ "New per-collection whole-ingest-job wall-clock budget, seconds. Omitted = leave the "
155
+ "current value unchanged; a set value overrides the global WORKER_JOB_TIMEOUT_SECONDS."
156
+ ),
157
+ )
123
158
  fields: list[FieldSpec] | None = Field(
124
159
  default=None,
125
160
  description="The TARGET schema (diffed by field name; omitted fields are removed).",
@@ -148,6 +148,8 @@ class ChunkInfo(BaseModel):
148
148
  parent_id (str | None): Parent chunk (hierarchical chunking).
149
149
  block_ids (list[str]): Composing IR block ids, in assembly order.
150
150
  metadata (list[MetadataValue]): Per-chunk generated metadata values.
151
+ heading_path (list[str]): The chunk's section breadcrumb (outer→inner headings).
152
+ page (int | None): Page of the chunk's primary block; None when it has no located block.
151
153
  """
152
154
 
153
155
  id: str = Field(description="The chunk UUID (doubles as the Qdrant point id).")
@@ -165,6 +167,14 @@ class ChunkInfo(BaseModel):
165
167
  metadata: list[MetadataValue] = Field(
166
168
  default_factory=list, description="Per-chunk generated metadata values."
167
169
  )
170
+ heading_path: list[str] = Field(
171
+ default_factory=list,
172
+ description="The chunk's section breadcrumb (outer→inner headings); [] when none.",
173
+ )
174
+ page: int | None = Field(
175
+ default=None,
176
+ description="Page of the chunk's primary block; None when it has no located block.",
177
+ )
168
178
 
169
179
 
170
180
  class ChunkEnabledPatch(BaseModel):
@@ -0,0 +1,165 @@
1
+ # ====== Code Summary ======
2
+ # Response models for the jobs resource, mirrored field-for-field from the DocForge backend router
3
+ # models: the live ingestion status, one node of the execution trace, the full trace, and the live
4
+ # per-worker activity view.
5
+
6
+ # ====== Standard Library Imports ======
7
+ from datetime import datetime
8
+
9
+ # ====== Third-Party Library Imports ======
10
+ from pydantic import BaseModel, Field
11
+
12
+
13
+ class JobStatus(BaseModel):
14
+ """
15
+ One ingestion job's live state — written by the worker, only read here.
16
+
17
+ Attributes:
18
+ job_id (str): The job row's UUID.
19
+ document_id (str): The document being ingested.
20
+ collection_id (str): Its collection.
21
+ status (str): queued / running / done / failed.
22
+ progress (int): 0–100 (completed pipeline nodes over total).
23
+ current_stage (str | None): The node currently (or last) executed.
24
+ error (str | None): The failure, verbatim — only set when status is failed.
25
+ attempt (int): arq retry attempt (1 = first run).
26
+ started_at (datetime | None): When the worker picked it up.
27
+ finished_at (datetime | None): When it ended (done or failed).
28
+ items_done (int | None): Child items finished in the current fan-out stage (None off it).
29
+ items_total (int | None): The current fan-out stage's width (None when not in a fan-out).
30
+ failed_node_id (str | None): The deepest node that raised — only set on a failed job.
31
+ failed_node_kind (str | None): That node's kind/family label — only set on a failed job.
32
+ failed_item_index (int | None): The fan-out item index the failure sits in (None outside one).
33
+ error_type (str | None): The exception class name of the failure (e.g. "TimeoutError").
34
+ """
35
+
36
+ job_id: str = Field(description="The job row's UUID.")
37
+ document_id: str = Field(description="The document being ingested.")
38
+ collection_id: str = Field(description="Its collection.")
39
+ status: str = Field(description="queued / running / done / failed.")
40
+ progress: int = Field(description="0-100, completed pipeline nodes over total.")
41
+ current_stage: str | None = Field(default=None, description="Node currently/last executed.")
42
+ error: str | None = Field(default=None, description="Failure detail when status=failed.")
43
+ attempt: int = Field(description="arq retry attempt (1 = first run).")
44
+ started_at: datetime | None = Field(default=None, description="Picked up by the worker at.")
45
+ finished_at: datetime | None = Field(default=None, description="Ended (done or failed) at.")
46
+ updated_at: datetime = Field(description="Last progress/lifecycle write (freezes on a wedge).")
47
+ stalled: bool = Field(
48
+ description="A RUNNING job idle past the stall threshold — an early wedge warning."
49
+ )
50
+ total_prompt_tokens: int = Field(
51
+ description="Prompt tokens billed across this job's paid text-gen calls."
52
+ )
53
+ total_completion_tokens: int = Field(
54
+ description="Completion tokens billed across this job's paid text-gen calls."
55
+ )
56
+ cost_usd: float = Field(
57
+ description="USD cost of this job's paid calls (0 when nothing priceable)."
58
+ )
59
+ items_done: int | None = Field(
60
+ default=None, description="Child items finished in the current fan-out stage (None off it)."
61
+ )
62
+ items_total: int | None = Field(
63
+ default=None, description="The current fan-out stage's width (None when not in a fan-out)."
64
+ )
65
+ failed_node_id: str | None = Field(
66
+ default=None, description="Deepest node that raised — only set on a failed job."
67
+ )
68
+ failed_node_kind: str | None = Field(
69
+ default=None, description="That node's kind/family label — only set on a failed job."
70
+ )
71
+ failed_item_index: int | None = Field(
72
+ default=None, description="Fan-out item index the failure sits in (None outside a fan-out)."
73
+ )
74
+ error_type: str | None = Field(
75
+ default=None, description="Exception class name of the failure (e.g. 'TimeoutError')."
76
+ )
77
+
78
+
79
+ class JobEvent(BaseModel):
80
+ """
81
+ One node of the job's execution trace — written by the worker at each stage end.
82
+
83
+ Attributes:
84
+ stage (str): The pipeline node id.
85
+ status (str): success / failed / skipped.
86
+ node_kind (str | None): The stage's structural kind (action/group/foreach) or the node's
87
+ concrete kind — None for rows written before this column landed.
88
+ started_at (datetime | None): Node start.
89
+ finished_at (datetime | None): Node end.
90
+ detail (str | None): Duration, or the error when failed.
91
+ """
92
+
93
+ stage: str = Field(description="The pipeline node id.")
94
+ status: str = Field(description="success / failed / skipped.")
95
+ node_kind: str | None = Field(
96
+ default=None,
97
+ description="The stage's structural kind (action/group/foreach) or the node's concrete "
98
+ "kind — None for rows written before this column landed.",
99
+ )
100
+ started_at: datetime | None = Field(default=None, description="Node start.")
101
+ finished_at: datetime | None = Field(default=None, description="Node end.")
102
+ detail: str | None = Field(default=None, description="Duration, or the error when failed.")
103
+ prompt_tokens: int | None = Field(
104
+ default=None, description="Prompt tokens billed by this stage; null when it made none."
105
+ )
106
+ completion_tokens: int | None = Field(
107
+ default=None, description="Completion tokens billed by this stage; null when none."
108
+ )
109
+ cost_usd: float | None = Field(
110
+ default=None, description="USD cost of this stage; null when no usage or unknown price."
111
+ )
112
+
113
+
114
+ class JobTrace(BaseModel):
115
+ """
116
+ A job's full per-node trace, in execution order.
117
+
118
+ Attributes:
119
+ job_id (str): The traced job.
120
+ events (list[JobEvent]): One entry per stage run.
121
+ """
122
+
123
+ job_id: str = Field(description="The traced job.")
124
+ events: list[JobEvent] = Field(default_factory=list, description="One entry per stage run.")
125
+
126
+
127
+ class WorkerActivity(BaseModel):
128
+ """
129
+ One worker's live activity — its heartbeat-derived liveness plus any RUNNING jobs it owns.
130
+
131
+ Attributes:
132
+ worker_id (str): The worker's stable id (its hostname).
133
+ alive (bool): Its heartbeat is fresher than the liveness threshold.
134
+ busy (bool): It currently owns at least one RUNNING job.
135
+ last_seen (datetime | None): Its last heartbeat tick (None when no heartbeat row exists).
136
+ started_at (datetime | None): When the worker process registered (None when no heartbeat).
137
+ jobs (list[JobStatus]): Its running jobs, live.
138
+ """
139
+
140
+ worker_id: str = Field(description="The worker's stable id (its hostname).")
141
+ alive: bool = Field(description="Heartbeat fresher than the liveness threshold.")
142
+ busy: bool = Field(description="Owns at least one RUNNING job right now.")
143
+ last_seen: datetime | None = Field(
144
+ default=None, description="Last heartbeat tick (None when the worker has no heartbeat row)."
145
+ )
146
+ started_at: datetime | None = Field(
147
+ default=None, description="When the worker process registered (None when no heartbeat)."
148
+ )
149
+ jobs: list[JobStatus] = Field(default_factory=list, description="Its running jobs, live.")
150
+
151
+
152
+ class WorkersLive(BaseModel):
153
+ """
154
+ Everything running right now, grouped by worker — the monitoring view.
155
+
156
+ Attributes:
157
+ workers (list[WorkerActivity]): One entry per active worker.
158
+ """
159
+
160
+ workers: list[WorkerActivity] = Field(
161
+ default_factory=list, description="One entry per active worker."
162
+ )
163
+
164
+
165
+ __all__ = ["JobStatus", "JobEvent", "JobTrace", "WorkerActivity", "WorkersLive"]
@@ -41,8 +41,6 @@ class SearchRequest(BaseModel):
41
41
  limit (int): Number of fused results to return.
42
42
  filters (dict[str, Any] | None): Exact/any-of constraints on the FILTERABLE fields.
43
43
  search_in (list[SearchTarget] | None): Fields × modalities to search. None → content on both.
44
- use_late_interaction (bool | None): Opt into the ColBERT re-score. None → off for this query.
45
- rescore_pool_size (int | None): Size of the fused candidate pool the ColBERT stage re-scores.
46
44
  """
47
45
 
48
46
  query: str = Field(min_length=1, description="The natural-language query to search for.")
@@ -56,15 +54,25 @@ class SearchRequest(BaseModel):
56
54
  description="Fields × modalities to search (content and/or metadata). None → content on "
57
55
  "both semantic and lexical (the unchanged default).",
58
56
  )
59
- use_late_interaction: bool | None = Field(
60
- default=None,
61
- description="Opt into the ColBERT late-interaction re-score for this query. None → off.",
57
+
58
+
59
+ class BlockLocation(BaseModel):
60
+ """
61
+ One source block's location on the page — enough for a UI to draw a box over the hit.
62
+
63
+ Attributes:
64
+ page (int | None): The page the block sits on. None for a page-less document (no page
65
+ render) — distinct from a genuine 0-based page index 0.
66
+ bbox (list[float]): The block's bounding box ``[x0, y0, x1, y1]``, NORMALISED to [0, 1] —
67
+ multiply each component by the page image's width/height to draw it in pixels.
68
+ """
69
+
70
+ page: int | None = Field(
71
+ description="The page the block sits on; None for a page-less document (no page render)."
62
72
  )
63
- rescore_pool_size: int | None = Field(
64
- default=None,
65
- ge=1,
66
- le=1000,
67
- description="Fused candidate pool size the ColBERT stage re-scores. None → node/store default.",
73
+ bbox: list[float] = Field(
74
+ description="Bounding box [x0, y0, x1, y1] NORMALISED to [0, 1] — multiply by the page "
75
+ "image width/height to draw it in pixels.",
68
76
  )
69
77
 
70
78
 
@@ -79,14 +87,53 @@ class SearchHit(BaseModel):
79
87
  text (str): The chunk's enriched text.
80
88
  chunk_index (int): The chunk's ordinal within its document.
81
89
  token_count (int): The chunk's token count.
90
+ block_ids (list[str]): The IR block ids the chunk was assembled from (assembly order).
91
+ page (int | None): The page of the chunk's primary (leading) block — draw the box here.
92
+ bbox (list[float] | None): The primary block's NORMALISED [0, 1] bounding box.
93
+ block_locations (list[BlockLocation]): Every source block's page + bbox (draw them all).
82
94
  """
83
95
 
84
96
  chunk_id: str = Field(description="The chunk's UUID.")
85
97
  document_id: str = Field(description="The owning document's UUID.")
98
+ filename: str | None = Field(
99
+ default=None, description="The source document's filename — the hit's human identity."
100
+ )
101
+ document_title: str | None = Field(
102
+ default=None, description="The source document's title (empty parsed titles → null)."
103
+ )
104
+ heading_path: list[str] = Field(
105
+ default_factory=list,
106
+ description="The chunk's section ancestry, top-down (e.g. ['Article 7 — Audit rights']) — "
107
+ "so a hit self-cites the section/clause it came from. Empty when the chunk sits under no "
108
+ "section.",
109
+ )
110
+ metadata: dict[str, Any] = Field(
111
+ default_factory=dict,
112
+ description="The document's filterable metadata (field → value) — so a hit self-cites "
113
+ "without a second GET /documents/{id}.",
114
+ )
86
115
  score: float = Field(description="Fused RRF score (higher is better).")
87
116
  text: str = Field(description="The chunk's enriched text.")
88
117
  chunk_index: int = Field(description="Ordinal within the document.")
89
118
  token_count: int = Field(description="Token count of the chunk.")
119
+ block_ids: list[str] = Field(
120
+ default_factory=list,
121
+ description="The IR block ids the chunk was assembled from, in assembly order.",
122
+ )
123
+ page: int | None = Field(
124
+ default=None,
125
+ description="The page of the chunk's primary (leading) block — where to draw the box. "
126
+ "None when the chunk carries no block location.",
127
+ )
128
+ bbox: list[float] | None = Field(
129
+ default=None,
130
+ description="The primary block's bounding box [x0, y0, x1, y1], NORMALISED to [0, 1] — "
131
+ "multiply by the page image width/height to draw it. None when unlocated.",
132
+ )
133
+ block_locations: list[BlockLocation] = Field(
134
+ default_factory=list,
135
+ description="Every source block's page + NORMALISED bbox — draw one box per block.",
136
+ )
90
137
 
91
138
 
92
139
  class SearchResponse(BaseModel):
@@ -103,8 +150,8 @@ class SearchResponse(BaseModel):
103
150
  hits: list[SearchHit] = Field(default_factory=list, description="Ranked hits, best first.")
104
151
  debug_info: dict[str, Any] | None = Field(
105
152
  default=None,
106
- description="Non-fatal diagnostics (e.g. late_interaction_skipped). None when empty.",
153
+ description="Non-fatal diagnostics about how the search ran. None when empty.",
107
154
  )
108
155
 
109
156
 
110
- __all__ = ["SearchTarget", "SearchRequest", "SearchHit", "SearchResponse"]
157
+ __all__ = ["SearchTarget", "SearchRequest", "BlockLocation", "SearchHit", "SearchResponse"]