docforge-sdk 0.1.2__tar.gz → 0.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/.gitignore +8 -0
- {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/PKG-INFO +2 -3
- {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/README.md +1 -2
- {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/docforge_sdk/__init__.py +2 -1
- {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/docforge_sdk/_version.py +1 -1
- {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/docforge_sdk/models/__init__.py +2 -1
- {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/docforge_sdk/models/collections.py +36 -1
- {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/docforge_sdk/models/explorer.py +10 -0
- docforge_sdk-0.3.0/docforge_sdk/models/jobs.py +165 -0
- {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/docforge_sdk/models/search.py +59 -12
- docforge_sdk-0.3.0/tests/openapi_snapshot.json +1 -0
- {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/tests/parity_map.py +8 -1
- {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/tests/unit/test_jobs.py +5 -0
- docforge_sdk-0.1.2/docforge_sdk/models/jobs.py +0 -100
- docforge_sdk-0.1.2/tests/openapi_snapshot.json +0 -1
- {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/.python-version +0 -0
- {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/LICENSE +0 -0
- {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/docforge_sdk/_exceptions.py +0 -0
- {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/docforge_sdk/_requestspec.py +0 -0
- {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/docforge_sdk/_transport_async.py +0 -0
- {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/docforge_sdk/_transport_base.py +0 -0
- {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/docforge_sdk/_transport_sync.py +0 -0
- {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/docforge_sdk/client.py +0 -0
- {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/docforge_sdk/models/_shared.py +0 -0
- {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/docforge_sdk/models/auth.py +0 -0
- {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/docforge_sdk/models/blobs.py +0 -0
- {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/docforge_sdk/models/documents.py +0 -0
- {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/docforge_sdk/models/health.py +0 -0
- {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/docforge_sdk/models/ir.py +0 -0
- {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/docforge_sdk/models/pipelines.py +0 -0
- {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/docforge_sdk/py.typed +0 -0
- {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/docforge_sdk/resources/__init__.py +0 -0
- {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/docforge_sdk/resources/_base.py +0 -0
- {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/docforge_sdk/resources/auth.py +0 -0
- {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/docforge_sdk/resources/blobs.py +0 -0
- {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/docforge_sdk/resources/collections.py +0 -0
- {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/docforge_sdk/resources/documents.py +0 -0
- {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/docforge_sdk/resources/explorer.py +0 -0
- {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/docforge_sdk/resources/health.py +0 -0
- {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/docforge_sdk/resources/jobs.py +0 -0
- {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/docforge_sdk/resources/pipelines.py +0 -0
- {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/docforge_sdk/resources/search.py +0 -0
- {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/pyproject.toml +0 -0
- {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/tests/__init__.py +0 -0
- {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/tests/check_schema_drift.py +0 -0
- {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/tests/live/__init__.py +0 -0
- {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/tests/live/test_openapi_parity_live.py +0 -0
- {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/tests/unit/__init__.py +0 -0
- {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/tests/unit/test_auth.py +0 -0
- {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/tests/unit/test_blobs.py +0 -0
- {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/tests/unit/test_collections.py +0 -0
- {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/tests/unit/test_documents.py +0 -0
- {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/tests/unit/test_explorer.py +0 -0
- {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/tests/unit/test_health.py +0 -0
- {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/tests/unit/test_models_offline_parity.py +0 -0
- {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/tests/unit/test_pipelines.py +0 -0
- {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/tests/unit/test_resource_parity.py +0 -0
- {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/tests/unit/test_search.py +0 -0
- {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/tests/unit/test_transport.py +0 -0
- {docforge_sdk-0.1.2 → docforge_sdk-0.3.0}/uv.lock +0 -0
|
@@ -84,8 +84,16 @@ htmlcov/
|
|
|
84
84
|
!src/docforge/tests/units/build/**
|
|
85
85
|
src/docforge/app/frontend/scripts/node_modules/
|
|
86
86
|
|
|
87
|
+
# UI-screenshot dev tooling — internal, kept private like .claude/ (Playwright-via-Docker recipe).
|
|
88
|
+
src/docforge/app/frontend/scripts/ui-shot.mjs
|
|
89
|
+
src/docforge/app/frontend/scripts/*.png
|
|
90
|
+
src/docforge/app/frontend/UI-SCREENSHOT.md
|
|
91
|
+
|
|
87
92
|
# Retired AI agent-memory snapshots — internal, not part of the public docs.
|
|
88
93
|
docs/archive/agent-memory-legacy/
|
|
89
94
|
|
|
90
95
|
# Root AI project instructions — local dev tooling, kept private (like .claude/).
|
|
91
96
|
/CLAUDE.md
|
|
97
|
+
|
|
98
|
+
# Docker Compose interpolation vars (copied from .env.example) — may hold local pins
|
|
99
|
+
/.env
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: docforge-sdk
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.3.0
|
|
4
4
|
Summary: Typed async + sync Python client for the DocForge REST API (standalone, zero server deps).
|
|
5
5
|
Project-URL: Homepage, https://github.com/Florian-BARRE/docforge
|
|
6
6
|
Project-URL: Repository, https://github.com/Florian-BARRE/docforge
|
|
@@ -322,8 +322,7 @@ with Client("http://localhost:10040", api_token="df_...") as client:
|
|
|
322
322
|
```
|
|
323
323
|
|
|
324
324
|
Key `SearchRequest` fields: `query` (required), `limit` (1–100, default 10), `filters`
|
|
325
|
-
(`dict[str, Any]`), `search_in` (`list[SearchTarget]` to pick which vectors to query)
|
|
326
|
-
`use_late_interaction`, `rescore_pool_size`.
|
|
325
|
+
(`dict[str, Any]`), `search_in` (`list[SearchTarget]` to pick which vectors to query).
|
|
327
326
|
|
|
328
327
|
### Explore a document (pages, IR, chunks)
|
|
329
328
|
|
|
@@ -299,8 +299,7 @@ with Client("http://localhost:10040", api_token="df_...") as client:
|
|
|
299
299
|
```
|
|
300
300
|
|
|
301
301
|
Key `SearchRequest` fields: `query` (required), `limit` (1–100, default 10), `filters`
|
|
302
|
-
(`dict[str, Any]`), `search_in` (`list[SearchTarget]` to pick which vectors to query)
|
|
303
|
-
`use_late_interaction`, `rescore_pool_size`.
|
|
302
|
+
(`dict[str, Any]`), `search_in` (`list[SearchTarget]` to pick which vectors to query).
|
|
304
303
|
|
|
305
304
|
### Explore a document (pages, IR, chunks)
|
|
306
305
|
|
|
@@ -78,7 +78,7 @@ from .models.pipelines import (
|
|
|
78
78
|
)
|
|
79
79
|
|
|
80
80
|
# ------------------- Search models ------------------- #
|
|
81
|
-
from .models.search import SearchHit, SearchRequest, SearchResponse, SearchTarget
|
|
81
|
+
from .models.search import BlockLocation, SearchHit, SearchRequest, SearchResponse, SearchTarget
|
|
82
82
|
|
|
83
83
|
# ------------------- Public API ------------------- #
|
|
84
84
|
__all__ = [
|
|
@@ -131,6 +131,7 @@ __all__ = [
|
|
|
131
131
|
# Search
|
|
132
132
|
"SearchTarget",
|
|
133
133
|
"SearchRequest",
|
|
134
|
+
"BlockLocation",
|
|
134
135
|
"SearchHit",
|
|
135
136
|
"SearchResponse",
|
|
136
137
|
# Jobs
|
|
@@ -62,7 +62,7 @@ from .pipelines import (
|
|
|
62
62
|
)
|
|
63
63
|
|
|
64
64
|
# ------------------- Search models ------------------- #
|
|
65
|
-
from .search import SearchHit, SearchRequest, SearchResponse, SearchTarget
|
|
65
|
+
from .search import BlockLocation, SearchHit, SearchRequest, SearchResponse, SearchTarget
|
|
66
66
|
|
|
67
67
|
# ------------------- Public API ------------------- #
|
|
68
68
|
__all__ = [
|
|
@@ -111,6 +111,7 @@ __all__ = [
|
|
|
111
111
|
# Search
|
|
112
112
|
"SearchTarget",
|
|
113
113
|
"SearchRequest",
|
|
114
|
+
"BlockLocation",
|
|
114
115
|
"SearchHit",
|
|
115
116
|
"SearchResponse",
|
|
116
117
|
# Jobs
|
|
@@ -5,7 +5,7 @@
|
|
|
5
5
|
|
|
6
6
|
# ====== Standard Library Imports ======
|
|
7
7
|
from datetime import datetime
|
|
8
|
-
from typing import Any
|
|
8
|
+
from typing import Any, Literal
|
|
9
9
|
|
|
10
10
|
# ====== Third-Party Library Imports ======
|
|
11
11
|
from pydantic import BaseModel, Field
|
|
@@ -56,6 +56,8 @@ class CollectionModel(BaseModel):
|
|
|
56
56
|
name (str): Unique human name.
|
|
57
57
|
supported_formats (list[str]): Accepted upload extensions (e.g. pdf).
|
|
58
58
|
max_file_size_bytes (int): Upload size ceiling, bytes.
|
|
59
|
+
job_timeout_seconds (float | None): Per-collection whole-ingest-job wall-clock budget,
|
|
60
|
+
seconds. None = inherit the worker's global WORKER_JOB_TIMEOUT_SECONDS default.
|
|
59
61
|
needs_reindex (bool): True when a config change requires reindexing.
|
|
60
62
|
created_at (datetime | None): Creation timestamp.
|
|
61
63
|
pipeline (dict[str, Any]): The ingestion pipeline blob (the graph).
|
|
@@ -67,6 +69,14 @@ class CollectionModel(BaseModel):
|
|
|
67
69
|
name: str = Field(description="Unique human name.")
|
|
68
70
|
supported_formats: list[str] = Field(description="Accepted upload extensions (e.g. pdf).")
|
|
69
71
|
max_file_size_bytes: int = Field(description="Upload size ceiling, bytes.")
|
|
72
|
+
job_timeout_seconds: float | None = Field(
|
|
73
|
+
default=None,
|
|
74
|
+
gt=0,
|
|
75
|
+
description=(
|
|
76
|
+
"Per-collection whole-ingest-job wall-clock budget, seconds. None = inherit the "
|
|
77
|
+
"worker's global WORKER_JOB_TIMEOUT_SECONDS default."
|
|
78
|
+
),
|
|
79
|
+
)
|
|
70
80
|
needs_reindex: bool = Field(description="True when a config change requires reindexing.")
|
|
71
81
|
created_at: datetime | None = Field(default=None, description="Creation timestamp.")
|
|
72
82
|
pipeline: dict[str, Any] = Field(description="The ingestion pipeline blob (the graph).")
|
|
@@ -84,6 +94,8 @@ class CreateCollectionRequest(BaseModel):
|
|
|
84
94
|
name (str): Unique human name.
|
|
85
95
|
supported_formats (list[str]): Accepted upload extensions (e.g. pdf).
|
|
86
96
|
max_file_size_bytes (int): Upload size ceiling, bytes.
|
|
97
|
+
job_timeout_seconds (float | None): Per-collection whole-ingest-job wall-clock budget,
|
|
98
|
+
seconds. None = inherit the worker's global WORKER_JOB_TIMEOUT_SECONDS default.
|
|
87
99
|
fields (list[FieldSpec]): The FULL schema, declared up front (vector space is fixed).
|
|
88
100
|
pipeline (dict[str, Any] | None): The pipeline blob; omitted → the product default.
|
|
89
101
|
"""
|
|
@@ -91,6 +103,14 @@ class CreateCollectionRequest(BaseModel):
|
|
|
91
103
|
name: str = Field(description="Unique human name.")
|
|
92
104
|
supported_formats: list[str] = Field(description="Accepted upload extensions (e.g. pdf).")
|
|
93
105
|
max_file_size_bytes: int = Field(description="Upload size ceiling, bytes.")
|
|
106
|
+
job_timeout_seconds: float | None = Field(
|
|
107
|
+
default=None,
|
|
108
|
+
gt=0,
|
|
109
|
+
description=(
|
|
110
|
+
"Per-collection whole-ingest-job wall-clock budget, seconds. None = inherit the "
|
|
111
|
+
"worker's global WORKER_JOB_TIMEOUT_SECONDS default."
|
|
112
|
+
),
|
|
113
|
+
)
|
|
94
114
|
fields: list[FieldSpec] = Field(
|
|
95
115
|
default_factory=list,
|
|
96
116
|
description="The FULL schema, declared up front (vector space is fixed at creation).",
|
|
@@ -99,6 +119,11 @@ class CreateCollectionRequest(BaseModel):
|
|
|
99
119
|
default=None,
|
|
100
120
|
description="The pipeline blob; omitted → the product default (all stages wired).",
|
|
101
121
|
)
|
|
122
|
+
preset: Literal["standard", "light"] | None = Field(
|
|
123
|
+
default=None,
|
|
124
|
+
description="Stock-blob selector (ignored when pipeline is set); 'light' = fast, "
|
|
125
|
+
"enrichment-free core.",
|
|
126
|
+
)
|
|
102
127
|
|
|
103
128
|
|
|
104
129
|
class UpdateCollectionRequest(BaseModel):
|
|
@@ -109,6 +134,8 @@ class UpdateCollectionRequest(BaseModel):
|
|
|
109
134
|
name (str | None): New unique name.
|
|
110
135
|
supported_formats (list[str] | None): New accepted upload extensions.
|
|
111
136
|
max_file_size_bytes (int | None): New size ceiling, bytes.
|
|
137
|
+
job_timeout_seconds (float | None): New per-collection whole-ingest-job wall-clock
|
|
138
|
+
budget, seconds. Omitted = leave the current value unchanged.
|
|
112
139
|
fields (list[FieldSpec] | None): The TARGET schema (diffed by field name).
|
|
113
140
|
pipeline (dict[str, Any] | None): New pipeline blob (validated before storage).
|
|
114
141
|
search (dict[str, Any] | None): New search graph blob ({} = stock default).
|
|
@@ -120,6 +147,14 @@ class UpdateCollectionRequest(BaseModel):
|
|
|
120
147
|
default=None, description="New accepted upload extensions."
|
|
121
148
|
)
|
|
122
149
|
max_file_size_bytes: int | None = Field(default=None, description="New size ceiling, bytes.")
|
|
150
|
+
job_timeout_seconds: float | None = Field(
|
|
151
|
+
default=None,
|
|
152
|
+
gt=0,
|
|
153
|
+
description=(
|
|
154
|
+
"New per-collection whole-ingest-job wall-clock budget, seconds. Omitted = leave the "
|
|
155
|
+
"current value unchanged; a set value overrides the global WORKER_JOB_TIMEOUT_SECONDS."
|
|
156
|
+
),
|
|
157
|
+
)
|
|
123
158
|
fields: list[FieldSpec] | None = Field(
|
|
124
159
|
default=None,
|
|
125
160
|
description="The TARGET schema (diffed by field name; omitted fields are removed).",
|
|
@@ -148,6 +148,8 @@ class ChunkInfo(BaseModel):
|
|
|
148
148
|
parent_id (str | None): Parent chunk (hierarchical chunking).
|
|
149
149
|
block_ids (list[str]): Composing IR block ids, in assembly order.
|
|
150
150
|
metadata (list[MetadataValue]): Per-chunk generated metadata values.
|
|
151
|
+
heading_path (list[str]): The chunk's section breadcrumb (outer→inner headings).
|
|
152
|
+
page (int | None): Page of the chunk's primary block; None when it has no located block.
|
|
151
153
|
"""
|
|
152
154
|
|
|
153
155
|
id: str = Field(description="The chunk UUID (doubles as the Qdrant point id).")
|
|
@@ -165,6 +167,14 @@ class ChunkInfo(BaseModel):
|
|
|
165
167
|
metadata: list[MetadataValue] = Field(
|
|
166
168
|
default_factory=list, description="Per-chunk generated metadata values."
|
|
167
169
|
)
|
|
170
|
+
heading_path: list[str] = Field(
|
|
171
|
+
default_factory=list,
|
|
172
|
+
description="The chunk's section breadcrumb (outer→inner headings); [] when none.",
|
|
173
|
+
)
|
|
174
|
+
page: int | None = Field(
|
|
175
|
+
default=None,
|
|
176
|
+
description="Page of the chunk's primary block; None when it has no located block.",
|
|
177
|
+
)
|
|
168
178
|
|
|
169
179
|
|
|
170
180
|
class ChunkEnabledPatch(BaseModel):
|
|
@@ -0,0 +1,165 @@
|
|
|
1
|
+
# ====== Code Summary ======
|
|
2
|
+
# Response models for the jobs resource, mirrored field-for-field from the DocForge backend router
|
|
3
|
+
# models: the live ingestion status, one node of the execution trace, the full trace, and the live
|
|
4
|
+
# per-worker activity view.
|
|
5
|
+
|
|
6
|
+
# ====== Standard Library Imports ======
|
|
7
|
+
from datetime import datetime
|
|
8
|
+
|
|
9
|
+
# ====== Third-Party Library Imports ======
|
|
10
|
+
from pydantic import BaseModel, Field
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
class JobStatus(BaseModel):
|
|
14
|
+
"""
|
|
15
|
+
One ingestion job's live state — written by the worker, only read here.
|
|
16
|
+
|
|
17
|
+
Attributes:
|
|
18
|
+
job_id (str): The job row's UUID.
|
|
19
|
+
document_id (str): The document being ingested.
|
|
20
|
+
collection_id (str): Its collection.
|
|
21
|
+
status (str): queued / running / done / failed.
|
|
22
|
+
progress (int): 0–100 (completed pipeline nodes over total).
|
|
23
|
+
current_stage (str | None): The node currently (or last) executed.
|
|
24
|
+
error (str | None): The failure, verbatim — only set when status is failed.
|
|
25
|
+
attempt (int): arq retry attempt (1 = first run).
|
|
26
|
+
started_at (datetime | None): When the worker picked it up.
|
|
27
|
+
finished_at (datetime | None): When it ended (done or failed).
|
|
28
|
+
items_done (int | None): Child items finished in the current fan-out stage (None off it).
|
|
29
|
+
items_total (int | None): The current fan-out stage's width (None when not in a fan-out).
|
|
30
|
+
failed_node_id (str | None): The deepest node that raised — only set on a failed job.
|
|
31
|
+
failed_node_kind (str | None): That node's kind/family label — only set on a failed job.
|
|
32
|
+
failed_item_index (int | None): The fan-out item index the failure sits in (None outside one).
|
|
33
|
+
error_type (str | None): The exception class name of the failure (e.g. "TimeoutError").
|
|
34
|
+
"""
|
|
35
|
+
|
|
36
|
+
job_id: str = Field(description="The job row's UUID.")
|
|
37
|
+
document_id: str = Field(description="The document being ingested.")
|
|
38
|
+
collection_id: str = Field(description="Its collection.")
|
|
39
|
+
status: str = Field(description="queued / running / done / failed.")
|
|
40
|
+
progress: int = Field(description="0-100, completed pipeline nodes over total.")
|
|
41
|
+
current_stage: str | None = Field(default=None, description="Node currently/last executed.")
|
|
42
|
+
error: str | None = Field(default=None, description="Failure detail when status=failed.")
|
|
43
|
+
attempt: int = Field(description="arq retry attempt (1 = first run).")
|
|
44
|
+
started_at: datetime | None = Field(default=None, description="Picked up by the worker at.")
|
|
45
|
+
finished_at: datetime | None = Field(default=None, description="Ended (done or failed) at.")
|
|
46
|
+
updated_at: datetime = Field(description="Last progress/lifecycle write (freezes on a wedge).")
|
|
47
|
+
stalled: bool = Field(
|
|
48
|
+
description="A RUNNING job idle past the stall threshold — an early wedge warning."
|
|
49
|
+
)
|
|
50
|
+
total_prompt_tokens: int = Field(
|
|
51
|
+
description="Prompt tokens billed across this job's paid text-gen calls."
|
|
52
|
+
)
|
|
53
|
+
total_completion_tokens: int = Field(
|
|
54
|
+
description="Completion tokens billed across this job's paid text-gen calls."
|
|
55
|
+
)
|
|
56
|
+
cost_usd: float = Field(
|
|
57
|
+
description="USD cost of this job's paid calls (0 when nothing priceable)."
|
|
58
|
+
)
|
|
59
|
+
items_done: int | None = Field(
|
|
60
|
+
default=None, description="Child items finished in the current fan-out stage (None off it)."
|
|
61
|
+
)
|
|
62
|
+
items_total: int | None = Field(
|
|
63
|
+
default=None, description="The current fan-out stage's width (None when not in a fan-out)."
|
|
64
|
+
)
|
|
65
|
+
failed_node_id: str | None = Field(
|
|
66
|
+
default=None, description="Deepest node that raised — only set on a failed job."
|
|
67
|
+
)
|
|
68
|
+
failed_node_kind: str | None = Field(
|
|
69
|
+
default=None, description="That node's kind/family label — only set on a failed job."
|
|
70
|
+
)
|
|
71
|
+
failed_item_index: int | None = Field(
|
|
72
|
+
default=None, description="Fan-out item index the failure sits in (None outside a fan-out)."
|
|
73
|
+
)
|
|
74
|
+
error_type: str | None = Field(
|
|
75
|
+
default=None, description="Exception class name of the failure (e.g. 'TimeoutError')."
|
|
76
|
+
)
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
class JobEvent(BaseModel):
|
|
80
|
+
"""
|
|
81
|
+
One node of the job's execution trace — written by the worker at each stage end.
|
|
82
|
+
|
|
83
|
+
Attributes:
|
|
84
|
+
stage (str): The pipeline node id.
|
|
85
|
+
status (str): success / failed / skipped.
|
|
86
|
+
node_kind (str | None): The stage's structural kind (action/group/foreach) or the node's
|
|
87
|
+
concrete kind — None for rows written before this column landed.
|
|
88
|
+
started_at (datetime | None): Node start.
|
|
89
|
+
finished_at (datetime | None): Node end.
|
|
90
|
+
detail (str | None): Duration, or the error when failed.
|
|
91
|
+
"""
|
|
92
|
+
|
|
93
|
+
stage: str = Field(description="The pipeline node id.")
|
|
94
|
+
status: str = Field(description="success / failed / skipped.")
|
|
95
|
+
node_kind: str | None = Field(
|
|
96
|
+
default=None,
|
|
97
|
+
description="The stage's structural kind (action/group/foreach) or the node's concrete "
|
|
98
|
+
"kind — None for rows written before this column landed.",
|
|
99
|
+
)
|
|
100
|
+
started_at: datetime | None = Field(default=None, description="Node start.")
|
|
101
|
+
finished_at: datetime | None = Field(default=None, description="Node end.")
|
|
102
|
+
detail: str | None = Field(default=None, description="Duration, or the error when failed.")
|
|
103
|
+
prompt_tokens: int | None = Field(
|
|
104
|
+
default=None, description="Prompt tokens billed by this stage; null when it made none."
|
|
105
|
+
)
|
|
106
|
+
completion_tokens: int | None = Field(
|
|
107
|
+
default=None, description="Completion tokens billed by this stage; null when none."
|
|
108
|
+
)
|
|
109
|
+
cost_usd: float | None = Field(
|
|
110
|
+
default=None, description="USD cost of this stage; null when no usage or unknown price."
|
|
111
|
+
)
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
class JobTrace(BaseModel):
|
|
115
|
+
"""
|
|
116
|
+
A job's full per-node trace, in execution order.
|
|
117
|
+
|
|
118
|
+
Attributes:
|
|
119
|
+
job_id (str): The traced job.
|
|
120
|
+
events (list[JobEvent]): One entry per stage run.
|
|
121
|
+
"""
|
|
122
|
+
|
|
123
|
+
job_id: str = Field(description="The traced job.")
|
|
124
|
+
events: list[JobEvent] = Field(default_factory=list, description="One entry per stage run.")
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
class WorkerActivity(BaseModel):
|
|
128
|
+
"""
|
|
129
|
+
One worker's live activity — its heartbeat-derived liveness plus any RUNNING jobs it owns.
|
|
130
|
+
|
|
131
|
+
Attributes:
|
|
132
|
+
worker_id (str): The worker's stable id (its hostname).
|
|
133
|
+
alive (bool): Its heartbeat is fresher than the liveness threshold.
|
|
134
|
+
busy (bool): It currently owns at least one RUNNING job.
|
|
135
|
+
last_seen (datetime | None): Its last heartbeat tick (None when no heartbeat row exists).
|
|
136
|
+
started_at (datetime | None): When the worker process registered (None when no heartbeat).
|
|
137
|
+
jobs (list[JobStatus]): Its running jobs, live.
|
|
138
|
+
"""
|
|
139
|
+
|
|
140
|
+
worker_id: str = Field(description="The worker's stable id (its hostname).")
|
|
141
|
+
alive: bool = Field(description="Heartbeat fresher than the liveness threshold.")
|
|
142
|
+
busy: bool = Field(description="Owns at least one RUNNING job right now.")
|
|
143
|
+
last_seen: datetime | None = Field(
|
|
144
|
+
default=None, description="Last heartbeat tick (None when the worker has no heartbeat row)."
|
|
145
|
+
)
|
|
146
|
+
started_at: datetime | None = Field(
|
|
147
|
+
default=None, description="When the worker process registered (None when no heartbeat)."
|
|
148
|
+
)
|
|
149
|
+
jobs: list[JobStatus] = Field(default_factory=list, description="Its running jobs, live.")
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
class WorkersLive(BaseModel):
|
|
153
|
+
"""
|
|
154
|
+
Everything running right now, grouped by worker — the monitoring view.
|
|
155
|
+
|
|
156
|
+
Attributes:
|
|
157
|
+
workers (list[WorkerActivity]): One entry per active worker.
|
|
158
|
+
"""
|
|
159
|
+
|
|
160
|
+
workers: list[WorkerActivity] = Field(
|
|
161
|
+
default_factory=list, description="One entry per active worker."
|
|
162
|
+
)
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
__all__ = ["JobStatus", "JobEvent", "JobTrace", "WorkerActivity", "WorkersLive"]
|
|
@@ -41,8 +41,6 @@ class SearchRequest(BaseModel):
|
|
|
41
41
|
limit (int): Number of fused results to return.
|
|
42
42
|
filters (dict[str, Any] | None): Exact/any-of constraints on the FILTERABLE fields.
|
|
43
43
|
search_in (list[SearchTarget] | None): Fields × modalities to search. None → content on both.
|
|
44
|
-
use_late_interaction (bool | None): Opt into the ColBERT re-score. None → off for this query.
|
|
45
|
-
rescore_pool_size (int | None): Size of the fused candidate pool the ColBERT stage re-scores.
|
|
46
44
|
"""
|
|
47
45
|
|
|
48
46
|
query: str = Field(min_length=1, description="The natural-language query to search for.")
|
|
@@ -56,15 +54,25 @@ class SearchRequest(BaseModel):
|
|
|
56
54
|
description="Fields × modalities to search (content and/or metadata). None → content on "
|
|
57
55
|
"both semantic and lexical (the unchanged default).",
|
|
58
56
|
)
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
class BlockLocation(BaseModel):
|
|
60
|
+
"""
|
|
61
|
+
One source block's location on the page — enough for a UI to draw a box over the hit.
|
|
62
|
+
|
|
63
|
+
Attributes:
|
|
64
|
+
page (int | None): The page the block sits on. None for a page-less document (no page
|
|
65
|
+
render) — distinct from a genuine 0-based page index 0.
|
|
66
|
+
bbox (list[float]): The block's bounding box ``[x0, y0, x1, y1]``, NORMALISED to [0, 1] —
|
|
67
|
+
multiply each component by the page image's width/height to draw it in pixels.
|
|
68
|
+
"""
|
|
69
|
+
|
|
70
|
+
page: int | None = Field(
|
|
71
|
+
description="The page the block sits on; None for a page-less document (no page render)."
|
|
62
72
|
)
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
le=1000,
|
|
67
|
-
description="Fused candidate pool size the ColBERT stage re-scores. None → node/store default.",
|
|
73
|
+
bbox: list[float] = Field(
|
|
74
|
+
description="Bounding box [x0, y0, x1, y1] NORMALISED to [0, 1] — multiply by the page "
|
|
75
|
+
"image width/height to draw it in pixels.",
|
|
68
76
|
)
|
|
69
77
|
|
|
70
78
|
|
|
@@ -79,14 +87,53 @@ class SearchHit(BaseModel):
|
|
|
79
87
|
text (str): The chunk's enriched text.
|
|
80
88
|
chunk_index (int): The chunk's ordinal within its document.
|
|
81
89
|
token_count (int): The chunk's token count.
|
|
90
|
+
block_ids (list[str]): The IR block ids the chunk was assembled from (assembly order).
|
|
91
|
+
page (int | None): The page of the chunk's primary (leading) block — draw the box here.
|
|
92
|
+
bbox (list[float] | None): The primary block's NORMALISED [0, 1] bounding box.
|
|
93
|
+
block_locations (list[BlockLocation]): Every source block's page + bbox (draw them all).
|
|
82
94
|
"""
|
|
83
95
|
|
|
84
96
|
chunk_id: str = Field(description="The chunk's UUID.")
|
|
85
97
|
document_id: str = Field(description="The owning document's UUID.")
|
|
98
|
+
filename: str | None = Field(
|
|
99
|
+
default=None, description="The source document's filename — the hit's human identity."
|
|
100
|
+
)
|
|
101
|
+
document_title: str | None = Field(
|
|
102
|
+
default=None, description="The source document's title (empty parsed titles → null)."
|
|
103
|
+
)
|
|
104
|
+
heading_path: list[str] = Field(
|
|
105
|
+
default_factory=list,
|
|
106
|
+
description="The chunk's section ancestry, top-down (e.g. ['Article 7 — Audit rights']) — "
|
|
107
|
+
"so a hit self-cites the section/clause it came from. Empty when the chunk sits under no "
|
|
108
|
+
"section.",
|
|
109
|
+
)
|
|
110
|
+
metadata: dict[str, Any] = Field(
|
|
111
|
+
default_factory=dict,
|
|
112
|
+
description="The document's filterable metadata (field → value) — so a hit self-cites "
|
|
113
|
+
"without a second GET /documents/{id}.",
|
|
114
|
+
)
|
|
86
115
|
score: float = Field(description="Fused RRF score (higher is better).")
|
|
87
116
|
text: str = Field(description="The chunk's enriched text.")
|
|
88
117
|
chunk_index: int = Field(description="Ordinal within the document.")
|
|
89
118
|
token_count: int = Field(description="Token count of the chunk.")
|
|
119
|
+
block_ids: list[str] = Field(
|
|
120
|
+
default_factory=list,
|
|
121
|
+
description="The IR block ids the chunk was assembled from, in assembly order.",
|
|
122
|
+
)
|
|
123
|
+
page: int | None = Field(
|
|
124
|
+
default=None,
|
|
125
|
+
description="The page of the chunk's primary (leading) block — where to draw the box. "
|
|
126
|
+
"None when the chunk carries no block location.",
|
|
127
|
+
)
|
|
128
|
+
bbox: list[float] | None = Field(
|
|
129
|
+
default=None,
|
|
130
|
+
description="The primary block's bounding box [x0, y0, x1, y1], NORMALISED to [0, 1] — "
|
|
131
|
+
"multiply by the page image width/height to draw it. None when unlocated.",
|
|
132
|
+
)
|
|
133
|
+
block_locations: list[BlockLocation] = Field(
|
|
134
|
+
default_factory=list,
|
|
135
|
+
description="Every source block's page + NORMALISED bbox — draw one box per block.",
|
|
136
|
+
)
|
|
90
137
|
|
|
91
138
|
|
|
92
139
|
class SearchResponse(BaseModel):
|
|
@@ -103,8 +150,8 @@ class SearchResponse(BaseModel):
|
|
|
103
150
|
hits: list[SearchHit] = Field(default_factory=list, description="Ranked hits, best first.")
|
|
104
151
|
debug_info: dict[str, Any] | None = Field(
|
|
105
152
|
default=None,
|
|
106
|
-
description="Non-fatal diagnostics
|
|
153
|
+
description="Non-fatal diagnostics about how the search ran. None when empty.",
|
|
107
154
|
)
|
|
108
155
|
|
|
109
156
|
|
|
110
|
-
__all__ = ["SearchTarget", "SearchRequest", "SearchHit", "SearchResponse"]
|
|
157
|
+
__all__ = ["SearchTarget", "SearchRequest", "BlockLocation", "SearchHit", "SearchResponse"]
|