fred-capability-documents 4.4.3__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,21 @@
1
+ # Copyright Thales 2026
2
+ #
3
+ # Licensed under the Apache License, Version 2.0 (the "License");
4
+ # you may not use this file except in compliance with the License.
5
+ # You may obtain a copy of the License at
6
+ #
7
+ # http://www.apache.org/licenses/LICENSE-2.0
8
+ #
9
+ # Unless required by applicable law or agreed to in writing, software
10
+ # distributed under the License is distributed on an "AS IS" BASIS,
11
+ # WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
12
+ # See the License for the specific language governing permissions and
13
+ # limitations under the License.
14
+
15
+ """Document capabilities over the Knowledge Flow ports.
16
+
17
+ One subpackage per capability, each registered as its own `fred.capabilities`
18
+ entry point in `pyproject.toml`. `document_read_common` is the one shared
19
+ module: config, pagination and error shaping the reading pair and
20
+ `document_similarity` have in common.
21
+ """
@@ -0,0 +1,28 @@
1
+ # Copyright Thales 2026
2
+ #
3
+ # Licensed under the Apache License, Version 2.0 (the "License");
4
+ # you may not use this file except in compliance with the License.
5
+ # You may obtain a copy of the License at
6
+ #
7
+ # http://www.apache.org/licenses/LICENSE-2.0
8
+ #
9
+ # Unless required by applicable law or agreed to in writing, software
10
+ # distributed under the License is distributed on an "AS IS" BASIS,
11
+ # WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
12
+ # See the License for the specific language governing permissions and
13
+ # limitations under the License.
14
+
15
+ """
16
+ `DocumentExtractCapability` (DOCREAD-01) — exhaustive, paginated document
17
+ extraction.
18
+
19
+ Installing fred-capability-documents registers it via the `fred.capabilities` entry point
20
+ (`document_extract`). Pairs with `document_verbatim` over the shared
21
+ `document_markdown` port; see `document_read_common`.
22
+ """
23
+
24
+ from __future__ import annotations
25
+
26
+ from .capability import DocumentExtractCapability
27
+
28
+ __all__ = ["DocumentExtractCapability"]
@@ -0,0 +1,208 @@
1
+ # Copyright Thales 2026
2
+ #
3
+ # Licensed under the Apache License, Version 2.0 (the "License");
4
+ # you may not use this file except in compliance with the License.
5
+ # You may obtain a copy of the License at
6
+ #
7
+ # http://www.apache.org/licenses/LICENSE-2.0
8
+ #
9
+ # Unless required by applicable law or agreed to in writing, software
10
+ # distributed under the License is distributed on an "AS IS" BASIS,
11
+ # WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
12
+ # See the License for the specific language governing permissions and
13
+ # limitations under the License.
14
+
15
+ """
16
+ `DocumentExtractCapability` (DOCREAD-01) — exhaustively extract information from a
17
+ document, nothing omitted.
18
+
19
+ Phase 2 (2026-08-07): the tool no longer pages the document into the agent's own
20
+ context (which made the agent burst many token-heavy model calls and trip the
21
+ provider's rate limit). It now makes ONE call to the `document_extraction` port,
22
+ which runs the exhaustive map-reduce server-side in Knowledge Flow — mapping over
23
+ EVERY chunk with bounded concurrency and 429 backoff, then de-duplicating without
24
+ compressing — and returns the consolidated list. `document_verbatim`'s
25
+ positional read stays on the paginated `document_markdown` port; only exhaustive
26
+ extraction moved server-side.
27
+
28
+ Admin-gated (class default), registered via the `document_extract`
29
+ `fred.capabilities` entry point.
30
+ """
31
+
32
+ from __future__ import annotations
33
+
34
+ import time
35
+ from collections.abc import Sequence
36
+
37
+ from fred_sdk.contracts.capability import (
38
+ AgentCapability,
39
+ CapabilityContext,
40
+ CapabilityManifest,
41
+ EmptyModel,
42
+ HitlGateRequest,
43
+ HitlSpec,
44
+ )
45
+ from fred_sdk.contracts.context import (
46
+ ToolContentBlock,
47
+ ToolContentKind,
48
+ ToolInvocationResult,
49
+ )
50
+ from fred_sdk.contracts.models import FieldSpec, UIHints
51
+ from fred_sdk.contracts.runtime import DocumentScopeRefusedError
52
+ from langchain_core.tools import BaseTool, tool
53
+ from pydantic import BaseModel
54
+
55
+ from fred_capability_documents.document_read_common import (
56
+ document_scope_refusal,
57
+ document_tool_failure,
58
+ )
59
+
60
+
61
+ class DocumentExtractConfig(BaseModel):
62
+ """Agent-creation / stored config of the document-extract capability.
63
+
64
+ `require_confirmation` gates each `extract_from_document` call behind a human
65
+ proceed/cancel (default on): extraction is token-heavy, so the confirmation
66
+ is shown BEFORE any LLM work is spent. An admin who trusts this agent's usage
67
+ can turn it off per instance. Same mechanism/field name as
68
+ `document_summarize` (RFC §5.4)."""
69
+
70
+ require_confirmation: bool = True
71
+
72
+
73
+ class DocumentExtractCapability(
74
+ AgentCapability[DocumentExtractConfig, DocumentExtractConfig, EmptyModel]
75
+ ):
76
+ """Exhaustive, server-side document extraction. Single tool, no chat
77
+ controls, no turn options; HITL-gated by default (token-heavy), configurable
78
+ per instance via `require_confirmation`."""
79
+
80
+ manifest = CapabilityManifest(
81
+ id="document_extract",
82
+ # Pre-GA: version stays 0.1.0 while the platform has not shipped.
83
+ version="0.1.0",
84
+ name="capability.document_extract.name",
85
+ description="capability.document_extract.description",
86
+ icon="find_in_page",
87
+ config_fields=[
88
+ FieldSpec(
89
+ key="require_confirmation",
90
+ type="boolean",
91
+ title="capability.document_extract.fields.require_confirmation.title",
92
+ description="capability.document_extract.fields.require_confirmation.description",
93
+ default=True,
94
+ ui=UIHints(group="safety"),
95
+ ),
96
+ ],
97
+ # team_scope left at the class default (ADMIN_GATED).
98
+ )
99
+ ConfigModel = DocumentExtractConfig
100
+
101
+ def tools(
102
+ self,
103
+ ctx: CapabilityContext[DocumentExtractConfig, EmptyModel],
104
+ ) -> Sequence[BaseTool]:
105
+ services = ctx.services
106
+
107
+ @tool("extract_from_document", response_format="content_and_artifact")
108
+ async def extract_from_document(
109
+ document_uid: str,
110
+ what_to_extract: str,
111
+ ) -> tuple[str, ToolInvocationResult]:
112
+ """Exhaustively extract information from a document, nothing omitted —
113
+ use instead of summarize_document whenever an item could be missed.
114
+
115
+ Use this when the user wants a COMPLETE, nothing-missed answer over a
116
+ whole document — e.g. "list ALL the requirements in this spec",
117
+ "every deadline", "each obligation and its owner". One call reads the
118
+ ENTIRE document (server-side) and returns a consolidated,
119
+ de-duplicated list of every matching item — you do NOT page through
120
+ the document yourself.
121
+
122
+ When to use a different tool instead:
123
+ - do NOT use summarize_document for this — a summary silently drops
124
+ items and gives a half-complete answer;
125
+ - to read one specific spot verbatim (e.g. "the first paragraph") →
126
+ read_document.
127
+
128
+ `what_to_extract` describes precisely what to enumerate (e.g.
129
+ "functional requirements", "dates and their surrounding context").
130
+ `document_uid` MUST be the document's opaque uid, not its name — get
131
+ it from a search hit's 'uid', the document tree, or the
132
+ conversation's attached-files list. NEVER repeat the uid in your
133
+ answer; refer to the document by its display name.
134
+
135
+ The returned list is already complete: present it to the user, do not
136
+ call this tool again for the same request.
137
+ """
138
+
139
+ port = services.document_extraction
140
+ if port is None:
141
+ raise RuntimeError(
142
+ "extract_from_document: RuntimeServices.document_extraction "
143
+ "is not available on this execution path."
144
+ )
145
+
146
+ started = time.monotonic()
147
+ try:
148
+ result = await port.extract(document_uid, instruction=what_to_extract)
149
+ except DocumentScopeRefusedError as exc:
150
+ return document_scope_refusal(
151
+ tool_ref="extract_from_document",
152
+ action="extract from the document",
153
+ exc=exc,
154
+ )
155
+ except Exception as exc:
156
+ message, artifact = document_tool_failure(
157
+ tool_ref="extract_from_document",
158
+ action="extract from the document",
159
+ exc=exc,
160
+ elapsed_s=time.monotonic() - started,
161
+ document_uid=document_uid,
162
+ )
163
+ return message, artifact
164
+
165
+ if result.item_count == 0:
166
+ content = (
167
+ f"No items matching “{what_to_extract}” were found in the document."
168
+ )
169
+ else:
170
+ content = result.extraction
171
+ if result.truncated:
172
+ content += (
173
+ "\n\n[Note: the document exceeded the processing cap and "
174
+ "was read head+tail; some middle content may be omitted.]"
175
+ )
176
+
177
+ artifact = ToolInvocationResult(
178
+ tool_ref="extract_from_document",
179
+ blocks=(ToolContentBlock(kind=ToolContentKind.TEXT, text=content),),
180
+ )
181
+ return content, artifact
182
+
183
+ return [extract_from_document]
184
+
185
+ def hitl_specs(self) -> Sequence[HitlSpec]:
186
+ """Gate `extract_from_document` behind a human proceed/cancel, on by
187
+ default and per-instance configurable via `require_confirmation`
188
+ (RFC §5.4), the same mechanism as `document_summarize`. `require=False`
189
+ defers to the `when` predicate, which reads the resolved instance config
190
+ fresh at gate time (`request.context.config`) — so it always reflects the
191
+ CURRENT `require_confirmation`, not whatever was set at assembly time.
192
+ The gate runs BEFORE the tool, so a cancel spends no extraction tokens."""
193
+
194
+ return [
195
+ HitlSpec(
196
+ tool="extract_from_document",
197
+ require=False,
198
+ when=_confirmation_required,
199
+ )
200
+ ]
201
+
202
+
203
+ def _confirmation_required(request: HitlGateRequest) -> bool:
204
+ """Whether this agent instance still wants a human's proceed/cancel before
205
+ `extract_from_document` runs — the resolved `require_confirmation` config
206
+ value (default `True`, see `DocumentExtractConfig`)."""
207
+
208
+ return bool(request.context.config.require_confirmation)
@@ -0,0 +1,30 @@
1
+ # Copyright Thales 2026
2
+ #
3
+ # Licensed under the Apache License, Version 2.0 (the "License");
4
+ # you may not use this file except in compliance with the License.
5
+ # You may obtain a copy of the License at
6
+ #
7
+ # http://www.apache.org/licenses/LICENSE-2.0
8
+ #
9
+ # Unless required by applicable law or agreed to in writing, software
10
+ # distributed under the License is distributed on an "AS IS" BASIS,
11
+ # WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
12
+ # See the License for the specific language governing permissions and
13
+ # limitations under the License.
14
+
15
+ """
16
+ `DocumentLabelSearchCapability` — exhaustive, paginated business-label
17
+ resolution, kept deliberately separate from `document_access`.
18
+
19
+ Installing fred-capability-documents registers it via the `fred.capabilities` entry point
20
+ (`document_label_search`), separate from `document_access` so a team admin
21
+ opts into it explicitly, per team.
22
+ """
23
+
24
+ from __future__ import annotations
25
+
26
+ from .capability import DocumentLabelSearchCapability
27
+
28
+ __all__ = [
29
+ "DocumentLabelSearchCapability",
30
+ ]
@@ -0,0 +1,249 @@
1
+ # Copyright Thales 2026
2
+ #
3
+ # Licensed under the Apache License, Version 2.0 (the "License");
4
+ # you may not use this file except in compliance with the License.
5
+ # You may obtain a copy of the License at
6
+ #
7
+ # http://www.apache.org/licenses/LICENSE-2.0
8
+ #
9
+ # Unless required by applicable law or agreed to in writing, software
10
+ # distributed under the License is distributed on an "AS IS" BASIS,
11
+ # WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
12
+ # See the License for the specific language governing permissions and
13
+ # limitations under the License.
14
+
15
+ """
16
+ `DocumentLabelSearchCapability` — exhaustive, paginated business-label
17
+ resolution.
18
+
19
+ Why this is its own capability, not a second tool on `document_access`:
20
+ - `document_access` is `DEFAULT_ON` — every agent that ships with it gets its
21
+ tool set for free. Adding a label-search tool there would mean every such
22
+ agent picks up a second, semantically-adjacent tool the moment it ships: a
23
+ harder tool-selection call for the model than the pre-existing
24
+ search-vs-tree split (both would answer "documents with label X",
25
+ differing only on the where/all-of-them axis), with a fail-quiet failure
26
+ mode (an incomplete-but-plausible-looking answer) instead of a fail-loud
27
+ one. Splitting into its own admin-gated capability means NO existing
28
+ agent's tool set changes; a team admin opts a specific agent into label
29
+ search deliberately, one team at a time.
30
+ - `list_document_tree` (`document_access`) does NOT filter by label at all,
31
+ by design (see its own docstring) — it renders a folder tree, and a size-
32
+ budgeted tree is the wrong response shape for "give me every document
33
+ labeled X". This capability's `list_documents_by_label` is the only
34
+ label-search surface: exhaustive, deterministic, paginated.
35
+
36
+ Doctrine (RFC §3.5, §3.8, §10) — same as `document_access`/
37
+ `document_summarize`:
38
+ - reaches the platform ONLY through the typed `RuntimeServices.document_tree`
39
+ port's `list_by_label` method (shared plumbing with `document_access`'s
40
+ `tree()` — one port, two capabilities, not two ports); the per-turn
41
+ binding and the raw access token NEVER enter `CapabilityContext`
42
+ - the tool signature exposes ONLY LLM arguments; scope and identity reach it
43
+ through the middleware closure, never the tool schema
44
+
45
+ Identifier hygiene (hard rule, shared with `document_access`): `document_uid`
46
+ is an internal working identifier for the agent's own tool calls (e.g.
47
+ chaining into `document_summarize`). The docstring instructs the model to
48
+ NEVER repeat it to the end user — answers refer to documents by display name
49
+ only.
50
+
51
+ KNOWN GAP (documented on `DocumentTreePort.list_by_label` itself): unlike
52
+ `document_access`'s `tree()`, this tool does not respect a `library_tag_ids`
53
+ binding — Knowledge Flow's label resolution narrows only by document-level
54
+ READ permission, never by folder/library scope. An agent bound to a specific
55
+ library set still sees every readable document carrying a label, corpus-
56
+ wide. Not silently papered over; narrowing this is follow-up work.
57
+ """
58
+
59
+ from __future__ import annotations
60
+
61
+ import time
62
+ from collections.abc import Sequence
63
+
64
+ from fred_sdk.contracts.capability import (
65
+ AgentCapability,
66
+ CapabilityContext,
67
+ CapabilityManifest,
68
+ EmptyModel,
69
+ )
70
+ from fred_sdk.contracts.context import (
71
+ ToolContentBlock,
72
+ ToolContentKind,
73
+ ToolInvocationResult,
74
+ )
75
+ from fred_sdk.contracts.runtime import DocumentLabelPageResult, unwrap_run_stop_error
76
+ from langchain_core.tools import BaseTool, tool
77
+
78
+ _KF_SERVICE = "Knowledge Flow"
79
+
80
+ # Wire bounds of the Knowledge Flow endpoint's `limit` validation — clamp
81
+ # client-side so an out-of-range LLM value degrades gracefully instead of
82
+ # 422ing.
83
+ _LABEL_PAGE_LIMIT_BOUNDS = (1, 500)
84
+
85
+
86
+ def _clamp(value: int, bounds: tuple[int, int]) -> int:
87
+ low, high = bounds
88
+ return max(low, min(value, high))
89
+
90
+
91
+ def _document_tool_failure(
92
+ *,
93
+ tool_ref: str,
94
+ action: str,
95
+ exc: Exception,
96
+ elapsed_s: float,
97
+ ) -> tuple[str, ToolInvocationResult]:
98
+ """Turn any label-search failure into a non-empty, actionable error
99
+ message plus an ``is_error=True`` artifact — same convention as
100
+ `document_access`/`document_summarize`'s `_document_tool_failure`
101
+ (duplicated, not imported: these capabilities are independently
102
+ installable/removable and must not import each other's internals)."""
103
+
104
+ err_type = type(exc).__name__
105
+ raw = str(exc).strip()
106
+ timed_out = bool(getattr(exc, "timed_out", False))
107
+ status_code = getattr(exc, "status_code", None)
108
+
109
+ if timed_out:
110
+ cause = f"the {_KF_SERVICE} service timed out after {elapsed_s:.0f}s"
111
+ elif status_code is not None:
112
+ cause = f"the {_KF_SERVICE} service returned HTTP {status_code}"
113
+ else:
114
+ cause = f"the {_KF_SERVICE} service call failed after {elapsed_s:.0f}s"
115
+
116
+ detail = f": {raw}" if raw else ""
117
+ message = f"Could not {action}: {cause} [{err_type}{detail}]."
118
+ return message, ToolInvocationResult(
119
+ tool_ref=tool_ref,
120
+ is_error=True,
121
+ blocks=(ToolContentBlock(kind=ToolContentKind.TEXT, text=message),),
122
+ )
123
+
124
+
125
+ class DocumentLabelSearchCapability(
126
+ AgentCapability[EmptyModel, EmptyModel, EmptyModel]
127
+ ):
128
+ """
129
+ Exhaustive, paginated business-label resolution — a single tool, no
130
+ config, no chat controls, no turn options. See the module docstring for
131
+ why this is its own admin-gated capability rather than living inside
132
+ `document_access`.
133
+ """
134
+
135
+ manifest = CapabilityManifest(
136
+ id="document_label_search",
137
+ # Pre-GA: version stays 0.1.0 while the platform has not shipped —
138
+ # config-surface changes land without bumps.
139
+ version="0.1.0",
140
+ name="capability.document_label_search.name",
141
+ description="capability.document_label_search.description",
142
+ icon="label",
143
+ # No new chat part / side panel / router / owned table.
144
+ # team_scope intentionally left at the class default (ADMIN_GATED,
145
+ # RFC §7, §8.3), same reasoning as `document_summarize`: this tool
146
+ # has no config-shaped trigger a user can reason about — a team
147
+ # admin must explicitly enable it before any agent in that team can
148
+ # even select it. That is the whole point of the split (see module
149
+ # docstring): zero blast radius on `document_access`'s existing,
150
+ # already-shipped baseline tool set.
151
+ )
152
+ ConfigModel = EmptyModel
153
+
154
+ def tools(
155
+ self,
156
+ ctx: CapabilityContext[EmptyModel, EmptyModel],
157
+ ) -> Sequence[BaseTool]:
158
+ """
159
+ Build the single label-search tool, bound to the turn's typed
160
+ context (RFC §3.2, §5). `AgentCapability.middleware()`'s default
161
+ wraps this for `create_agent()`; no ReAct-loop-specific hook is
162
+ needed.
163
+
164
+ Return-convention note:
165
+ kept as `@tool(..., response_format="content_and_artifact")`
166
+ returning a `(content, ToolInvocationResult)` tuple — see
167
+ `document_access`'s `tools()` docstring for the full rationale;
168
+ identical constraint here.
169
+ """
170
+
171
+ services = ctx.services
172
+
173
+ @tool("list_documents_by_label", response_format="content_and_artifact")
174
+ async def list_documents_by_label(
175
+ label: str,
176
+ offset: int = 0,
177
+ limit: int = 50,
178
+ ) -> tuple[str, ToolInvocationResult]:
179
+ """List EVERY document carrying a business label, exhaustively and page by page.
180
+
181
+ Call this when the user wants a complete answer about a label —
182
+ "list every DAT document", "how many MEX documents are there",
183
+ "find all documents tagged DAT". This tool is exhaustive and
184
+ deterministic: it is the only label-search surface available to
185
+ you, there is no folder-tree filter by label anywhere else.
186
+
187
+ Each call returns ONE page: matching documents as "name [uid]",
188
+ plus `total` (the full match count across all pages) and
189
+ `has_more`. When `has_more` is true, call again with
190
+ `offset=<next_offset from this response>` to continue. NEVER
191
+ report a count or a "complete" list to the user after a single
192
+ call without checking `has_more` — a label can match far more
193
+ documents than one page holds.
194
+
195
+ The bracketed uid is an internal working id for chaining into
196
+ other tools (e.g. summarize_document) only: NEVER repeat it in
197
+ your answer to the user — refer to documents by display name.
198
+ """
199
+
200
+ port = services.document_tree
201
+ if port is None:
202
+ raise RuntimeError(
203
+ "document_label_search: RuntimeServices.document_tree is "
204
+ "not available on this execution path."
205
+ )
206
+
207
+ effective_limit = _clamp(limit, _LABEL_PAGE_LIMIT_BOUNDS)
208
+ effective_offset = max(offset, 0)
209
+ started = time.monotonic()
210
+ try:
211
+ result: DocumentLabelPageResult = await port.list_by_label(
212
+ label=label,
213
+ offset=effective_offset,
214
+ limit=effective_limit,
215
+ )
216
+ except Exception as exc:
217
+ run_stop = unwrap_run_stop_error(exc)
218
+ if run_stop is not None:
219
+ raise run_stop from None
220
+ return _document_tool_failure(
221
+ tool_ref="list_documents_by_label",
222
+ action="list documents by label",
223
+ exc=exc,
224
+ elapsed_s=time.monotonic() - started,
225
+ )
226
+
227
+ if not result.documents:
228
+ text = f'No documents found with label "{result.label}".'
229
+ else:
230
+ lines = [
231
+ f'label="{result.label}" total={result.total} offset={result.offset} limit={result.limit}'
232
+ ]
233
+ lines.extend(
234
+ f"{doc.document_name} [{doc.document_uid}]"
235
+ for doc in result.documents
236
+ )
237
+ if result.has_more and result.next_offset is not None:
238
+ lines.append(
239
+ f"... {result.total - result.next_offset} more — call again with offset={result.next_offset} to continue."
240
+ )
241
+ text = "\n".join(lines)
242
+
243
+ artifact = ToolInvocationResult(
244
+ tool_ref="list_documents_by_label",
245
+ blocks=(ToolContentBlock(kind=ToolContentKind.TEXT, text=text),),
246
+ )
247
+ return text, artifact
248
+
249
+ return [list_documents_by_label]