fred-capability-document-access 4.4.3__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,36 @@
1
+ # Copyright Thales 2026
2
+ #
3
+ # Licensed under the Apache License, Version 2.0 (the "License");
4
+ # you may not use this file except in compliance with the License.
5
+ # You may obtain a copy of the License at
6
+ #
7
+ # http://www.apache.org/licenses/LICENSE-2.0
8
+ #
9
+ # Unless required by applicable law or agreed to in writing, software
10
+ # distributed under the License is distributed on an "AS IS" BASIS,
11
+ # WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
12
+ # See the License for the specific language governing permissions and
13
+ # limitations under the License.
14
+
15
+ """Scoped document access over the Knowledge Flow corpus (`document_access`).
16
+
17
+ Installing this package registers the capability through its
18
+ `fred.capabilities` entry point; see `capability.py` for the scoping
19
+ precedence and the deliberate deferrals.
20
+ """
21
+
22
+ from fred_capability_document_access.capability import (
23
+ DOCUMENT_ACCESS_TOOL_REF,
24
+ DocumentAccessCapability,
25
+ DocumentAccessConfig,
26
+ DocumentAccessTurnOptions,
27
+ narrow_scope_ids,
28
+ )
29
+
30
+ __all__ = [
31
+ "DOCUMENT_ACCESS_TOOL_REF",
32
+ "DocumentAccessCapability",
33
+ "DocumentAccessConfig",
34
+ "DocumentAccessTurnOptions",
35
+ "narrow_scope_ids",
36
+ ]
@@ -0,0 +1,781 @@
1
+ # Copyright Thales 2026
2
+ #
3
+ # Licensed under the Apache License, Version 2.0 (the "License");
4
+ # you may not use this file except in compliance with the License.
5
+ # You may obtain a copy of the License at
6
+ #
7
+ # http://www.apache.org/licenses/LICENSE-2.0
8
+ #
9
+ # Unless required by applicable law or agreed to in writing, software
10
+ # distributed under the License is distributed on an "AS IS" BASIS,
11
+ # WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
12
+ # See the License for the specific language governing permissions and
13
+ # limitations under the License.
14
+
15
+ """
16
+ `DocumentAccessCapability` (RFC §3, §10) — the canonical real capability.
17
+
18
+ Why this module exists:
19
+ - it is the first REAL (non-tracer) capability: live document tools wired to
20
+ platform services through typed SDK ports, plus static config-field scoping
21
+ and one computed chat-turn narrowing control
22
+ - it doubles as the canonical in-tree reference a capability author copies
23
+
24
+ What this capability ships (and deliberately does NOT):
25
+ - `search_documents_using_vectorization`, the vector-search tool, wired live
26
+ through `ctx.services.document_search`
27
+ - `list_document_tree`: wired live through `ctx.services.document_tree`,
28
+ backed by Knowledge Flow's document tree endpoint. Failures surface as
29
+ `is_error` tool results with actionable detail (timeout/HTTP status), never
30
+ raised exceptions.
31
+ - NOT `summarize_document` — that tool is its own capability,
32
+ `document_summarize` (`fred-capability-documents`, its own package),
33
+ admin-gated and HITL-gated separately (RFC §10.1). `list_document_tree`'s
34
+ emitted uids are commonly chained into that capability's tool, but the two
35
+ are independently selectable — this capability does not depend on it.
36
+ - still deferred: the tree's trailing "Session attachments" section — a
37
+ pod-reachable session-attachment enumeration does not exist yet
38
+ (attachments are only a *search* scope today). The tree tool therefore lists
39
+ the corpus only, and is registered only while `team_documents` is on.
40
+
41
+ Identifier hygiene (hard rule): document uids and tag ids are internal working
42
+ identifiers for the agent's own tool calls (scope filters, cross-capability
43
+ chaining into `document_summarize`). The agent uses them freely, but every
44
+ LLM-facing docstring instructs the model to NEVER repeat them to the end
45
+ user — answers refer to documents by display name only.
46
+
47
+ Doctrine (RFC §3.5, §3.8, §10):
48
+ - the capability reaches the platform ONLY through typed optional ports on
49
+ `RuntimeServices`; the per-turn binding and the raw access token NEVER enter
50
+ `CapabilityContext`
51
+ - the tool signature exposes ONLY LLM arguments (`question`, `top_k`); scope and
52
+ identity reach the tool through the middleware closure, never the tool schema
53
+
54
+ Scoping precedence (`turn_option ⊆ capability_config ⊆ session_binding`):
55
+ - HERE both tools narrow their stored-config scope (`config.library_tag_ids` /
56
+ `config.document_uids`) by the per-turn `document_scope` selection
57
+ (`turn_options`), enforcing `turn_option ⊆ capability_config`; a library pick
58
+ and a document pick each narrow on their own, and UNION when both are made -
59
+ "that library, plus that file", the rule Knowledge Flow already applies to
60
+ search hits;
61
+ - the runtime adapter then bounds the result by the session binding's own scope,
62
+ enforcing `⊆ session_binding` (see `DocumentSearchAdapter`).
63
+
64
+ The builtin `knowledge.search` (`TOOL_REF_KNOWLEDGE_SEARCH`) also exposes vector
65
+ search, scoped from `RuntimeContext`. Prefer this capability for per-capability
66
+ configuration and turn scoping; do not wire both search tools on one instance.
67
+ """
68
+
69
+ from __future__ import annotations
70
+
71
+ import json
72
+ import logging
73
+ import time
74
+ from collections.abc import Sequence
75
+
76
+ from fred_core.store.vector_search import (
77
+ DEFAULT_MIN_SOURCE_SCORE_RATIO,
78
+ select_citable_sources,
79
+ )
80
+ from fred_sdk.contracts.capability import (
81
+ AgentCapability,
82
+ CapabilityContext,
83
+ CapabilityManifest,
84
+ ChatControlSpec,
85
+ ScopePrivate,
86
+ TeamScopePolicy,
87
+ )
88
+ from fred_sdk.contracts.context import (
89
+ ToolContentBlock,
90
+ ToolContentKind,
91
+ ToolInvocationResult,
92
+ )
93
+ from fred_sdk.contracts.models import (
94
+ DocumentScopeControlParams,
95
+ FieldSpec,
96
+ RagScopeControlParams,
97
+ SearchPolicyControlParams,
98
+ UIHints,
99
+ )
100
+ from fred_sdk.contracts.runtime import (
101
+ DocumentSearchResult,
102
+ DocumentTreeResult,
103
+ unwrap_run_stop_error,
104
+ )
105
+ from langchain_core.tools import BaseTool, tool
106
+ from pydantic import BaseModel, Field, TypeAdapter, model_validator
107
+
108
+ # The tool-result `tool_ref` this capability stamps on its artifact — distinct
109
+ # from the builtin `knowledge.search` ref so the two paths stay traceable apart.
110
+ DOCUMENT_ACCESS_TOOL_REF = "document_access"
111
+
112
+ # Vector-search policy values Knowledge Flow accepts (mirrors the MCP catalog's
113
+ # `chat_options.search_policy` enum). None on the config means "let the session
114
+ # binding decide".
115
+ _SEARCH_POLICIES = ("strict", "hybrid", "semantic")
116
+ _RAG_SCOPES = ("corpus_only", "hybrid", "general_only")
117
+
118
+ # Only the fields the LLM needs for citation, reasoning, and tool chaining are
119
+ # exposed to the model. URL and operational fields are excluded so the model
120
+ # cannot reproduce broken or internal paths. `uid` stays: it is the working
121
+ # identifier the (separate) document_summarize capability's tool takes —
122
+ # internal to the agent, never repeated to the end user (each tool docstring
123
+ # says so).
124
+ _LLM_FIELDS = frozenset(
125
+ {"uid", "title", "content", "file_name", "page", "section", "score"}
126
+ )
127
+
128
+ _KF_SERVICE = "Knowledge Flow"
129
+
130
+ logger = logging.getLogger(__name__)
131
+
132
+ _TREE_MAX_CHARS_BOUNDS = (500, 20_000)
133
+
134
+
135
+ def _clamp(value: int, bounds: tuple[int, int]) -> int:
136
+ low, high = bounds
137
+ return max(low, min(value, high))
138
+
139
+
140
+ def _document_tool_failure(
141
+ *,
142
+ tool_ref: str,
143
+ action: str,
144
+ exc: Exception,
145
+ elapsed_s: float,
146
+ document_uid: str | None = None,
147
+ ) -> tuple[str, ToolInvocationResult]:
148
+ """Turn any document tool-call failure into a non-empty, actionable error
149
+ message plus an ``is_error=True`` artifact.
150
+
151
+ The v2 ReAct runtime surfaces ``ToolInvocationResult.is_error`` directly to
152
+ the user (and suppresses LLM hallucination), so a failing tool MUST return
153
+ such a result instead of raising — a raised exception is re-raised by the
154
+ default ``ToolNode`` handler, which leaves the tool call pending in the
155
+ trace and yields an empty error detail to the UI.
156
+
157
+ Transport detail (timeout, HTTP status) arrives via the SDK-typed
158
+ `DocumentPortCallError` attributes the adapters stamp — this module never
159
+ imports the adapter's HTTP stack.
160
+ """
161
+
162
+ # Server-side, with the stack: the model-facing message below deliberately
163
+ # compresses the failure, so without this line a programming error caught
164
+ # by the broad handlers (a TypeError from a renamed port kwarg, say) would
165
+ # degrade into a plausible "service call failed" and never surface anywhere
166
+ # a developer looks. Degrading is right for the turn; being silent about
167
+ # WHY is not.
168
+ logger.error(
169
+ "Document tool failure (%s, %.1fs) — degraded to an is_error artifact.",
170
+ action,
171
+ elapsed_s,
172
+ exc_info=exc,
173
+ )
174
+
175
+ err_type = type(exc).__name__
176
+ raw = str(exc).strip()
177
+ timed_out = bool(getattr(exc, "timed_out", False))
178
+ status_code = getattr(exc, "status_code", None)
179
+
180
+ # `structured` means the adapter identified the failure, so `cause` below
181
+ # already states it precisely and `raw` only repeats it — for a 401 that
182
+ # produced "returned HTTP 401 [DocumentPortCallError: Client error '401
183
+ # Unauthorized' for url ...]". Since the adapter now redacts URLs out of
184
+ # `raw`, what survived was worse than redundant: an unbalanced quote and
185
+ # "For more information check:" pointing at nothing.
186
+ #
187
+ # The unstructured branch keeps both, because there `raw` is the only
188
+ # information there is ("All connection attempts failed") and `err_type`
189
+ # is how an unexpected exception — a TypeError from a renamed port kwarg,
190
+ # the case the logger above exists for — reaches someone who can act on it.
191
+ structured = timed_out or status_code is not None
192
+
193
+ if timed_out:
194
+ cause = f"the {_KF_SERVICE} service timed out after {elapsed_s:.0f}s"
195
+ elif status_code is not None:
196
+ cause = f"the {_KF_SERVICE} service returned HTTP {status_code}"
197
+ else:
198
+ cause = f"the {_KF_SERVICE} service call failed after {elapsed_s:.0f}s"
199
+
200
+ target = f" (document_uid={document_uid})" if document_uid else ""
201
+ if structured:
202
+ message = f"Could not {action}{target}: {cause}."
203
+ else:
204
+ detail = f": {raw}" if raw else ""
205
+ message = f"Could not {action}{target}: {cause} [{err_type}{detail}]."
206
+ # `blocks` carries the same diagnostic as `content` (CAPAB-02, same reason
207
+ # as the success-path artifacts above): a Graph agent's plain-dict
208
+ # invocation keeps only the artifact half of a `content_and_artifact`
209
+ # return — an artifact with `is_error=True` but no message tells a Graph
210
+ # node THAT the call failed but not WHY.
211
+ return message, ToolInvocationResult(
212
+ tool_ref=tool_ref,
213
+ is_error=True,
214
+ blocks=(ToolContentBlock(kind=ToolContentKind.TEXT, text=message),),
215
+ )
216
+
217
+
218
+ def narrow_scope_ids(
219
+ outer: Sequence[str] | None, inner: Sequence[str] | None
220
+ ) -> list[str] | None:
221
+ """
222
+ Bound one scope level (`inner`) by a broader one (`outer`) — the capability
223
+ half of the scoping precedence.
224
+
225
+ Semantics (empty/None = "no bound at this level"):
226
+ - `inner` empty → inherit `outer` unchanged;
227
+ - `outer` empty → `outer` is unbounded, so keep `inner` as-is;
228
+ - both present → intersection, so the result is a subset of BOTH.
229
+
230
+ Used here as `narrow_scope_ids(capability_config, turn_option)` to enforce
231
+ `turn_option ⊆ capability_config`; the adapter applies the SAME primitive as
232
+ `narrow_scope_ids(session_binding, params)` to complete
233
+ `turn_option ⊆ capability_config ⊆ session_binding`.
234
+ """
235
+
236
+ if not inner:
237
+ return list(outer) if outer else None
238
+ if not outer:
239
+ return list(inner)
240
+ allowed = set(outer)
241
+ return [value for value in inner if value in allowed]
242
+
243
+
244
+ # Legacy keys parse like the fields they replace: "false" must stay false.
245
+ _LEGACY_BOOL = TypeAdapter(bool)
246
+
247
+
248
+ class DocumentAccessConfig(BaseModel):
249
+ """
250
+ Agent-creation / stored config of the document-access capability (RFC §3.2),
251
+ mirroring the legacy MCP search tool's configuration surface exactly.
252
+
253
+ `bind_libraries` + `library_tag_ids` pin the agent to a fixed library set
254
+ (the bound ids are IGNORED while `bind_libraries` is off, like the legacy
255
+ tool); `document_uids` NARROWS to specific documents. An empty list means
256
+ "no capability-side narrowing at this level" (the session binding still
257
+ bounds it). `default_top_k` and `search_policy` set retrieval defaults;
258
+ `attachments` and `team_documents` are the two document sources (at least
259
+ one must stay on); the `show_*` toggles pick which computed chat controls
260
+ the composer shows (library/document scope, search policy, RAG scope). When
261
+ the search-policy picker is shown, `search_policy` acts as the picker's
262
+ DEFAULT and the per-turn choice (RuntimeContext) wins at search time;
263
+ when hidden, it is enforced as-is. `min_source_score_ratio` bounds what's
264
+ citable as a "source" in the chat UI (RAG-DATASET-DISCOVERY-RFC.md §7) —
265
+ it never narrows what the model itself sees, only what a human is shown
266
+ as evidence.
267
+ """
268
+
269
+ library_tag_ids: ScopePrivate[list[str]] = []
270
+ document_uids: ScopePrivate[list[str]] = []
271
+ default_top_k: int = 8
272
+ search_policy: str | None = None
273
+ # The switch and its list form one team-owned setting: a copy that kept the
274
+ # switch on with an empty list would hide the library picker for nothing.
275
+ bind_libraries: ScopePrivate[bool] = False
276
+ show_library_selection: bool = True
277
+ show_document_selection: bool = True
278
+ # Sources: `attachments` drives the paperclip and the session search scope,
279
+ # `team_documents` the corpus scope, the scope pickers and the tree tool.
280
+ attachments: bool = True
281
+ team_documents: bool = True
282
+ show_search_policy_control: bool = True
283
+ show_rag_scope_control: bool = True
284
+ default_rag_scope: str | None = None
285
+ min_source_score_ratio: float = Field(
286
+ default=DEFAULT_MIN_SOURCE_SCORE_RATIO,
287
+ ge=0.0,
288
+ le=1.0,
289
+ description=(
290
+ "A hit must score at least this fraction of the best hit in the "
291
+ "same search call to be citable as a source. Does not affect what "
292
+ "the model itself can read — only the human-facing Sources panel."
293
+ ),
294
+ )
295
+
296
+ @model_validator(mode="before")
297
+ @classmethod
298
+ def _upgrade_legacy_slices(cls, data: object) -> object:
299
+ """Older slices revalidate without behavior change: the single scope
300
+ toggle maps onto the split library/document toggles, a pre-`bind_libraries`
301
+ library scope stays binding, and the paperclip/attachments-only pair maps
302
+ onto the two sources (legacy keys are dropped so they never round-trip)."""
303
+
304
+ if isinstance(data, dict):
305
+ data = dict(data)
306
+ paperclip = _LEGACY_BOOL.validate_python(
307
+ data.pop("show_attach_files_control", True)
308
+ )
309
+ only_attached = _LEGACY_BOOL.validate_python(
310
+ data.pop("search_attachments_only", False)
311
+ )
312
+ if "attachments" not in data and "team_documents" not in data:
313
+ data["attachments"] = paperclip
314
+ data["team_documents"] = not (paperclip and only_attached)
315
+ if (
316
+ "show_document_scope_control" in data
317
+ and "show_library_selection" not in data
318
+ and "show_document_selection" not in data
319
+ ):
320
+ shown = bool(data.get("show_document_scope_control"))
321
+ data["show_library_selection"] = shown
322
+ data["show_document_selection"] = shown
323
+ if "bind_libraries" not in data and data.get("library_tag_ids"):
324
+ data["bind_libraries"] = True
325
+ return data
326
+
327
+ @model_validator(mode="after")
328
+ def _require_a_source(self) -> DocumentAccessConfig:
329
+ if not (self.attachments or self.team_documents):
330
+ raise ValueError(
331
+ "document_access needs at least one source: attachments or "
332
+ "team_documents."
333
+ )
334
+ return self
335
+
336
+
337
+ class DocumentAccessTurnOptions(BaseModel):
338
+ """
339
+ Per-turn narrowing carried by the `document_scope` chat control (RFC §3.5).
340
+
341
+ Each field is `None` when the turn does not narrow that level; a present list
342
+ is intersected with the capability config scope (never widening it).
343
+ """
344
+
345
+ library_tag_ids: list[str] | None = None
346
+ document_uids: list[str] | None = None
347
+
348
+
349
+ class DocumentAccessCapability(
350
+ AgentCapability[
351
+ DocumentAccessConfig, DocumentAccessConfig, DocumentAccessTurnOptions
352
+ ]
353
+ ):
354
+ """
355
+ Document access over the platform corpus: vector search and tree listing,
356
+ wired through the `document_search` / `document_tree` ports. Config-field
357
+ scoping + one computed chat-turn narrowing control. See the module
358
+ docstring for the remaining deferral (session-attachment enumeration),
359
+ the duplicate-search-tool decision, and why `summarize_document` is a
360
+ separate capability (`document_summarize`).
361
+ """
362
+
363
+ manifest = CapabilityManifest(
364
+ id="document_access",
365
+ # Bump on every config-surface change: the version keys the stored-slice
366
+ # schema_version and the control-plane chat-controls cache.
367
+ version="0.2.0",
368
+ name="capability.document_access.name",
369
+ description="capability.document_access.description",
370
+ icon="find_in_page",
371
+ config_fields=[
372
+ FieldSpec(
373
+ key="attachments",
374
+ type="boolean",
375
+ title="capability.document_access.fields.attachments.title",
376
+ description="capability.document_access.fields.attachments.description",
377
+ default=True,
378
+ # `ui.group` drives the form's visual sections: the renderer
379
+ # draws a thin divider whenever the group changes between two
380
+ # consecutive visible fields.
381
+ ui=UIHints(group="sources"),
382
+ ),
383
+ FieldSpec(
384
+ key="team_documents",
385
+ type="boolean",
386
+ title="capability.document_access.fields.team_documents.title",
387
+ description="capability.document_access.fields.team_documents.description",
388
+ default=True,
389
+ ui=UIHints(group="sources"),
390
+ ),
391
+ FieldSpec(
392
+ key="show_library_selection",
393
+ type="boolean",
394
+ title="capability.document_access.fields.show_library_selection.title",
395
+ description="capability.document_access.fields.show_library_selection.description",
396
+ default=True,
397
+ ui=UIHints(group="scope", visible_when="team_documents"),
398
+ ),
399
+ FieldSpec(
400
+ key="bind_libraries",
401
+ type="boolean",
402
+ title="capability.document_access.fields.bind_libraries.title",
403
+ description="capability.document_access.fields.bind_libraries.description",
404
+ default=False,
405
+ ui=UIHints(group="scope", visible_when="team_documents"),
406
+ ),
407
+ FieldSpec(
408
+ key="library_tag_ids",
409
+ type="array",
410
+ item_type="string",
411
+ title="capability.document_access.fields.library_tag_ids.title",
412
+ description="capability.document_access.fields.library_tag_ids.description",
413
+ # Library/document tree picker, only shown while the binding
414
+ # toggle above is on (the ids are ignored otherwise).
415
+ ui=UIHints(
416
+ widget="document_libraries",
417
+ visible_when="bind_libraries",
418
+ group="scope",
419
+ ),
420
+ ),
421
+ FieldSpec(
422
+ key="show_document_selection",
423
+ type="boolean",
424
+ title="capability.document_access.fields.show_document_selection.title",
425
+ description="capability.document_access.fields.show_document_selection.description",
426
+ default=True,
427
+ ui=UIHints(group="scope", visible_when="team_documents"),
428
+ ),
429
+ FieldSpec(
430
+ key="default_top_k",
431
+ type="integer",
432
+ title="capability.document_access.fields.default_top_k.title",
433
+ description="capability.document_access.fields.default_top_k.description",
434
+ default=8,
435
+ min=1,
436
+ ui=UIHints(group="retrieval", advanced=True),
437
+ ),
438
+ FieldSpec(
439
+ key="min_source_score_ratio",
440
+ type="number",
441
+ title="capability.document_access.fields.min_source_score_ratio.title",
442
+ description="capability.document_access.fields.min_source_score_ratio.description",
443
+ default=DEFAULT_MIN_SOURCE_SCORE_RATIO,
444
+ min=0.0,
445
+ max=1.0,
446
+ ui=UIHints(group="retrieval", advanced=True),
447
+ ),
448
+ FieldSpec(
449
+ key="show_search_policy_control",
450
+ type="boolean",
451
+ title="capability.document_access.fields.show_search_policy_control.title",
452
+ description="capability.document_access.fields.show_search_policy_control.description",
453
+ default=True,
454
+ ui=UIHints(group="search_policy", advanced=True),
455
+ ),
456
+ FieldSpec(
457
+ key="search_policy",
458
+ type="select",
459
+ enum=list(_SEARCH_POLICIES),
460
+ title="capability.document_access.fields.search_policy.title",
461
+ description="capability.document_access.fields.search_policy.description",
462
+ ui=UIHints(group="search_policy", advanced=True),
463
+ ),
464
+ FieldSpec(
465
+ key="show_rag_scope_control",
466
+ type="boolean",
467
+ title="capability.document_access.fields.show_rag_scope_control.title",
468
+ description="capability.document_access.fields.show_rag_scope_control.description",
469
+ default=True,
470
+ ui=UIHints(group="rag_scope", advanced=True),
471
+ ),
472
+ FieldSpec(
473
+ key="default_rag_scope",
474
+ type="select",
475
+ enum=list(_RAG_SCOPES),
476
+ title="capability.document_access.fields.default_rag_scope.title",
477
+ description="capability.document_access.fields.default_rag_scope.description",
478
+ ui=UIHints(group="rag_scope", advanced=True),
479
+ ),
480
+ ],
481
+ # No new chat part / side panel / router / owned table — the pilot's
482
+ # smallest real surface (RFC §10). team_scope=default_on: baseline
483
+ # document access should work without a per-team admin gate (RFC §8.3).
484
+ team_scope=TeamScopePolicy.DEFAULT_ON,
485
+ )
486
+ ConfigModel = DocumentAccessConfig
487
+ TurnOptionsModel = DocumentAccessTurnOptions
488
+
489
+ def chat_controls(self, config: DocumentAccessConfig) -> list[ChatControlSpec]:
490
+ """
491
+ The stock composer controls (RFC §3.3), each behind its config toggle —
492
+ the same widget set the legacy MCP search tool emits, so both paths
493
+ offer the same chat surface. `bound_library_ids` pins the scope picker
494
+ to the capability's configured library scope (read-only) when one is
495
+ set. Search-policy/RAG-scope choices travel on `RuntimeContext`, which
496
+ the document-search adapter already honors.
497
+ """
498
+
499
+ controls: list[ChatControlSpec] = []
500
+ if config.attachments:
501
+ controls.append(ChatControlSpec(widget="attach_files"))
502
+ # Same visibility algebra as the legacy MCP tool: binding replaces the
503
+ # free library picker with a read-only pinned list; the document picker
504
+ # is independent. Without team documents there is no corpus to scope.
505
+ corpus = config.team_documents
506
+ bound = (
507
+ (config.library_tag_ids or None)
508
+ if corpus and config.bind_libraries
509
+ else None
510
+ )
511
+ show_libraries = (
512
+ corpus and not config.bind_libraries and config.show_library_selection
513
+ )
514
+ show_documents = corpus and config.show_document_selection
515
+ if show_libraries or show_documents or bound:
516
+ controls.append(
517
+ ChatControlSpec(
518
+ widget="document_scope",
519
+ params=DocumentScopeControlParams(
520
+ libraries=show_libraries or bool(bound),
521
+ documents=show_documents,
522
+ bound_library_ids=bound,
523
+ ),
524
+ )
525
+ )
526
+ if config.show_search_policy_control:
527
+ controls.append(
528
+ ChatControlSpec(
529
+ widget="search_policy",
530
+ params=(
531
+ SearchPolicyControlParams(default=config.search_policy) # type: ignore[arg-type]
532
+ if config.search_policy in _SEARCH_POLICIES
533
+ else SearchPolicyControlParams()
534
+ ),
535
+ )
536
+ )
537
+ if config.show_rag_scope_control:
538
+ default = config.default_rag_scope
539
+ controls.append(
540
+ ChatControlSpec(
541
+ widget="rag_scope",
542
+ params=RagScopeControlParams(
543
+ default=default if default in _RAG_SCOPES else "hybrid", # type: ignore[arg-type]
544
+ ),
545
+ )
546
+ )
547
+ return controls
548
+
549
+ def tools(
550
+ self,
551
+ ctx: CapabilityContext[DocumentAccessConfig, DocumentAccessTurnOptions],
552
+ ) -> Sequence[BaseTool]:
553
+ """
554
+ Build the single vector-search tool, bound to the turn's typed
555
+ context (RFC §3.2, §5). This is the ONLY runtime contribution of this
556
+ capability — `AgentCapability.middleware()`'s default wraps this for
557
+ `create_agent()`; no ReAct-loop-specific hook is needed.
558
+
559
+ Return-convention note:
560
+ kept as `@tool(..., response_format="content_and_artifact")` returning
561
+ a `(content, ToolInvocationResult)` tuple. Verified empirically
562
+ (`test_capability_tool_return_convention.py`) that this is correct for the only
563
+ execution path this tool goes through today — `create_agent()`'s real
564
+ ToolCall-based tool-calling loop, which builds a `ToolMessage` whose
565
+ `.artifact` carries the `ToolInvocationResult` (and its `.sources`)
566
+ intact. A plain-dict `.ainvoke()` call (the shape Graph's
567
+ `invoke_runtime_tool` and the MCP runtime-provider resolver both use)
568
+ does NOT preserve this: LangChain collapses a `content_and_artifact`
569
+ response to the bare content string with NO tuple and NO artifact at
570
+ all when there is no `ToolCall` to attach it to — worse than the
571
+ tuple-collapse the original plan assumed. Switching to a bare
572
+ `ToolInvocationResult` return (the runtime-provider convention)
573
+ would fix that path but breaks THIS one: without
574
+ `response_format="content_and_artifact"`, `create_agent()`'s ToolCall
575
+ loop stringifies the whole model into `ToolMessage.content` and never
576
+ populates `.artifact`. Since Phase 1 does not wire this tool into any
577
+ plain-dict invocation path (that's Phase 4), the existing convention
578
+ is correct as-is; Phase 2+ must adapt at the tool-carrier/assembly
579
+ seam rather than change this tool's return shape again.
580
+ """
581
+
582
+ config = ctx.config
583
+ turn = ctx.turn_options
584
+ services = ctx.services
585
+
586
+ # Capability-config ∩ turn-option → the params handed to the port. This
587
+ # enforces `turn_option ⊆ capability_config`; the adapter then bounds the
588
+ # result by the session binding (`⊆ session_binding`).
589
+ # Bound library ids only apply while the binding toggle is on — same
590
+ # semantics as the legacy tool (the tree's value is kept but inert
591
+ # when unbound).
592
+ bound_library_ids = (
593
+ (config.library_tag_ids or None)
594
+ if config.team_documents and config.bind_libraries
595
+ else None
596
+ )
597
+ scoped_library_tag_ids = narrow_scope_ids(
598
+ bound_library_ids, turn.library_tag_ids
599
+ )
600
+ scoped_document_uids = narrow_scope_ids(
601
+ config.document_uids or None, turn.document_uids
602
+ )
603
+ default_top_k = config.default_top_k if config.default_top_k > 0 else 8
604
+ # With the search-policy picker shown, the configured policy is only
605
+ # the picker's DEFAULT: pass None so the adapter falls back to the
606
+ # per-turn RuntimeContext value (which carries that default anyway).
607
+ # With the picker hidden, the configured policy is enforced.
608
+ search_policy = (
609
+ None if config.show_search_policy_control else config.search_policy
610
+ )
611
+ min_source_score_ratio = config.min_source_score_ratio
612
+
613
+ @tool(
614
+ "search_documents_using_vectorization",
615
+ response_format="content_and_artifact",
616
+ )
617
+ async def search_documents_using_vectorization(
618
+ question: str,
619
+ top_k: int | None = None,
620
+ ) -> tuple[str, ToolInvocationResult]:
621
+ """Search the selected document libraries using semantic similarity
622
+ (RAG) — call this BEFORE answering any factual, technical, or
623
+ domain-specific question.
624
+
625
+ The corpus may hold more specific or more recent information than
626
+ you already know. Skip this tool only for purely conversational
627
+ exchanges (greetings, thanks, clarifying what was just said).
628
+
629
+ Covers prose/text documents. If a hit describes a structured/tabular
630
+ dataset (a "dataset pointer"), do not answer from it directly — pivot
631
+ to the tabular/SQL tool it names instead.
632
+
633
+ Returns ranked hits with title and content. Only use information
634
+ actually present in the returned hits; never invent facts beyond
635
+ them.
636
+ """
637
+
638
+ port = services.document_search
639
+ if port is None:
640
+ # No platform port injected (e.g. a bare test harness). Fail
641
+ # LOUD in the tool result rather than silently returning nothing.
642
+ raise RuntimeError(
643
+ "document_access: RuntimeServices.document_search is not "
644
+ "available on this execution path."
645
+ )
646
+
647
+ effective_top_k = top_k if isinstance(top_k, int) and top_k > 0 else None
648
+ started = time.monotonic()
649
+ try:
650
+ result: DocumentSearchResult = await port.search(
651
+ question,
652
+ top_k=effective_top_k or default_top_k,
653
+ library_tag_ids=scoped_library_tag_ids,
654
+ document_uids=scoped_document_uids,
655
+ search_policy=search_policy,
656
+ include_attachments=config.attachments,
657
+ include_team_documents=config.team_documents,
658
+ )
659
+ except Exception as exc:
660
+ run_stop = unwrap_run_stop_error(exc)
661
+ if run_stop is not None:
662
+ raise run_stop from None
663
+ # Same contract as the sibling tools: a failing tool returns an
664
+ # `is_error=True` artifact rather than raising, or the default
665
+ # ToolNode handler re-raises and the whole turn dies with an
666
+ # empty error detail. Observed live on an expired-token 401
667
+ # (#2073 Item 3) — this is the most-used RAG tool, so it was
668
+ # the one that took the turn down.
669
+ return _document_tool_failure(
670
+ tool_ref=DOCUMENT_ACCESS_TOOL_REF,
671
+ action="search documents",
672
+ exc=exc,
673
+ elapsed_s=time.monotonic() - started,
674
+ )
675
+ hits = result.hits
676
+
677
+ content = {
678
+ "query": question,
679
+ "hits": [
680
+ {
681
+ k: v
682
+ for k, v in hit.model_dump(mode="json").items()
683
+ if k in _LLM_FIELDS
684
+ }
685
+ for hit in hits
686
+ ],
687
+ }
688
+ # `blocks` feed the LLM the full hit set (the model needs to see a
689
+ # dataset pointer to know to pivot to the tabular tool, and every
690
+ # hit to reason with) — `sources` (the chat Sources panel) is
691
+ # narrowed separately: never a pointer chunk (no real content to
692
+ # cite), and never a hit that's noise relative to the best match
693
+ # in this call (found live citing near-zero-relevance paragraphs
694
+ # from an unrelated document, RAG-DATASET-DISCOVERY-RFC.md §7).
695
+ artifact = ToolInvocationResult(
696
+ tool_ref=DOCUMENT_ACCESS_TOOL_REF,
697
+ blocks=(ToolContentBlock(kind=ToolContentKind.JSON, data=content),),
698
+ sources=select_citable_sources(
699
+ hits, min_score_ratio=min_source_score_ratio
700
+ ),
701
+ )
702
+ return json.dumps(content), artifact
703
+
704
+ @tool("list_document_tree", response_format="content_and_artifact")
705
+ async def list_document_tree(
706
+ working_directory: str | None = None,
707
+ max_chars: int = 6000,
708
+ ) -> tuple[str, ToolInvocationResult]:
709
+ """List the folders and documents in the user's document scope as a tree.
710
+
711
+ Call this first to orient on what's available before searching or
712
+ summarizing — it shows folder structure and, for each document, its
713
+ name, uid, and upload date, not its content.
714
+
715
+ Documents are rendered as "name [document_uid] (uploaded date)" —
716
+ use that uid as the `document_uid` argument to summarize_document.
717
+ Folder lines end with "/" and are rendered as "name [folder:id]/":
718
+ a folder id is NOT a document_uid — never pass it to
719
+ summarize_document (the call will fail). To summarize a folder's
720
+ content, summarize each document leaf under it instead. The
721
+ bracketed identifiers are internal working ids for YOUR tool
722
+ calls only: NEVER repeat them in your answer to the user — always
723
+ refer to documents by their display name.
724
+
725
+ `working_directory` narrows the listing to a specific folder (e.g.
726
+ "Sales/HR"); omit it to start from the root. The tree is rendered as
727
+ indented text, with documents appearing as leaves under every folder
728
+ they belong to (a document can be in more than one folder).
729
+
730
+ If the corpus is too large to show in full, the deepest branches are
731
+ pruned and a note tells you how many items were omitted — when that
732
+ happens, narrow `working_directory` or switch to
733
+ search_documents_using_vectorization instead of trying to browse
734
+ everything.
735
+ """
736
+
737
+ port = services.document_tree
738
+ if port is None:
739
+ raise RuntimeError(
740
+ "document_access: RuntimeServices.document_tree is not "
741
+ "available on this execution path."
742
+ )
743
+
744
+ effective_max_chars = _clamp(max_chars, _TREE_MAX_CHARS_BOUNDS)
745
+ started = time.monotonic()
746
+ try:
747
+ result: DocumentTreeResult = await port.tree(
748
+ working_directory=working_directory,
749
+ library_tag_ids=scoped_library_tag_ids,
750
+ document_uids=scoped_document_uids,
751
+ max_chars=effective_max_chars,
752
+ )
753
+ except Exception as exc:
754
+ run_stop = unwrap_run_stop_error(exc)
755
+ if run_stop is not None:
756
+ raise run_stop from None
757
+ return _document_tool_failure(
758
+ tool_ref="list_document_tree",
759
+ action="list the document tree",
760
+ exc=exc,
761
+ elapsed_s=time.monotonic() - started,
762
+ )
763
+ # `blocks` carries the same tree text as `content` (CAPAB-02): a
764
+ # Graph agent's plain-dict invocation keeps only the artifact half
765
+ # of a `content_and_artifact` return (`_adapt_capability_tool_for_graph`,
766
+ # `graph_runtime.py`) — an artifact with no payload silently loses
767
+ # the tree for a Graph node, exactly the "never silently degrade"
768
+ # failure RFC §3.9 forbids. ReAct is unaffected: `content` is
769
+ # still what the model reads.
770
+ artifact = ToolInvocationResult(
771
+ tool_ref="list_document_tree",
772
+ blocks=(ToolContentBlock(kind=ToolContentKind.TEXT, text=result.tree),),
773
+ )
774
+ return result.tree, artifact
775
+
776
+ tools: list[BaseTool] = [search_documents_using_vectorization]
777
+ # The tree lists the corpus only (no session-attachment enumeration yet,
778
+ # see module docstring), so it is dropped rather than registered-but-empty.
779
+ if config.team_documents:
780
+ tools.append(list_document_tree)
781
+ return tools
File without changes
@@ -0,0 +1,63 @@
1
+ Metadata-Version: 2.4
2
+ Name: fred-capability-document-access
3
+ Version: 4.4.3
4
+ Summary: Fred agent capability: scoped vector search and document-tree listing over the Knowledge Flow corpus (document_access).
5
+ Author-email: Thales <noreply@thalesgroup.com>
6
+ License: Apache-2.0
7
+ Project-URL: Homepage, https://site.fredlab.dev
8
+ Project-URL: Repository, https://github.com/ThalesGroup/fred
9
+ Classifier: License :: OSI Approved :: Apache Software License
10
+ Classifier: Programming Language :: Python :: 3
11
+ Classifier: Programming Language :: Python :: 3 :: Only
12
+ Classifier: Operating System :: OS Independent
13
+ Requires-Python: <3.13,>=3.12
14
+ Description-Content-Type: text/markdown
15
+ Requires-Dist: fred-core>=4.4.3
16
+ Requires-Dist: fred-sdk[agents]>=4.4.3
17
+ Requires-Dist: pydantic<3.0.0,>=2.7.0
18
+ Requires-Dist: langchain-core>=0.3.0
19
+ Provides-Extra: dev
20
+ Requires-Dist: bandit>=1.8.6; extra == "dev"
21
+ Requires-Dist: basedpyright==1.31.0; extra == "dev"
22
+ Requires-Dist: detect-secrets>=1.5.0; extra == "dev"
23
+ Requires-Dist: pytest>=8.4.2; extra == "dev"
24
+ Requires-Dist: pytest-asyncio>=1.2.0; extra == "dev"
25
+ Requires-Dist: pytest-cov>=6.2.1; extra == "dev"
26
+ Requires-Dist: pytest-socket>=0.7.0; extra == "dev"
27
+ Requires-Dist: ruff<0.16,>=0.15.22; extra == "dev"
28
+
29
+ # fred-capability-document-access
30
+
31
+ Fred agent capability giving an agent scoped access to the Knowledge Flow
32
+ document corpus: `search_documents_using_vectorization` (semantic/RAG search)
33
+ and `list_document_tree` (folder and document listing, names and uids only).
34
+
35
+ It is the reference implementation a capability author copies: real tools wired
36
+ to platform services through typed SDK ports, static config-field scoping, and
37
+ one computed chat-turn narrowing control — with no HTTP stack, no access token
38
+ and no per-turn binding anywhere in the capability.
39
+
40
+ ## Scoping
41
+
42
+ Three levels narrow each other, never widen:
43
+
44
+ turn_option ⊆ capability_config ⊆ session_binding
45
+
46
+ The capability enforces the first (`narrow_scope_ids`, in `capability.py`); the
47
+ runtime's `DocumentSearchAdapter` enforces the second. The tool signature the
48
+ model sees carries only `question` / `top_k` — scope and identity travel in the
49
+ middleware closure and can never be widened by the model.
50
+
51
+ ## Chat controls
52
+
53
+ `chat_controls(config)` emits up to four stock composer widgets behind their
54
+ config toggles: `attach_files`, `document_scope`, `search_policy`, `rag_scope`.
55
+ Their params are SDK models (`fred_sdk.contracts.models`) and carry the widget's
56
+ *default* only — the value the user picks travels on `RuntimeContext`.
57
+
58
+ ## Tests
59
+
60
+ `make test` runs offline: the ports are faked, no service is contacted. The dev
61
+ group depends on `fred-runtime` for `CapabilityRegistry` (entry-point discovery)
62
+ and `build_capability_context`; the capability itself depends only on
63
+ `fred-core` and `fred-sdk`.
@@ -0,0 +1,8 @@
1
+ fred_capability_document_access/__init__.py,sha256=X6RAL-lmFQV3uiUR3DuvRzTUsr7HnKWMUXLHtQO7DB8,1188
2
+ fred_capability_document_access/capability.py,sha256=95-V-MD5zX5HZhs4MLz3-HsRpa8jQeGiluJLnY18HyE,36092
3
+ fred_capability_document_access/py.typed,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
4
+ fred_capability_document_access-4.4.3.dist-info/METADATA,sha256=4Q_ITVcDyrRFHUvP_A9lMFvLIj32wiXj78Xus8-loBU,2808
5
+ fred_capability_document_access-4.4.3.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
6
+ fred_capability_document_access-4.4.3.dist-info/entry_points.txt,sha256=DWjm5ABec9HPVlM2GQWFgXoM1_I_eOJWvfZwvsTdOsE,106
7
+ fred_capability_document_access-4.4.3.dist-info/top_level.txt,sha256=t-N9Y97OGCjsBMG6mGSsvw6bqC9VvgvnIlA_lZPsLz0,32
8
+ fred_capability_document_access-4.4.3.dist-info/RECORD,,
@@ -0,0 +1,5 @@
1
+ Wheel-Version: 1.0
2
+ Generator: setuptools (84.0.0)
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
5
+
@@ -0,0 +1,2 @@
1
+ [fred.capabilities]
2
+ document_access = fred_capability_document_access.capability:DocumentAccessCapability
@@ -0,0 +1 @@
1
+ fred_capability_document_access