interloper-toolkit 0.87.0__tar.gz → 0.89.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (22) hide show
  1. {interloper_toolkit-0.87.0 → interloper_toolkit-0.89.0}/PKG-INFO +1 -1
  2. {interloper_toolkit-0.87.0 → interloper_toolkit-0.89.0}/pyproject.toml +2 -2
  3. {interloper_toolkit-0.87.0 → interloper_toolkit-0.89.0}/pyproject.toml.orig +2 -2
  4. {interloper_toolkit-0.87.0 → interloper_toolkit-0.89.0}/src/interloper_toolkit/__init__.py +7 -8
  5. interloper_toolkit-0.89.0/src/interloper_toolkit/analytics.py +360 -0
  6. interloper_toolkit-0.89.0/src/interloper_toolkit/authz.py +73 -0
  7. {interloper_toolkit-0.87.0 → interloper_toolkit-0.89.0}/src/interloper_toolkit/catalog.py +8 -4
  8. interloper_toolkit-0.89.0/src/interloper_toolkit/collection.py +477 -0
  9. {interloper_toolkit-0.87.0 → interloper_toolkit-0.89.0}/src/interloper_toolkit/context.py +4 -0
  10. interloper_toolkit-0.89.0/src/interloper_toolkit/errors.py +110 -0
  11. interloper_toolkit-0.89.0/src/interloper_toolkit/jobs.py +69 -0
  12. {interloper_toolkit-0.87.0 → interloper_toolkit-0.89.0}/src/interloper_toolkit/lineage.py +4 -4
  13. interloper_toolkit-0.89.0/src/interloper_toolkit/models.py +810 -0
  14. interloper_toolkit-0.89.0/src/interloper_toolkit/scheduling.py +622 -0
  15. interloper_toolkit-0.89.0/src/interloper_toolkit/sources.py +393 -0
  16. interloper_toolkit-0.89.0/src/interloper_toolkit/stats.py +80 -0
  17. interloper_toolkit-0.89.0/src/interloper_toolkit/utils.py +22 -0
  18. interloper_toolkit-0.87.0/src/interloper_toolkit/analytics.py +0 -164
  19. interloper_toolkit-0.87.0/src/interloper_toolkit/collection.py +0 -120
  20. interloper_toolkit-0.87.0/src/interloper_toolkit/models.py +0 -405
  21. interloper_toolkit-0.87.0/src/interloper_toolkit/scheduling.py +0 -172
  22. {interloper_toolkit-0.87.0 → interloper_toolkit-0.89.0}/README.md +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.3
2
2
  Name: interloper-toolkit
3
- Version: 0.87.0
3
+ Version: 0.89.0
4
4
  Summary: Interloper shared read-only tool functions for AI surfaces (agent, MCP)
5
5
  Author: Guillaume Onfroy
6
6
  Author-email: Guillaume Onfroy <guillaume@digitlcloud.com>
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "interloper-toolkit"
3
- version = "0.87.0"
3
+ version = "0.89.0"
4
4
  description = "Interloper shared read-only tool functions for AI surfaces (agent, MCP)"
5
5
  readme = "README.md"
6
6
  requires-python = ">=3.10"
@@ -57,7 +57,7 @@ convention = "google"
57
57
  "RUF069",
58
58
  "PLW0108",
59
59
  ]
60
- "src/interloper_toolkit/{analytics,catalog,collection,lineage,scheduling}.py" = [
60
+ "src/interloper_toolkit/{analytics,catalog,collection,jobs,lineage,scheduling,sources}.py" = [
61
61
  "D417",
62
62
  "DOC201",
63
63
  "BLE001",
@@ -3,7 +3,7 @@
3
3
  # ###############
4
4
  [project]
5
5
  name = "interloper-toolkit"
6
- version = "0.87.0"
6
+ version = "0.89.0"
7
7
  description = "Interloper shared read-only tool functions for AI surfaces (agent, MCP)"
8
8
  readme = "README.md"
9
9
  authors = [{ name = "Guillaume Onfroy", email = "guillaume@digitlcloud.com" }]
@@ -37,4 +37,4 @@ convention = "google"
37
37
  [tool.ruff.lint.per-file-ignores]
38
38
  "__init__.py" = ["F401", "F403"]
39
39
  "tests/**" = ["ANN", "F811", "D101", "D102", "D103", "D104", "RUF069", "PLW0108"]
40
- "src/interloper_toolkit/{analytics,catalog,collection,lineage,scheduling}.py" = ["D417", "DOC201", "BLE001"]
40
+ "src/interloper_toolkit/{analytics,catalog,collection,jobs,lineage,scheduling,sources}.py" = ["D417", "DOC201", "BLE001"]
@@ -7,17 +7,16 @@ the literal ``status`` field, never raising. The docstrings are LLM-facing:
7
7
  both the ADK agent and the MCP server surface them verbatim as tool
8
8
  descriptions.
9
9
 
10
- Almost every function here is read-only; the sole exceptions are
11
- :func:`interloper_toolkit.collection.bind_relation` and
12
- :func:`interloper_toolkit.collection.unbind_relation`, which write and are
13
- re-exported here as this package's whole write surface. A surface that must
14
- stay read-only (the MCP server's own registration is one) never registers
15
- those two; the ADK agent, whose own write tools already live beside them,
16
- does.
10
+ Reads take no role; writes declare the role they need with
11
+ :func:`~interloper_toolkit.authz.requires_role` and refuse below it. Which
12
+ functions a surface exposes is that surface's registration list (the MCP
13
+ server, for one, never registers ``create_connections``, whose arguments
14
+ would carry credentials through the client).
17
15
  """
18
16
 
17
+ from interloper_toolkit.authz import requires_role
19
18
  from interloper_toolkit.collection import bind_relation, unbind_relation
20
19
  from interloper_toolkit.context import ToolkitContext, serialize
21
20
  from interloper_toolkit.models import ToolError
22
21
 
23
- __all__ = ["ToolError", "ToolkitContext", "bind_relation", "serialize", "unbind_relation"]
22
+ __all__ = ["ToolError", "ToolkitContext", "bind_relation", "requires_role", "serialize", "unbind_relation"]
@@ -0,0 +1,360 @@
1
+ """Analytics tools — run statistics, partition coverage, and data freshness."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import datetime
6
+ from typing import Any
7
+ from uuid import UUID
8
+
9
+ from interloper.partitioning.time import TimePartition
10
+
11
+ from interloper_toolkit.context import ToolkitContext
12
+ from interloper_toolkit.models import (
13
+ AssetCoverage,
14
+ AssetCoverageRow,
15
+ FreshnessReport,
16
+ JobFreshness,
17
+ JobStats,
18
+ PartitionCoverage,
19
+ PartitionRange,
20
+ RunHistorySummary,
21
+ RunStats,
22
+ ToolError,
23
+ )
24
+ from interloper_toolkit.stats import percentile, window
25
+
26
+ _MISSING_RANGES_LIMIT = 20
27
+
28
+
29
+ def run_history_summary(
30
+ ctx: ToolkitContext,
31
+ component_id: str | None = None,
32
+ days: int = 7,
33
+ ) -> RunHistorySummary | ToolError:
34
+ """Summarize run statistics over a period.
35
+
36
+ Args:
37
+ component_id: Filter to a specific job UUID (optional, all jobs if omitted).
38
+ days: Number of days to look back (default 7).
39
+
40
+ Returns aggregate counts (total, success, failed, canceled), success
41
+ rate, and average duration over the runs that executed within the
42
+ period, each run stack counted by its latest attempt.
43
+ """
44
+ try:
45
+ jid = UUID(component_id) if component_id else None
46
+ cutoff = datetime.datetime.now(tz=datetime.timezone.utc) - datetime.timedelta(days=days)
47
+ total = ctx.store.runs.count(ctx.org_id, component_id=jid, after=cutoff)
48
+ runs = ctx.store.runs.list_all(ctx.org_id, component_id=jid, after=cutoff, limit=total)
49
+
50
+ by_status: dict[str, int] = {}
51
+ durations: list[float] = []
52
+ for r in runs:
53
+ by_status[r.status] = by_status.get(r.status, 0) + 1
54
+ if r.started_at and r.completed_at:
55
+ durations.append((r.completed_at - r.started_at).total_seconds())
56
+
57
+ success = by_status.get("success", 0)
58
+ return RunHistorySummary(
59
+ period_days=days,
60
+ component_id=component_id,
61
+ total_runs=total,
62
+ by_status=by_status,
63
+ success_rate=round(success / total, 2) if total > 0 else None,
64
+ avg_duration_seconds=round(sum(durations) / len(durations), 1) if durations else None,
65
+ )
66
+ except Exception as e:
67
+ return ToolError(error=str(e))
68
+
69
+
70
+ def partition_coverage(
71
+ ctx: ToolkitContext,
72
+ component_id: str,
73
+ start_date: str,
74
+ end_date: str,
75
+ ) -> PartitionCoverage | ToolError:
76
+ """Check partition coverage for a job over a date range.
77
+
78
+ Args:
79
+ component_id: UUID of the job.
80
+ start_date: Start date in ISO format (YYYY-MM-DD).
81
+ end_date: End date in ISO format (YYYY-MM-DD), inclusive.
82
+
83
+ Returns which dates have successful runs and which are missing.
84
+ """
85
+ try:
86
+ job = ctx.store.components.get(UUID(component_id), kind="job", org_id=ctx.org_id)
87
+ total = ctx.store.runs.count(ctx.org_id, component_id=job.id, status="success")
88
+ runs = ctx.store.runs.list_all(ctx.org_id, component_id=job.id, status="success", limit=total)
89
+
90
+ start = datetime.date.fromisoformat(start_date)
91
+ end = datetime.date.fromisoformat(end_date)
92
+
93
+ # Coverage is a daily question, so a run covers every day inside its partition.
94
+ covered: set[datetime.date] = set()
95
+ for r in runs:
96
+ if not r.partition_key:
97
+ continue
98
+ p_start, p_end = TimePartition.from_key(r.partition_key).bounds
99
+ if isinstance(p_start, datetime.datetime):
100
+ days = [p_start.date()]
101
+ else:
102
+ days = []
103
+ current = p_start
104
+ while current < p_end:
105
+ days.append(current)
106
+ current += datetime.timedelta(days=1)
107
+ covered.update(day for day in days if start <= day <= end)
108
+
109
+ # Build expected date range
110
+ expected: list[datetime.date] = []
111
+ current = start
112
+ while current <= end:
113
+ expected.append(current)
114
+ current += datetime.timedelta(days=1)
115
+
116
+ missing = sorted(set(expected) - covered)
117
+ coverage_pct = round(len(covered) / len(expected) * 100, 1) if expected else 100.0
118
+
119
+ return PartitionCoverage(
120
+ component_id=component_id,
121
+ start_date=start_date,
122
+ end_date=end_date,
123
+ total_days=len(expected),
124
+ covered_days=len(covered),
125
+ missing_days=len(missing),
126
+ coverage_percent=coverage_pct,
127
+ missing_dates=[d.isoformat() for d in missing],
128
+ )
129
+ except Exception as e:
130
+ return ToolError(error=str(e))
131
+
132
+
133
+ def freshness_check(ctx: ToolkitContext) -> FreshnessReport | ToolError:
134
+ """Check data freshness for all jobs.
135
+
136
+ Returns the last successful run timestamp for each job and flags
137
+ any that haven't succeeded in over 24 hours.
138
+ """
139
+ try:
140
+ jobs = ctx.store.components.list_all(ctx.org_id, kinds=["job"])
141
+ now = datetime.datetime.now(tz=datetime.timezone.utc)
142
+
143
+ results = []
144
+ for job in jobs:
145
+ if not (job.config or {}).get("enabled", True):
146
+ continue
147
+ component_id = job.id
148
+ runs = ctx.store.runs.list_all(ctx.org_id, component_id=component_id, status="success", limit=1)
149
+ last_success = runs[0] if runs else None
150
+
151
+ hours_since = None
152
+ if last_success and last_success.completed_at:
153
+ delta = now - last_success.completed_at
154
+ hours_since = round(delta.total_seconds() / 3600, 1)
155
+
156
+ results.append(JobFreshness(
157
+ job=job,
158
+ last_success_at=last_success.completed_at if last_success else None,
159
+ hours_since_success=hours_since,
160
+ stale=hours_since is None or hours_since > 24,
161
+ ))
162
+
163
+ stale_count = sum(1 for r in results if r.stale)
164
+ return FreshnessReport(
165
+ total_jobs=len(results),
166
+ stale_count=stale_count,
167
+ jobs=results,
168
+ )
169
+ except Exception as e:
170
+ return ToolError(error=str(e))
171
+
172
+
173
+ def run_stats(
174
+ ctx: ToolkitContext,
175
+ since: str | None = None,
176
+ until: str | None = None,
177
+ component_id: str | None = None,
178
+ limit: int = 25,
179
+ offset: int = 0,
180
+ ) -> RunStats | ToolError:
181
+ """Per-job run statistics over a window: verdicts, durations and retries.
182
+
183
+ Args:
184
+ since: ISO date or datetime the window opens at (default: 7 days ago).
185
+ until: ISO date or datetime the window closes before (default: open).
186
+ component_id: Restrict to this job UUID.
187
+ limit: Maximum number of jobs to return (default 25).
188
+ offset: Number of jobs to skip, for paging past the first page.
189
+
190
+ Returns one row per job that ran in the window, most failures first: run
191
+ stacks by final status, attempts, duration p50/p90/max in seconds, and
192
+ how many retried stacks healed or are still failing.
193
+ """
194
+ try:
195
+ start, end = window(since, until, default_days=7)
196
+ filters: dict[str, Any] = {
197
+ "component_id": UUID(component_id) if component_id else None,
198
+ "after": start,
199
+ "before": end,
200
+ "all_attempts": True,
201
+ }
202
+ total = ctx.store.runs.count(ctx.org_id, **filters)
203
+ runs = ctx.store.runs.list_all(ctx.org_id, **filters, limit=total)
204
+
205
+ stacks: dict[UUID | None, dict[UUID, list[Any]]] = {}
206
+ names: dict[UUID | None, str | None] = {}
207
+ for r in runs:
208
+ stacks.setdefault(r.component_id, {}).setdefault(r.root_run_id, []).append(r)
209
+ names.setdefault(r.component_id, r.target.name if r.target else None)
210
+
211
+ rows = []
212
+ for job_id, by_root in stacks.items():
213
+ by_status: dict[str, int] = {}
214
+ durations: list[float] = []
215
+ attempts = retried = healed = still_failing = 0
216
+ for chain in by_root.values():
217
+ latest = max(chain, key=lambda r: r.attempt)
218
+ by_status[latest.status] = by_status.get(latest.status, 0) + 1
219
+ attempts += len(chain)
220
+ durations += [
221
+ (r.completed_at - r.started_at).total_seconds() for r in chain if r.started_at and r.completed_at
222
+ ]
223
+ if len(chain) > 1:
224
+ retried += 1
225
+ healed += latest.status == "success"
226
+ still_failing += latest.status == "failed"
227
+ rows.append(
228
+ JobStats(
229
+ job_id=job_id,
230
+ job_name=names[job_id],
231
+ stacks=by_status,
232
+ attempts=attempts,
233
+ duration_p50_s=_rounded(percentile(durations, 50)),
234
+ duration_p90_s=_rounded(percentile(durations, 90)),
235
+ duration_max_s=_rounded(max(durations, default=None)),
236
+ stacks_retried=retried,
237
+ healed=healed,
238
+ still_failing=still_failing,
239
+ )
240
+ )
241
+ rows.sort(key=lambda j: (-j.stacks.get("failed", 0), j.job_name or ""))
242
+ page = rows[offset : offset + limit]
243
+ return RunStats(since=start, until=end, count=len(page), total=len(rows), jobs=page)
244
+ except Exception as e:
245
+ return ToolError(error=str(e))
246
+
247
+
248
+ def asset_coverage(
249
+ ctx: ToolkitContext,
250
+ component_id: str,
251
+ start_key: str,
252
+ end_key: str,
253
+ limit: int = 50,
254
+ offset: int = 0,
255
+ ) -> AssetCoverage | ToolError:
256
+ """Per-asset partition coverage of a job over a range of partition keys.
257
+
258
+ Unlike partition_coverage, which needs a whole run to have succeeded, an
259
+ asset counts as covered for a partition once any run's execution of it
260
+ succeeded, so a run where most assets succeeded shows what is actually
261
+ missing. Only assets that executed at least once in the range appear.
262
+
263
+ Args:
264
+ component_id: UUID of the job.
265
+ start_key: First partition key, in the job's granularity (2026-07-01,
266
+ 2026-07, 2026, or 2026-07-01T13).
267
+ end_key: Last partition key, inclusive; must share the start key's
268
+ granularity.
269
+ limit: Maximum number of assets to return (default 50).
270
+ offset: Number of assets to skip, for paging past the first page.
271
+
272
+ Returns the page of assets, least covered first, each with its covered,
273
+ failed and never-run partition counts and its missing partitions as
274
+ ranges ready for a backfill, plus a rollup of the range's partitions by
275
+ whether all, some or none of the assets are covered.
276
+ """
277
+ try:
278
+ job = ctx.store.components.get(UUID(component_id), kind="job", org_id=ctx.org_id)
279
+ first, last = TimePartition.from_key(start_key), TimePartition.from_key(end_key)
280
+ if first.granularity is not last.granularity:
281
+ return ToolError(error=f"{start_key!r} and {end_key!r} are keys of different granularities")
282
+ granularity = first.granularity
283
+ expected = [granularity.format(value) for value in granularity.period_range(first.value, last.value)]
284
+
285
+ rows = ctx.store.events.partition_coverage(ctx.org_id, job.id, start_key, end_key)
286
+ keys: dict[UUID, str | None] = {}
287
+ attempted: dict[UUID, set[str]] = {}
288
+ covered: dict[UUID, set[str]] = {}
289
+ for row in rows:
290
+ keys.setdefault(row.component_id, row.component_key)
291
+ attempted.setdefault(row.component_id, set()).add(row.partition_key)
292
+ if row.succeeded:
293
+ covered.setdefault(row.component_id, set()).add(row.partition_key)
294
+
295
+ assets = []
296
+ for asset_id, asset_key in keys.items():
297
+ done = covered.get(asset_id, set())
298
+ missing = [key for key in expected if key not in done]
299
+ ranges = _ranges(missing, expected)
300
+ assets.append(
301
+ AssetCoverageRow(
302
+ asset_id=asset_id,
303
+ asset_key=asset_key,
304
+ covered=len(done),
305
+ failed=len(attempted[asset_id] - done),
306
+ never_run=len(missing) - len(attempted[asset_id] - done),
307
+ missing=ranges[:_MISSING_RANGES_LIMIT],
308
+ missing_ranges_total=len(ranges),
309
+ )
310
+ )
311
+ assets.sort(key=lambda a: (a.covered, a.asset_key or ""))
312
+
313
+ per_partition = [sum(key in covered.get(asset_id, ()) for asset_id in keys) for key in expected]
314
+ page = assets[offset : offset + limit]
315
+ return AssetCoverage(
316
+ component_id=job.id,
317
+ start_key=start_key,
318
+ end_key=end_key,
319
+ partitions=len(expected),
320
+ all_covered=sum(n == len(keys) for n in per_partition) if keys else 0,
321
+ partly_covered=sum(0 < n < len(keys) for n in per_partition),
322
+ none_covered=sum(n == 0 for n in per_partition) if keys else len(expected),
323
+ count=len(page),
324
+ total=len(assets),
325
+ assets=page,
326
+ )
327
+ except Exception as e:
328
+ return ToolError(error=str(e))
329
+
330
+
331
+ def _ranges(missing: list[str], expected: list[str]) -> list[PartitionRange]:
332
+ """Collapse the missing keys into runs of consecutive partitions.
333
+
334
+ Args:
335
+ missing: The uncovered keys, in *expected*'s order.
336
+ expected: Every key of the range, in order.
337
+
338
+ Returns:
339
+ One inclusive range per run of consecutive missing keys.
340
+ """
341
+ position = {key: index for index, key in enumerate(expected)}
342
+ ranges: list[PartitionRange] = []
343
+ for key in missing:
344
+ if ranges and position[key] == position[ranges[-1].end_key] + 1:
345
+ ranges[-1].end_key = key
346
+ else:
347
+ ranges.append(PartitionRange(start_key=key, end_key=key))
348
+ return ranges
349
+
350
+
351
+ def _rounded(value: float | None) -> float | None:
352
+ """Round a duration to a tenth of a second, passing ``None`` through.
353
+
354
+ Args:
355
+ value: Seconds, or ``None`` when there was nothing to measure.
356
+
357
+ Returns:
358
+ The rounded value, or ``None``.
359
+ """
360
+ return round(value, 1) if value is not None else None
@@ -0,0 +1,73 @@
1
+ """Role gating for the toolkit's write tools.
2
+
3
+ The context carries the caller's role in the organisation; a write declares
4
+ the role it needs and refuses, as a structured :class:`ToolError`, before its
5
+ body runs. The ranks and the refusal wording mirror the API's route gates, so
6
+ a caller can do through a tool exactly what it could do through the app.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import functools
12
+ import inspect
13
+ from collections.abc import Callable
14
+ from typing import Any, TypeVar, cast
15
+
16
+ from interloper_toolkit.context import ToolkitContext
17
+ from interloper_toolkit.models import ToolError
18
+
19
+ _ROLE_RANK = {"viewer": 0, "editor": 1, "admin": 2}
20
+
21
+ F = TypeVar("F", bound=Callable[..., Any])
22
+
23
+
24
+ def requires_role(minimum: str) -> Callable[[F], F]:
25
+ """Refuse a tool call whose context holds a role below *minimum*.
26
+
27
+ Works on sync and async tool functions alike; the context is the first
28
+ positional argument, as it is for every toolkit function. The wrapped
29
+ function keeps its name and docstring, which the AI surfaces adopt.
30
+
31
+ Args:
32
+ minimum: The lowest role allowed: ``viewer``, ``editor`` or ``admin``.
33
+
34
+ Returns:
35
+ The decorator to apply to a tool function.
36
+
37
+ Raises:
38
+ ValueError: If *minimum* is not a known role.
39
+ """
40
+ if minimum not in _ROLE_RANK:
41
+ raise ValueError(f"Unknown role {minimum!r}; expected one of {sorted(_ROLE_RANK)}")
42
+
43
+ def decorate(func: F) -> F:
44
+ if inspect.iscoroutinefunction(func):
45
+
46
+ @functools.wraps(func)
47
+ async def async_gate(ctx: ToolkitContext, *args: Any, **kwargs: Any) -> Any:
48
+ return denied(ctx.role, minimum) or await func(ctx, *args, **kwargs)
49
+
50
+ return cast(F, async_gate)
51
+
52
+ @functools.wraps(func)
53
+ def gate(ctx: ToolkitContext, *args: Any, **kwargs: Any) -> Any:
54
+ return denied(ctx.role, minimum) or func(ctx, *args, **kwargs)
55
+
56
+ return cast(F, gate)
57
+
58
+ return decorate
59
+
60
+
61
+ def denied(role: str, minimum: str) -> ToolError | None:
62
+ """The refusal for a role below *minimum*, or ``None`` when the role suffices.
63
+
64
+ Args:
65
+ role: The caller's role; an unknown one ranks below every known role.
66
+ minimum: The lowest role allowed.
67
+
68
+ Returns:
69
+ The structured error, or ``None`` when the call may proceed.
70
+ """
71
+ if _ROLE_RANK.get(role, -1) < _ROLE_RANK[minimum]:
72
+ return ToolError(error=f"Requires {minimum} role or higher")
73
+ return None
@@ -81,7 +81,7 @@ def list_definitions(
81
81
  if kind not in counts:
82
82
  return ToolError(
83
83
  error=f"No '{kind}' definitions in the catalog",
84
- valid_kinds=sorted(counts),
84
+ valid_values=sorted(counts),
85
85
  )
86
86
 
87
87
  collection_counts: dict[str, int] = {}
@@ -179,7 +179,7 @@ def get_asset_schema(ctx: ToolkitContext, source_key: str, asset_key: str) -> As
179
179
  return ToolError(error=str(e))
180
180
 
181
181
 
182
- def search_fields(ctx: ToolkitContext, query: str) -> FieldSearchResult | ToolError:
182
+ def search_fields(ctx: ToolkitContext, query: str, limit: int = 50, offset: int = 0) -> FieldSearchResult | ToolError:
183
183
  """Search for fields across all asset schemas in the catalog matching a query string.
184
184
 
185
185
  Searches every source definition's schemas, whether or not the source is
@@ -187,8 +187,11 @@ def search_fields(ctx: ToolkitContext, query: str) -> FieldSearchResult | ToolEr
187
187
 
188
188
  Args:
189
189
  query: Substring to search for in field names and descriptions (case-insensitive).
190
+ limit: Maximum number of matches to return (default 50).
191
+ offset: Number of matches to skip, for paging past the first page.
190
192
 
191
- Returns matching fields grouped by source and asset, with field type and description.
193
+ Returns the page of matching fields grouped by source and asset, with
194
+ field type and description, plus the total number of matches.
192
195
  """
193
196
  try:
194
197
  query_lower = query.lower()
@@ -215,7 +218,8 @@ def search_fields(ctx: ToolkitContext, query: str) -> FieldSearchResult | ToolEr
215
218
  description=field_desc,
216
219
  ))
217
220
 
218
- return FieldSearchResult(query=query, match_count=len(matches), matches=matches)
221
+ page = matches[offset : offset + limit]
222
+ return FieldSearchResult(query=query, match_count=len(page), total=len(matches), matches=page)
219
223
  except Exception as e:
220
224
  return ToolError(error=str(e))
221
225