interloper-toolkit 0.87.0__tar.gz → 0.89.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {interloper_toolkit-0.87.0 → interloper_toolkit-0.89.0}/PKG-INFO +1 -1
- {interloper_toolkit-0.87.0 → interloper_toolkit-0.89.0}/pyproject.toml +2 -2
- {interloper_toolkit-0.87.0 → interloper_toolkit-0.89.0}/pyproject.toml.orig +2 -2
- {interloper_toolkit-0.87.0 → interloper_toolkit-0.89.0}/src/interloper_toolkit/__init__.py +7 -8
- interloper_toolkit-0.89.0/src/interloper_toolkit/analytics.py +360 -0
- interloper_toolkit-0.89.0/src/interloper_toolkit/authz.py +73 -0
- {interloper_toolkit-0.87.0 → interloper_toolkit-0.89.0}/src/interloper_toolkit/catalog.py +8 -4
- interloper_toolkit-0.89.0/src/interloper_toolkit/collection.py +477 -0
- {interloper_toolkit-0.87.0 → interloper_toolkit-0.89.0}/src/interloper_toolkit/context.py +4 -0
- interloper_toolkit-0.89.0/src/interloper_toolkit/errors.py +110 -0
- interloper_toolkit-0.89.0/src/interloper_toolkit/jobs.py +69 -0
- {interloper_toolkit-0.87.0 → interloper_toolkit-0.89.0}/src/interloper_toolkit/lineage.py +4 -4
- interloper_toolkit-0.89.0/src/interloper_toolkit/models.py +810 -0
- interloper_toolkit-0.89.0/src/interloper_toolkit/scheduling.py +622 -0
- interloper_toolkit-0.89.0/src/interloper_toolkit/sources.py +393 -0
- interloper_toolkit-0.89.0/src/interloper_toolkit/stats.py +80 -0
- interloper_toolkit-0.89.0/src/interloper_toolkit/utils.py +22 -0
- interloper_toolkit-0.87.0/src/interloper_toolkit/analytics.py +0 -164
- interloper_toolkit-0.87.0/src/interloper_toolkit/collection.py +0 -120
- interloper_toolkit-0.87.0/src/interloper_toolkit/models.py +0 -405
- interloper_toolkit-0.87.0/src/interloper_toolkit/scheduling.py +0 -172
- {interloper_toolkit-0.87.0 → interloper_toolkit-0.89.0}/README.md +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "interloper-toolkit"
|
|
3
|
-
version = "0.
|
|
3
|
+
version = "0.89.0"
|
|
4
4
|
description = "Interloper shared read-only tool functions for AI surfaces (agent, MCP)"
|
|
5
5
|
readme = "README.md"
|
|
6
6
|
requires-python = ">=3.10"
|
|
@@ -57,7 +57,7 @@ convention = "google"
|
|
|
57
57
|
"RUF069",
|
|
58
58
|
"PLW0108",
|
|
59
59
|
]
|
|
60
|
-
"src/interloper_toolkit/{analytics,catalog,collection,lineage,scheduling}.py" = [
|
|
60
|
+
"src/interloper_toolkit/{analytics,catalog,collection,jobs,lineage,scheduling,sources}.py" = [
|
|
61
61
|
"D417",
|
|
62
62
|
"DOC201",
|
|
63
63
|
"BLE001",
|
|
@@ -3,7 +3,7 @@
|
|
|
3
3
|
# ###############
|
|
4
4
|
[project]
|
|
5
5
|
name = "interloper-toolkit"
|
|
6
|
-
version = "0.
|
|
6
|
+
version = "0.89.0"
|
|
7
7
|
description = "Interloper shared read-only tool functions for AI surfaces (agent, MCP)"
|
|
8
8
|
readme = "README.md"
|
|
9
9
|
authors = [{ name = "Guillaume Onfroy", email = "guillaume@digitlcloud.com" }]
|
|
@@ -37,4 +37,4 @@ convention = "google"
|
|
|
37
37
|
[tool.ruff.lint.per-file-ignores]
|
|
38
38
|
"__init__.py" = ["F401", "F403"]
|
|
39
39
|
"tests/**" = ["ANN", "F811", "D101", "D102", "D103", "D104", "RUF069", "PLW0108"]
|
|
40
|
-
"src/interloper_toolkit/{analytics,catalog,collection,lineage,scheduling}.py" = ["D417", "DOC201", "BLE001"]
|
|
40
|
+
"src/interloper_toolkit/{analytics,catalog,collection,jobs,lineage,scheduling,sources}.py" = ["D417", "DOC201", "BLE001"]
|
|
@@ -7,17 +7,16 @@ the literal ``status`` field, never raising. The docstrings are LLM-facing:
|
|
|
7
7
|
both the ADK agent and the MCP server surface them verbatim as tool
|
|
8
8
|
descriptions.
|
|
9
9
|
|
|
10
|
-
|
|
11
|
-
:func
|
|
12
|
-
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
those two; the ADK agent, whose own write tools already live beside them,
|
|
16
|
-
does.
|
|
10
|
+
Reads take no role; writes declare the role they need with
|
|
11
|
+
:func:`~interloper_toolkit.authz.requires_role` and refuse below it. Which
|
|
12
|
+
functions a surface exposes is that surface's registration list (the MCP
|
|
13
|
+
server, for one, never registers ``create_connections``, whose arguments
|
|
14
|
+
would carry credentials through the client).
|
|
17
15
|
"""
|
|
18
16
|
|
|
17
|
+
from interloper_toolkit.authz import requires_role
|
|
19
18
|
from interloper_toolkit.collection import bind_relation, unbind_relation
|
|
20
19
|
from interloper_toolkit.context import ToolkitContext, serialize
|
|
21
20
|
from interloper_toolkit.models import ToolError
|
|
22
21
|
|
|
23
|
-
__all__ = ["ToolError", "ToolkitContext", "bind_relation", "serialize", "unbind_relation"]
|
|
22
|
+
__all__ = ["ToolError", "ToolkitContext", "bind_relation", "requires_role", "serialize", "unbind_relation"]
|
|
@@ -0,0 +1,360 @@
|
|
|
1
|
+
"""Analytics tools — run statistics, partition coverage, and data freshness."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import datetime
|
|
6
|
+
from typing import Any
|
|
7
|
+
from uuid import UUID
|
|
8
|
+
|
|
9
|
+
from interloper.partitioning.time import TimePartition
|
|
10
|
+
|
|
11
|
+
from interloper_toolkit.context import ToolkitContext
|
|
12
|
+
from interloper_toolkit.models import (
|
|
13
|
+
AssetCoverage,
|
|
14
|
+
AssetCoverageRow,
|
|
15
|
+
FreshnessReport,
|
|
16
|
+
JobFreshness,
|
|
17
|
+
JobStats,
|
|
18
|
+
PartitionCoverage,
|
|
19
|
+
PartitionRange,
|
|
20
|
+
RunHistorySummary,
|
|
21
|
+
RunStats,
|
|
22
|
+
ToolError,
|
|
23
|
+
)
|
|
24
|
+
from interloper_toolkit.stats import percentile, window
|
|
25
|
+
|
|
26
|
+
_MISSING_RANGES_LIMIT = 20
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def run_history_summary(
|
|
30
|
+
ctx: ToolkitContext,
|
|
31
|
+
component_id: str | None = None,
|
|
32
|
+
days: int = 7,
|
|
33
|
+
) -> RunHistorySummary | ToolError:
|
|
34
|
+
"""Summarize run statistics over a period.
|
|
35
|
+
|
|
36
|
+
Args:
|
|
37
|
+
component_id: Filter to a specific job UUID (optional, all jobs if omitted).
|
|
38
|
+
days: Number of days to look back (default 7).
|
|
39
|
+
|
|
40
|
+
Returns aggregate counts (total, success, failed, canceled), success
|
|
41
|
+
rate, and average duration over the runs that executed within the
|
|
42
|
+
period, each run stack counted by its latest attempt.
|
|
43
|
+
"""
|
|
44
|
+
try:
|
|
45
|
+
jid = UUID(component_id) if component_id else None
|
|
46
|
+
cutoff = datetime.datetime.now(tz=datetime.timezone.utc) - datetime.timedelta(days=days)
|
|
47
|
+
total = ctx.store.runs.count(ctx.org_id, component_id=jid, after=cutoff)
|
|
48
|
+
runs = ctx.store.runs.list_all(ctx.org_id, component_id=jid, after=cutoff, limit=total)
|
|
49
|
+
|
|
50
|
+
by_status: dict[str, int] = {}
|
|
51
|
+
durations: list[float] = []
|
|
52
|
+
for r in runs:
|
|
53
|
+
by_status[r.status] = by_status.get(r.status, 0) + 1
|
|
54
|
+
if r.started_at and r.completed_at:
|
|
55
|
+
durations.append((r.completed_at - r.started_at).total_seconds())
|
|
56
|
+
|
|
57
|
+
success = by_status.get("success", 0)
|
|
58
|
+
return RunHistorySummary(
|
|
59
|
+
period_days=days,
|
|
60
|
+
component_id=component_id,
|
|
61
|
+
total_runs=total,
|
|
62
|
+
by_status=by_status,
|
|
63
|
+
success_rate=round(success / total, 2) if total > 0 else None,
|
|
64
|
+
avg_duration_seconds=round(sum(durations) / len(durations), 1) if durations else None,
|
|
65
|
+
)
|
|
66
|
+
except Exception as e:
|
|
67
|
+
return ToolError(error=str(e))
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def partition_coverage(
|
|
71
|
+
ctx: ToolkitContext,
|
|
72
|
+
component_id: str,
|
|
73
|
+
start_date: str,
|
|
74
|
+
end_date: str,
|
|
75
|
+
) -> PartitionCoverage | ToolError:
|
|
76
|
+
"""Check partition coverage for a job over a date range.
|
|
77
|
+
|
|
78
|
+
Args:
|
|
79
|
+
component_id: UUID of the job.
|
|
80
|
+
start_date: Start date in ISO format (YYYY-MM-DD).
|
|
81
|
+
end_date: End date in ISO format (YYYY-MM-DD), inclusive.
|
|
82
|
+
|
|
83
|
+
Returns which dates have successful runs and which are missing.
|
|
84
|
+
"""
|
|
85
|
+
try:
|
|
86
|
+
job = ctx.store.components.get(UUID(component_id), kind="job", org_id=ctx.org_id)
|
|
87
|
+
total = ctx.store.runs.count(ctx.org_id, component_id=job.id, status="success")
|
|
88
|
+
runs = ctx.store.runs.list_all(ctx.org_id, component_id=job.id, status="success", limit=total)
|
|
89
|
+
|
|
90
|
+
start = datetime.date.fromisoformat(start_date)
|
|
91
|
+
end = datetime.date.fromisoformat(end_date)
|
|
92
|
+
|
|
93
|
+
# Coverage is a daily question, so a run covers every day inside its partition.
|
|
94
|
+
covered: set[datetime.date] = set()
|
|
95
|
+
for r in runs:
|
|
96
|
+
if not r.partition_key:
|
|
97
|
+
continue
|
|
98
|
+
p_start, p_end = TimePartition.from_key(r.partition_key).bounds
|
|
99
|
+
if isinstance(p_start, datetime.datetime):
|
|
100
|
+
days = [p_start.date()]
|
|
101
|
+
else:
|
|
102
|
+
days = []
|
|
103
|
+
current = p_start
|
|
104
|
+
while current < p_end:
|
|
105
|
+
days.append(current)
|
|
106
|
+
current += datetime.timedelta(days=1)
|
|
107
|
+
covered.update(day for day in days if start <= day <= end)
|
|
108
|
+
|
|
109
|
+
# Build expected date range
|
|
110
|
+
expected: list[datetime.date] = []
|
|
111
|
+
current = start
|
|
112
|
+
while current <= end:
|
|
113
|
+
expected.append(current)
|
|
114
|
+
current += datetime.timedelta(days=1)
|
|
115
|
+
|
|
116
|
+
missing = sorted(set(expected) - covered)
|
|
117
|
+
coverage_pct = round(len(covered) / len(expected) * 100, 1) if expected else 100.0
|
|
118
|
+
|
|
119
|
+
return PartitionCoverage(
|
|
120
|
+
component_id=component_id,
|
|
121
|
+
start_date=start_date,
|
|
122
|
+
end_date=end_date,
|
|
123
|
+
total_days=len(expected),
|
|
124
|
+
covered_days=len(covered),
|
|
125
|
+
missing_days=len(missing),
|
|
126
|
+
coverage_percent=coverage_pct,
|
|
127
|
+
missing_dates=[d.isoformat() for d in missing],
|
|
128
|
+
)
|
|
129
|
+
except Exception as e:
|
|
130
|
+
return ToolError(error=str(e))
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
def freshness_check(ctx: ToolkitContext) -> FreshnessReport | ToolError:
|
|
134
|
+
"""Check data freshness for all jobs.
|
|
135
|
+
|
|
136
|
+
Returns the last successful run timestamp for each job and flags
|
|
137
|
+
any that haven't succeeded in over 24 hours.
|
|
138
|
+
"""
|
|
139
|
+
try:
|
|
140
|
+
jobs = ctx.store.components.list_all(ctx.org_id, kinds=["job"])
|
|
141
|
+
now = datetime.datetime.now(tz=datetime.timezone.utc)
|
|
142
|
+
|
|
143
|
+
results = []
|
|
144
|
+
for job in jobs:
|
|
145
|
+
if not (job.config or {}).get("enabled", True):
|
|
146
|
+
continue
|
|
147
|
+
component_id = job.id
|
|
148
|
+
runs = ctx.store.runs.list_all(ctx.org_id, component_id=component_id, status="success", limit=1)
|
|
149
|
+
last_success = runs[0] if runs else None
|
|
150
|
+
|
|
151
|
+
hours_since = None
|
|
152
|
+
if last_success and last_success.completed_at:
|
|
153
|
+
delta = now - last_success.completed_at
|
|
154
|
+
hours_since = round(delta.total_seconds() / 3600, 1)
|
|
155
|
+
|
|
156
|
+
results.append(JobFreshness(
|
|
157
|
+
job=job,
|
|
158
|
+
last_success_at=last_success.completed_at if last_success else None,
|
|
159
|
+
hours_since_success=hours_since,
|
|
160
|
+
stale=hours_since is None or hours_since > 24,
|
|
161
|
+
))
|
|
162
|
+
|
|
163
|
+
stale_count = sum(1 for r in results if r.stale)
|
|
164
|
+
return FreshnessReport(
|
|
165
|
+
total_jobs=len(results),
|
|
166
|
+
stale_count=stale_count,
|
|
167
|
+
jobs=results,
|
|
168
|
+
)
|
|
169
|
+
except Exception as e:
|
|
170
|
+
return ToolError(error=str(e))
|
|
171
|
+
|
|
172
|
+
|
|
173
|
+
def run_stats(
|
|
174
|
+
ctx: ToolkitContext,
|
|
175
|
+
since: str | None = None,
|
|
176
|
+
until: str | None = None,
|
|
177
|
+
component_id: str | None = None,
|
|
178
|
+
limit: int = 25,
|
|
179
|
+
offset: int = 0,
|
|
180
|
+
) -> RunStats | ToolError:
|
|
181
|
+
"""Per-job run statistics over a window: verdicts, durations and retries.
|
|
182
|
+
|
|
183
|
+
Args:
|
|
184
|
+
since: ISO date or datetime the window opens at (default: 7 days ago).
|
|
185
|
+
until: ISO date or datetime the window closes before (default: open).
|
|
186
|
+
component_id: Restrict to this job UUID.
|
|
187
|
+
limit: Maximum number of jobs to return (default 25).
|
|
188
|
+
offset: Number of jobs to skip, for paging past the first page.
|
|
189
|
+
|
|
190
|
+
Returns one row per job that ran in the window, most failures first: run
|
|
191
|
+
stacks by final status, attempts, duration p50/p90/max in seconds, and
|
|
192
|
+
how many retried stacks healed or are still failing.
|
|
193
|
+
"""
|
|
194
|
+
try:
|
|
195
|
+
start, end = window(since, until, default_days=7)
|
|
196
|
+
filters: dict[str, Any] = {
|
|
197
|
+
"component_id": UUID(component_id) if component_id else None,
|
|
198
|
+
"after": start,
|
|
199
|
+
"before": end,
|
|
200
|
+
"all_attempts": True,
|
|
201
|
+
}
|
|
202
|
+
total = ctx.store.runs.count(ctx.org_id, **filters)
|
|
203
|
+
runs = ctx.store.runs.list_all(ctx.org_id, **filters, limit=total)
|
|
204
|
+
|
|
205
|
+
stacks: dict[UUID | None, dict[UUID, list[Any]]] = {}
|
|
206
|
+
names: dict[UUID | None, str | None] = {}
|
|
207
|
+
for r in runs:
|
|
208
|
+
stacks.setdefault(r.component_id, {}).setdefault(r.root_run_id, []).append(r)
|
|
209
|
+
names.setdefault(r.component_id, r.target.name if r.target else None)
|
|
210
|
+
|
|
211
|
+
rows = []
|
|
212
|
+
for job_id, by_root in stacks.items():
|
|
213
|
+
by_status: dict[str, int] = {}
|
|
214
|
+
durations: list[float] = []
|
|
215
|
+
attempts = retried = healed = still_failing = 0
|
|
216
|
+
for chain in by_root.values():
|
|
217
|
+
latest = max(chain, key=lambda r: r.attempt)
|
|
218
|
+
by_status[latest.status] = by_status.get(latest.status, 0) + 1
|
|
219
|
+
attempts += len(chain)
|
|
220
|
+
durations += [
|
|
221
|
+
(r.completed_at - r.started_at).total_seconds() for r in chain if r.started_at and r.completed_at
|
|
222
|
+
]
|
|
223
|
+
if len(chain) > 1:
|
|
224
|
+
retried += 1
|
|
225
|
+
healed += latest.status == "success"
|
|
226
|
+
still_failing += latest.status == "failed"
|
|
227
|
+
rows.append(
|
|
228
|
+
JobStats(
|
|
229
|
+
job_id=job_id,
|
|
230
|
+
job_name=names[job_id],
|
|
231
|
+
stacks=by_status,
|
|
232
|
+
attempts=attempts,
|
|
233
|
+
duration_p50_s=_rounded(percentile(durations, 50)),
|
|
234
|
+
duration_p90_s=_rounded(percentile(durations, 90)),
|
|
235
|
+
duration_max_s=_rounded(max(durations, default=None)),
|
|
236
|
+
stacks_retried=retried,
|
|
237
|
+
healed=healed,
|
|
238
|
+
still_failing=still_failing,
|
|
239
|
+
)
|
|
240
|
+
)
|
|
241
|
+
rows.sort(key=lambda j: (-j.stacks.get("failed", 0), j.job_name or ""))
|
|
242
|
+
page = rows[offset : offset + limit]
|
|
243
|
+
return RunStats(since=start, until=end, count=len(page), total=len(rows), jobs=page)
|
|
244
|
+
except Exception as e:
|
|
245
|
+
return ToolError(error=str(e))
|
|
246
|
+
|
|
247
|
+
|
|
248
|
+
def asset_coverage(
|
|
249
|
+
ctx: ToolkitContext,
|
|
250
|
+
component_id: str,
|
|
251
|
+
start_key: str,
|
|
252
|
+
end_key: str,
|
|
253
|
+
limit: int = 50,
|
|
254
|
+
offset: int = 0,
|
|
255
|
+
) -> AssetCoverage | ToolError:
|
|
256
|
+
"""Per-asset partition coverage of a job over a range of partition keys.
|
|
257
|
+
|
|
258
|
+
Unlike partition_coverage, which needs a whole run to have succeeded, an
|
|
259
|
+
asset counts as covered for a partition once any run's execution of it
|
|
260
|
+
succeeded, so a run where most assets succeeded shows what is actually
|
|
261
|
+
missing. Only assets that executed at least once in the range appear.
|
|
262
|
+
|
|
263
|
+
Args:
|
|
264
|
+
component_id: UUID of the job.
|
|
265
|
+
start_key: First partition key, in the job's granularity (2026-07-01,
|
|
266
|
+
2026-07, 2026, or 2026-07-01T13).
|
|
267
|
+
end_key: Last partition key, inclusive; must share the start key's
|
|
268
|
+
granularity.
|
|
269
|
+
limit: Maximum number of assets to return (default 50).
|
|
270
|
+
offset: Number of assets to skip, for paging past the first page.
|
|
271
|
+
|
|
272
|
+
Returns the page of assets, least covered first, each with its covered,
|
|
273
|
+
failed and never-run partition counts and its missing partitions as
|
|
274
|
+
ranges ready for a backfill, plus a rollup of the range's partitions by
|
|
275
|
+
whether all, some or none of the assets are covered.
|
|
276
|
+
"""
|
|
277
|
+
try:
|
|
278
|
+
job = ctx.store.components.get(UUID(component_id), kind="job", org_id=ctx.org_id)
|
|
279
|
+
first, last = TimePartition.from_key(start_key), TimePartition.from_key(end_key)
|
|
280
|
+
if first.granularity is not last.granularity:
|
|
281
|
+
return ToolError(error=f"{start_key!r} and {end_key!r} are keys of different granularities")
|
|
282
|
+
granularity = first.granularity
|
|
283
|
+
expected = [granularity.format(value) for value in granularity.period_range(first.value, last.value)]
|
|
284
|
+
|
|
285
|
+
rows = ctx.store.events.partition_coverage(ctx.org_id, job.id, start_key, end_key)
|
|
286
|
+
keys: dict[UUID, str | None] = {}
|
|
287
|
+
attempted: dict[UUID, set[str]] = {}
|
|
288
|
+
covered: dict[UUID, set[str]] = {}
|
|
289
|
+
for row in rows:
|
|
290
|
+
keys.setdefault(row.component_id, row.component_key)
|
|
291
|
+
attempted.setdefault(row.component_id, set()).add(row.partition_key)
|
|
292
|
+
if row.succeeded:
|
|
293
|
+
covered.setdefault(row.component_id, set()).add(row.partition_key)
|
|
294
|
+
|
|
295
|
+
assets = []
|
|
296
|
+
for asset_id, asset_key in keys.items():
|
|
297
|
+
done = covered.get(asset_id, set())
|
|
298
|
+
missing = [key for key in expected if key not in done]
|
|
299
|
+
ranges = _ranges(missing, expected)
|
|
300
|
+
assets.append(
|
|
301
|
+
AssetCoverageRow(
|
|
302
|
+
asset_id=asset_id,
|
|
303
|
+
asset_key=asset_key,
|
|
304
|
+
covered=len(done),
|
|
305
|
+
failed=len(attempted[asset_id] - done),
|
|
306
|
+
never_run=len(missing) - len(attempted[asset_id] - done),
|
|
307
|
+
missing=ranges[:_MISSING_RANGES_LIMIT],
|
|
308
|
+
missing_ranges_total=len(ranges),
|
|
309
|
+
)
|
|
310
|
+
)
|
|
311
|
+
assets.sort(key=lambda a: (a.covered, a.asset_key or ""))
|
|
312
|
+
|
|
313
|
+
per_partition = [sum(key in covered.get(asset_id, ()) for asset_id in keys) for key in expected]
|
|
314
|
+
page = assets[offset : offset + limit]
|
|
315
|
+
return AssetCoverage(
|
|
316
|
+
component_id=job.id,
|
|
317
|
+
start_key=start_key,
|
|
318
|
+
end_key=end_key,
|
|
319
|
+
partitions=len(expected),
|
|
320
|
+
all_covered=sum(n == len(keys) for n in per_partition) if keys else 0,
|
|
321
|
+
partly_covered=sum(0 < n < len(keys) for n in per_partition),
|
|
322
|
+
none_covered=sum(n == 0 for n in per_partition) if keys else len(expected),
|
|
323
|
+
count=len(page),
|
|
324
|
+
total=len(assets),
|
|
325
|
+
assets=page,
|
|
326
|
+
)
|
|
327
|
+
except Exception as e:
|
|
328
|
+
return ToolError(error=str(e))
|
|
329
|
+
|
|
330
|
+
|
|
331
|
+
def _ranges(missing: list[str], expected: list[str]) -> list[PartitionRange]:
|
|
332
|
+
"""Collapse the missing keys into runs of consecutive partitions.
|
|
333
|
+
|
|
334
|
+
Args:
|
|
335
|
+
missing: The uncovered keys, in *expected*'s order.
|
|
336
|
+
expected: Every key of the range, in order.
|
|
337
|
+
|
|
338
|
+
Returns:
|
|
339
|
+
One inclusive range per run of consecutive missing keys.
|
|
340
|
+
"""
|
|
341
|
+
position = {key: index for index, key in enumerate(expected)}
|
|
342
|
+
ranges: list[PartitionRange] = []
|
|
343
|
+
for key in missing:
|
|
344
|
+
if ranges and position[key] == position[ranges[-1].end_key] + 1:
|
|
345
|
+
ranges[-1].end_key = key
|
|
346
|
+
else:
|
|
347
|
+
ranges.append(PartitionRange(start_key=key, end_key=key))
|
|
348
|
+
return ranges
|
|
349
|
+
|
|
350
|
+
|
|
351
|
+
def _rounded(value: float | None) -> float | None:
|
|
352
|
+
"""Round a duration to a tenth of a second, passing ``None`` through.
|
|
353
|
+
|
|
354
|
+
Args:
|
|
355
|
+
value: Seconds, or ``None`` when there was nothing to measure.
|
|
356
|
+
|
|
357
|
+
Returns:
|
|
358
|
+
The rounded value, or ``None``.
|
|
359
|
+
"""
|
|
360
|
+
return round(value, 1) if value is not None else None
|
|
@@ -0,0 +1,73 @@
|
|
|
1
|
+
"""Role gating for the toolkit's write tools.
|
|
2
|
+
|
|
3
|
+
The context carries the caller's role in the organisation; a write declares
|
|
4
|
+
the role it needs and refuses, as a structured :class:`ToolError`, before its
|
|
5
|
+
body runs. The ranks and the refusal wording mirror the API's route gates, so
|
|
6
|
+
a caller can do through a tool exactly what it could do through the app.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import functools
|
|
12
|
+
import inspect
|
|
13
|
+
from collections.abc import Callable
|
|
14
|
+
from typing import Any, TypeVar, cast
|
|
15
|
+
|
|
16
|
+
from interloper_toolkit.context import ToolkitContext
|
|
17
|
+
from interloper_toolkit.models import ToolError
|
|
18
|
+
|
|
19
|
+
_ROLE_RANK = {"viewer": 0, "editor": 1, "admin": 2}
|
|
20
|
+
|
|
21
|
+
F = TypeVar("F", bound=Callable[..., Any])
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def requires_role(minimum: str) -> Callable[[F], F]:
|
|
25
|
+
"""Refuse a tool call whose context holds a role below *minimum*.
|
|
26
|
+
|
|
27
|
+
Works on sync and async tool functions alike; the context is the first
|
|
28
|
+
positional argument, as it is for every toolkit function. The wrapped
|
|
29
|
+
function keeps its name and docstring, which the AI surfaces adopt.
|
|
30
|
+
|
|
31
|
+
Args:
|
|
32
|
+
minimum: The lowest role allowed: ``viewer``, ``editor`` or ``admin``.
|
|
33
|
+
|
|
34
|
+
Returns:
|
|
35
|
+
The decorator to apply to a tool function.
|
|
36
|
+
|
|
37
|
+
Raises:
|
|
38
|
+
ValueError: If *minimum* is not a known role.
|
|
39
|
+
"""
|
|
40
|
+
if minimum not in _ROLE_RANK:
|
|
41
|
+
raise ValueError(f"Unknown role {minimum!r}; expected one of {sorted(_ROLE_RANK)}")
|
|
42
|
+
|
|
43
|
+
def decorate(func: F) -> F:
|
|
44
|
+
if inspect.iscoroutinefunction(func):
|
|
45
|
+
|
|
46
|
+
@functools.wraps(func)
|
|
47
|
+
async def async_gate(ctx: ToolkitContext, *args: Any, **kwargs: Any) -> Any:
|
|
48
|
+
return denied(ctx.role, minimum) or await func(ctx, *args, **kwargs)
|
|
49
|
+
|
|
50
|
+
return cast(F, async_gate)
|
|
51
|
+
|
|
52
|
+
@functools.wraps(func)
|
|
53
|
+
def gate(ctx: ToolkitContext, *args: Any, **kwargs: Any) -> Any:
|
|
54
|
+
return denied(ctx.role, minimum) or func(ctx, *args, **kwargs)
|
|
55
|
+
|
|
56
|
+
return cast(F, gate)
|
|
57
|
+
|
|
58
|
+
return decorate
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def denied(role: str, minimum: str) -> ToolError | None:
|
|
62
|
+
"""The refusal for a role below *minimum*, or ``None`` when the role suffices.
|
|
63
|
+
|
|
64
|
+
Args:
|
|
65
|
+
role: The caller's role; an unknown one ranks below every known role.
|
|
66
|
+
minimum: The lowest role allowed.
|
|
67
|
+
|
|
68
|
+
Returns:
|
|
69
|
+
The structured error, or ``None`` when the call may proceed.
|
|
70
|
+
"""
|
|
71
|
+
if _ROLE_RANK.get(role, -1) < _ROLE_RANK[minimum]:
|
|
72
|
+
return ToolError(error=f"Requires {minimum} role or higher")
|
|
73
|
+
return None
|
|
@@ -81,7 +81,7 @@ def list_definitions(
|
|
|
81
81
|
if kind not in counts:
|
|
82
82
|
return ToolError(
|
|
83
83
|
error=f"No '{kind}' definitions in the catalog",
|
|
84
|
-
|
|
84
|
+
valid_values=sorted(counts),
|
|
85
85
|
)
|
|
86
86
|
|
|
87
87
|
collection_counts: dict[str, int] = {}
|
|
@@ -179,7 +179,7 @@ def get_asset_schema(ctx: ToolkitContext, source_key: str, asset_key: str) -> As
|
|
|
179
179
|
return ToolError(error=str(e))
|
|
180
180
|
|
|
181
181
|
|
|
182
|
-
def search_fields(ctx: ToolkitContext, query: str) -> FieldSearchResult | ToolError:
|
|
182
|
+
def search_fields(ctx: ToolkitContext, query: str, limit: int = 50, offset: int = 0) -> FieldSearchResult | ToolError:
|
|
183
183
|
"""Search for fields across all asset schemas in the catalog matching a query string.
|
|
184
184
|
|
|
185
185
|
Searches every source definition's schemas, whether or not the source is
|
|
@@ -187,8 +187,11 @@ def search_fields(ctx: ToolkitContext, query: str) -> FieldSearchResult | ToolEr
|
|
|
187
187
|
|
|
188
188
|
Args:
|
|
189
189
|
query: Substring to search for in field names and descriptions (case-insensitive).
|
|
190
|
+
limit: Maximum number of matches to return (default 50).
|
|
191
|
+
offset: Number of matches to skip, for paging past the first page.
|
|
190
192
|
|
|
191
|
-
Returns matching fields grouped by source and asset, with
|
|
193
|
+
Returns the page of matching fields grouped by source and asset, with
|
|
194
|
+
field type and description, plus the total number of matches.
|
|
192
195
|
"""
|
|
193
196
|
try:
|
|
194
197
|
query_lower = query.lower()
|
|
@@ -215,7 +218,8 @@ def search_fields(ctx: ToolkitContext, query: str) -> FieldSearchResult | ToolEr
|
|
|
215
218
|
description=field_desc,
|
|
216
219
|
))
|
|
217
220
|
|
|
218
|
-
|
|
221
|
+
page = matches[offset : offset + limit]
|
|
222
|
+
return FieldSearchResult(query=query, match_count=len(page), total=len(matches), matches=page)
|
|
219
223
|
except Exception as e:
|
|
220
224
|
return ToolError(error=str(e))
|
|
221
225
|
|