unstract-cli 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,297 @@
1
+ """`unstract clone` -- copying one organization's resources into another.
2
+
3
+ Two endpoints, each with its own key, so this command takes them as flags rather
4
+ than from a profile: a profile describes one connection.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ import logging
10
+ import re
11
+ from typing import Any
12
+
13
+ import click
14
+ from requests.exceptions import InvalidHeader, RequestException
15
+ from unstract.clone.context import (
16
+ DEFAULT_CONCURRENCY,
17
+ CloneOptions,
18
+ OrgEndpoint,
19
+ )
20
+ from unstract.clone.exceptions import CloneError, PlatformAPIError
21
+ from unstract.clone.orchestrator import clone as run_clone
22
+ from unstract.clone.report import CloneReport
23
+
24
+ from unstract_cli.app import Context, cli, pass_context
25
+ from unstract_cli.commands.common import finish
26
+ from unstract_cli.core.clients import UNSENDABLE
27
+ from unstract_cli.core.errors import (
28
+ CLIError,
29
+ ExitCode,
30
+ error_from_status,
31
+ remember_secret,
32
+ )
33
+ from unstract_cli.core.output import OutputFormat, diagnostic, emit_text
34
+
35
+ # Mirrors the units `unstract.clone.cli` accepts, so both spellings of this
36
+ # command take the same strings.
37
+ _SIZE_UNITS = {
38
+ "B": 1,
39
+ "K": 1024,
40
+ "KB": 1024,
41
+ "M": 1024**2,
42
+ "MB": 1024**2,
43
+ "G": 1024**3,
44
+ "GB": 1024**3,
45
+ }
46
+ _SIZE_RE = re.compile(r"^\s*(\d+(?:\.\d+)?)\s*([A-Za-z]*)\s*$")
47
+
48
+
49
+ def _parse_size(value: str) -> int:
50
+ """Accept ``25``, ``25MB``, ``1.5GB`` etc. Returns bytes."""
51
+ match = _SIZE_RE.match(value)
52
+ if not match:
53
+ raise click.BadParameter(f"can't parse size {value!r}")
54
+ number, unit = match.group(1), match.group(2).upper() or "B"
55
+ if unit not in _SIZE_UNITS:
56
+ raise click.BadParameter(
57
+ f"unknown size unit {unit!r}; use one of {sorted(_SIZE_UNITS)}"
58
+ )
59
+ return int(float(number) * _SIZE_UNITS[unit])
60
+
61
+
62
+ def _split_csv(value: str | None) -> tuple[str, ...] | None:
63
+ if not value:
64
+ return None
65
+ return tuple(part.strip() for part in value.split(",") if part.strip())
66
+
67
+
68
+ @cli.command("clone")
69
+ @click.option("--source-url", required=True, help="Base URL of the source deployment.")
70
+ @click.option(
71
+ "--source-org", required=True, help="Source organization_id (slug in the URL path)."
72
+ )
73
+ @click.option(
74
+ "--source-key",
75
+ envvar="UNSTRACT_SRC_PLATFORM_KEY",
76
+ required=True,
77
+ help="Source admin's Platform API key (or env UNSTRACT_SRC_PLATFORM_KEY).",
78
+ )
79
+ @click.option("--target-url", required=True, help="Base URL of the target deployment.")
80
+ @click.option(
81
+ "--target-org", required=True, help="Target organization_id (slug in the URL path)."
82
+ )
83
+ @click.option(
84
+ "--target-key",
85
+ envvar="UNSTRACT_TGT_PLATFORM_KEY",
86
+ required=True,
87
+ help="Target admin's Platform API key (or env UNSTRACT_TGT_PLATFORM_KEY).",
88
+ )
89
+ @click.option(
90
+ "--dry-run", is_flag=True, help="Plan only -- do not write anything to the target."
91
+ )
92
+ @click.option(
93
+ "--include", default=None, help="Comma-separated phases to run (default: all)."
94
+ )
95
+ @click.option("--exclude", default=None, help="Comma-separated phases to skip.")
96
+ @click.option(
97
+ "--on-name-conflict",
98
+ type=click.Choice(["adopt", "abort"]),
99
+ default="adopt",
100
+ show_default=True,
101
+ help="What to do when a like-named entity exists on the target.",
102
+ )
103
+ @click.option(
104
+ "--api-prefix",
105
+ default="api/v1",
106
+ show_default=True,
107
+ help="Backend URL prefix, matching the deployment's own.",
108
+ )
109
+ @click.option(
110
+ "--file-strategy",
111
+ type=click.Choice(["platform_api", "skip"]),
112
+ default="platform_api",
113
+ show_default=True,
114
+ help="How to move Prompt Studio documents. 'skip' copies metadata only.",
115
+ )
116
+ @click.option("--skip-files", is_flag=True, help="Alias for --file-strategy=skip.")
117
+ @click.option(
118
+ "--max-file-size",
119
+ default="25MB",
120
+ show_default=True,
121
+ help="Per-file cap for the files phase. Oversize files are reported, not fatal.",
122
+ )
123
+ @click.option(
124
+ "--concurrency",
125
+ type=click.IntRange(min=1, max=32),
126
+ default=DEFAULT_CONCURRENCY,
127
+ show_default=True,
128
+ help="Per-phase worker count. 1 is strictly sequential.",
129
+ )
130
+ @click.option(
131
+ "--clone-group-members",
132
+ is_flag=True,
133
+ help="Also add group members on the target, matched by email.",
134
+ )
135
+ @pass_context
136
+ def clone(
137
+ ctx: Context,
138
+ source_url: str,
139
+ source_org: str,
140
+ source_key: str,
141
+ target_url: str,
142
+ target_org: str,
143
+ target_key: str,
144
+ **params: Any,
145
+ ) -> None:
146
+ """Copy an organization's resources into another organization.
147
+
148
+ Adapters, connectors, workflows, pipelines, API deployments, Prompt Studio
149
+ projects and their files, user groups and sharing state. Run --dry-run first:
150
+ it reports what would be written without writing it.
151
+ """
152
+ for key in (source_key, target_key):
153
+ remember_secret(key)
154
+ _configure_logging(ctx)
155
+
156
+ options = CloneOptions(
157
+ dry_run=params["dry_run"],
158
+ include=_split_csv(params["include"]),
159
+ exclude=_split_csv(params["exclude"]) or (),
160
+ on_name_conflict=params["on_name_conflict"],
161
+ verbose=ctx.verbosity > 0,
162
+ file_strategy="skip" if params["skip_files"] else params["file_strategy"],
163
+ max_file_size=_parse_size(params["max_file_size"]),
164
+ concurrency=params["concurrency"],
165
+ clone_group_members=params["clone_group_members"],
166
+ )
167
+
168
+ def endpoint(url: str, org: str, key: str) -> OrgEndpoint:
169
+ return OrgEndpoint(
170
+ base_url=url,
171
+ organization_id=org,
172
+ platform_key=key,
173
+ api_path_prefix=params["api_prefix"],
174
+ )
175
+
176
+ try:
177
+ report = run_clone(
178
+ endpoint(source_url, source_org, source_key),
179
+ endpoint(target_url, target_org, target_key),
180
+ options,
181
+ )
182
+ except PlatformAPIError as exc:
183
+ # The exception appends the response body to its own message, and
184
+ # `error.message` is published as a one-line summary; the body reaches
185
+ # the caller through `details`.
186
+ message = str(exc).split("\n body:", 1)[0]
187
+ if exc.status_code:
188
+ raise error_from_status(
189
+ int(exc.status_code), message, details=exc.body
190
+ ) from exc
191
+ raise CLIError(
192
+ message,
193
+ ExitCode.SERVER_ERROR,
194
+ details=exc.body,
195
+ retryable=True,
196
+ hint="The Platform API did not answer. Check the URLs and connectivity.",
197
+ ) from exc
198
+ except CloneError as exc:
199
+ raise CLIError(
200
+ str(exc),
201
+ ExitCode.USAGE,
202
+ hint="The clone could not start. Check the URLs, orgs and keys.",
203
+ ) from exc
204
+ except InvalidHeader as exc:
205
+ # The message quotes the offending header value -- the key -- and it
206
+ # arrives `repr`-escaped, so the scrub cannot match it either.
207
+ raise CLIError(
208
+ "A request header could not be built.",
209
+ ExitCode.USAGE,
210
+ hint=(
211
+ "A platform key most likely carries a newline or a control "
212
+ "character. Check how it is stored."
213
+ ),
214
+ ) from exc
215
+ except UNSENDABLE as exc:
216
+ raise CLIError(
217
+ str(exc) or type(exc).__name__,
218
+ ExitCode.USAGE,
219
+ hint=(
220
+ "The request could not be built. Check the URLs, and any proxy "
221
+ "variables, for a typo."
222
+ ),
223
+ ) from exc
224
+ except RequestException as exc:
225
+ raise CLIError(
226
+ str(exc) or type(exc).__name__,
227
+ ExitCode.SERVER_ERROR,
228
+ retryable=True,
229
+ hint="The request failed in transit rather than being answered.",
230
+ ) from exc
231
+
232
+ _finish(ctx, report)
233
+
234
+
235
+ def _configure_logging(ctx: Context) -> None:
236
+ """Send the orchestrator's progress to stderr, at the run's own verbosity."""
237
+ logging.basicConfig(
238
+ level=logging.WARNING
239
+ if ctx.quiet
240
+ else (logging.DEBUG if ctx.verbosity else logging.INFO),
241
+ format="%(asctime)s %(levelname)-7s %(name)s: %(message)s",
242
+ datefmt="%H:%M:%S",
243
+ )
244
+
245
+
246
+ def _skipped(report: CloneReport) -> dict[str, Any]:
247
+ """What the run did not copy, summarised at the top of the payload.
248
+
249
+ Skipping an oversize or unsupported file is reported rather than fatal, so
250
+ the run still exits 0; a consumer reading only the exit code would otherwise
251
+ have to walk the whole report to discover documents that never arrived.
252
+ """
253
+ by_phase = {phase.name: phase.skipped for phase in report.phases if phase.skipped}
254
+ return {
255
+ "total": sum(by_phase.values()),
256
+ "by_phase": by_phase,
257
+ "oversize_files": len(report.oversize_files),
258
+ "unsupported_files": len(report.unsupported_files),
259
+ }
260
+
261
+
262
+ def _finish(ctx: Context, report: CloneReport) -> None:
263
+ """Emit the report, then fail if the clone did not fully succeed."""
264
+ failure = None
265
+ if report.aborted:
266
+ failure = f"Clone aborted: {report.abort_reason}"
267
+ elif failed := [phase.name for phase in report.phases if phase.failed]:
268
+ failure = f"Clone completed with failures in: {', '.join(sorted(failed))}"
269
+
270
+ skipped = _skipped(report)
271
+ counts = {
272
+ **skipped["by_phase"],
273
+ "oversize files": skipped["oversize_files"],
274
+ "unsupported files": skipped["unsupported_files"],
275
+ }
276
+ if named := ", ".join(f"{what} {n}" for what, n in counts.items() if n):
277
+ # On stderr in every format: a skip does not fail the run.
278
+ diagnostic(f"Skipped: {named}.", quiet=ctx.quiet, verbosity=ctx.verbosity)
279
+
280
+ payload = {**report.as_dict(), "skipped": skipped}
281
+ rendered = ctx.output is OutputFormat.TABLE
282
+ if rendered:
283
+ emit_text(report.render(), secrets=ctx.secrets())
284
+ elif not failure:
285
+ finish(ctx, payload)
286
+
287
+ if failure:
288
+ raise CLIError(
289
+ failure,
290
+ ExitCode.GENERIC,
291
+ details=None if rendered else payload,
292
+ hint="The report lists what was copied and what was not. Re-running "
293
+ "adopts what already exists on the target rather than duplicating it.",
294
+ )
295
+
296
+
297
+ __all__ = ["clone"]
@@ -0,0 +1,114 @@
1
+ """Pieces every product command shares: the wait flags and result emission."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from collections.abc import Callable
6
+ from typing import Any
7
+
8
+ import click
9
+
10
+ from unstract_cli.app import Context
11
+ from unstract_cli.core.output import emit_result
12
+
13
+ #: Poll interval and the ceiling on the whole wait. Both are flags.
14
+ DEFAULT_INTERVAL = 3.0
15
+ DEFAULT_TIMEOUT = 300.0
16
+
17
+ #: Below this, polling is a busy loop against a metered service, not a wait.
18
+ MIN_INTERVAL = 0.1
19
+
20
+ F = Callable[..., Any]
21
+
22
+
23
+ def wait_options(*, default: bool = True) -> Callable[[F], F]:
24
+ """`--wait` and its two knobs.
25
+
26
+ ``--wait`` is a gate, not a duration: how long to wait is ``--timeout`` and
27
+ how often to check is ``--interval``, so neither has two spellings.
28
+ """
29
+
30
+ def decorate(func: F) -> F:
31
+ for option in reversed(
32
+ [
33
+ click.option(
34
+ "--wait/--no-wait",
35
+ default=default,
36
+ help="Poll until the job reaches a terminal state.",
37
+ ),
38
+ click.option(
39
+ "--interval",
40
+ # Bounded below: an interval of zero polls a metered service
41
+ # as fast as the loop can issue calls.
42
+ type=click.FloatRange(min=MIN_INTERVAL),
43
+ default=DEFAULT_INTERVAL,
44
+ show_default=True,
45
+ help="Seconds between polls.",
46
+ ),
47
+ click.option(
48
+ "--timeout",
49
+ "wait_timeout",
50
+ # Zero is meaningful -- one poll, then give up -- but a
51
+ # negative deadline has already passed.
52
+ type=click.FloatRange(min=0),
53
+ default=DEFAULT_TIMEOUT,
54
+ show_default=True,
55
+ help="Seconds to wait before giving up. The job keeps running.",
56
+ ),
57
+ click.option(
58
+ "--save",
59
+ type=click.Path(dir_okay=False),
60
+ default=None,
61
+ help="Write the result here before printing it.",
62
+ ),
63
+ ]
64
+ ):
65
+ func = option(func)
66
+ return func
67
+
68
+ return decorate
69
+
70
+
71
+ def raw_fields(*fields: str) -> Callable[[click.Command], click.Command]:
72
+ """Declare what `--output raw` prints for this command, best answer first.
73
+
74
+ Several, because one command has several answers: a queued run replies with
75
+ a handle and no result, and a status read replies with a state until there
76
+ is a result. Raw prints the first of these the answer actually carries.
77
+
78
+ Recorded on the command so `--discover full` can report the whole list: a
79
+ caller asking for raw output has to know what it is going to get, and one
80
+ field named there would be wrong for every other shape the command returns.
81
+ """
82
+
83
+ def decorate(command: click.Command) -> click.Command:
84
+ command.raw_fields = fields
85
+ return command
86
+
87
+ return decorate
88
+
89
+
90
+ def finish(
91
+ ctx: Context,
92
+ data: Any,
93
+ *,
94
+ raw_fields: tuple[str, ...] = (),
95
+ meta: dict[str, Any] | None = None,
96
+ ) -> None:
97
+ """Emit one result envelope, scrubbing any resolved credential from it."""
98
+ emit_result(
99
+ data,
100
+ ctx.output,
101
+ meta=meta,
102
+ raw_fields=raw_fields,
103
+ secrets=ctx.secrets(),
104
+ )
105
+
106
+
107
+ __all__ = [
108
+ "DEFAULT_INTERVAL",
109
+ "DEFAULT_TIMEOUT",
110
+ "MIN_INTERVAL",
111
+ "finish",
112
+ "raw_fields",
113
+ "wait_options",
114
+ ]