alembic-runner 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,61 @@
1
+ """run alembic plan, review and apply from a host application."""
2
+
3
+ from .backend import CONFIG_KEYS, KINDS, TOKEN_ENV, Backend, BackendError, StateStore, token_env
4
+ from .plan import ApplyReport, DocumentError, DriftReport, Plan
5
+ from .runner import (
6
+ SUPPORTED,
7
+ ApplyOutcome,
8
+ Completed,
9
+ DriftOutcome,
10
+ Flow,
11
+ InventoryOutcome,
12
+ PlanOutcome,
13
+ Runner,
14
+ RunnerError,
15
+ StaleCheck,
16
+ )
17
+ from .status import (
18
+ ACTIVE,
19
+ TERMINAL,
20
+ TRANSITIONS,
21
+ RunStatus,
22
+ TransitionError,
23
+ check_transition,
24
+ failure_status,
25
+ )
26
+ from .workspace import Workspace, WorkspaceError
27
+
28
+ __version__ = "0.1.0"
29
+
30
+ __all__ = [
31
+ "ACTIVE",
32
+ "CONFIG_KEYS",
33
+ "KINDS",
34
+ "SUPPORTED",
35
+ "TERMINAL",
36
+ "TOKEN_ENV",
37
+ "TRANSITIONS",
38
+ "ApplyOutcome",
39
+ "ApplyReport",
40
+ "Backend",
41
+ "BackendError",
42
+ "Completed",
43
+ "DocumentError",
44
+ "DriftOutcome",
45
+ "DriftReport",
46
+ "Flow",
47
+ "InventoryOutcome",
48
+ "Plan",
49
+ "PlanOutcome",
50
+ "RunStatus",
51
+ "Runner",
52
+ "RunnerError",
53
+ "StaleCheck",
54
+ "StateStore",
55
+ "TransitionError",
56
+ "Workspace",
57
+ "WorkspaceError",
58
+ "check_transition",
59
+ "failure_status",
60
+ "token_env",
61
+ ]
@@ -0,0 +1,111 @@
1
+ """backends and state stores, as the environment and files a run hands to alembic."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from dataclasses import dataclass, field
6
+ from typing import Any
7
+
8
+ KINDS = ("netbox", "nautobot", "infrahub", "peeringdb", "external")
9
+
10
+ # the config keys a stored backend may set. anything that names a command, a
11
+ # working directory or an environment for a process stays in host settings:
12
+ # whoever can edit a backend must not be able to run something on the worker.
13
+ CONFIG_KEYS = {
14
+ "netbox": frozenset({"url", "instance"}),
15
+ "nautobot": frozenset({"url", "instance"}),
16
+ "infrahub": frozenset({"url", "instance"}),
17
+ "peeringdb": frozenset({"url", "instance"}),
18
+ "external": frozenset({"args", "setup", "timeout_seconds", "instance"}),
19
+ }
20
+
21
+ # the environment variable each built-in adapter reads its credential from.
22
+ TOKEN_ENV = {
23
+ "netbox": "NETBOX_TOKEN",
24
+ "nautobot": "NAUTOBOT_TOKEN",
25
+ "infrahub": "INFRAHUB_TOKEN",
26
+ "peeringdb": "PEERINGDB_API_KEY",
27
+ }
28
+
29
+
30
+ class BackendError(ValueError):
31
+ pass
32
+
33
+
34
+ @dataclass(frozen=True)
35
+ class Backend:
36
+ """one backend as a run sees it.
37
+
38
+ `config` is the non-secret part of an alembic backend config, without the
39
+ `backend:` key. `env` holds the credentials, by variable name. an external
40
+ backend's `command` comes from host settings, never from stored config.
41
+ """
42
+
43
+ kind: str
44
+ config: dict[str, Any] = field(default_factory=dict)
45
+ env: dict[str, str] = field(default_factory=dict)
46
+ command: str | None = None
47
+
48
+ def __post_init__(self) -> None:
49
+ if self.kind not in KINDS:
50
+ raise BackendError(f"unknown backend kind {self.kind!r}")
51
+ unknown = self.config.keys() - CONFIG_KEYS[self.kind]
52
+ if unknown:
53
+ raise BackendError(
54
+ f"a {self.kind} backend config cannot set {', '.join(sorted(unknown))}; "
55
+ f"it takes {', '.join(sorted(CONFIG_KEYS[self.kind]))}"
56
+ )
57
+ if self.kind == "external":
58
+ if not self.command:
59
+ raise BackendError("an external backend needs a command from host settings")
60
+ elif self.command:
61
+ raise BackendError(f"a {self.kind} backend takes no command")
62
+
63
+ def config_document(self) -> dict[str, Any]:
64
+ doc = {"backend": self.kind, **self.config}
65
+ if self.command:
66
+ doc["command"] = self.command
67
+ return doc
68
+
69
+
70
+ def token_env(kind: str, token: str) -> dict[str, str]:
71
+ """the environment that hands `token` to a built-in adapter of `kind`."""
72
+ try:
73
+ return {TOKEN_ENV[kind]: token}
74
+ except KeyError:
75
+ raise BackendError(f"a {kind} backend reads no token variable") from None
76
+
77
+
78
+ @dataclass(frozen=True)
79
+ class StateStore:
80
+ """where identity state lives. `local` keeps it in the flow directory."""
81
+
82
+ backend: str = "local"
83
+ postgres_url: str | None = None
84
+ postgres_tls: str | None = None
85
+
86
+ def __post_init__(self) -> None:
87
+ if self.backend not in ("local", "postgres"):
88
+ raise BackendError(f"unknown state backend {self.backend!r}")
89
+ if self.backend == "postgres" and not self.postgres_url:
90
+ raise BackendError("postgres state needs a postgres_url")
91
+
92
+ @classmethod
93
+ def from_settings(cls, settings: dict[str, Any] | None) -> StateStore:
94
+ settings = settings or {}
95
+ return cls(
96
+ backend=settings.get("backend", "local"),
97
+ postgres_url=settings.get("postgres_url"),
98
+ postgres_tls=settings.get("postgres_tls"),
99
+ )
100
+
101
+ def env(self, key: str) -> dict[str, str]:
102
+ if self.backend == "local":
103
+ return {"ALEMBIC_STATE_BACKEND": "local"}
104
+ env = {
105
+ "ALEMBIC_STATE_BACKEND": "postgres",
106
+ "ALEMBIC_STATE_POSTGRES_URL": self.postgres_url or "",
107
+ "ALEMBIC_STATE_KEY": key,
108
+ }
109
+ if self.postgres_tls:
110
+ env["ALEMBIC_STATE_POSTGRES_TLS"] = self.postgres_tls
111
+ return env
@@ -0,0 +1,197 @@
1
+ """run bookkeeping the host apps share: moves, locks, approval checks and the job
2
+ wrapper. it needs django, so only the host apps import it; the rest of the core
3
+ stays stdlib-only.
4
+
5
+ a host's run model carries the fields both apps define: `flow`, `kind`,
6
+ `status`, `requested_by`, `decided_by`, `decided_at`, `planned_at`,
7
+ `apply_started_at`, `finished_at`, `plan`, `plan_sha256`, `approved_sha256`,
8
+ `error`, `output`."""
9
+
10
+ from __future__ import annotations
11
+
12
+ from collections.abc import Callable, Iterable
13
+ from datetime import timedelta
14
+ from typing import Any
15
+
16
+ from django.core.exceptions import PermissionDenied, ValidationError
17
+ from django.db import IntegrityError, transaction
18
+ from django.utils import timezone
19
+
20
+ from .pipeline import Outcome, RunFailed, approved_plan
21
+ from .plan import Plan
22
+ from .status import RunStatus, TransitionError, check_transition, failure_status
23
+
24
+ S = RunStatus
25
+
26
+ # a new plan of a flow takes these runs' place.
27
+ SUPERSEDABLE = (S.AWAITING_APPROVAL.value, S.APPLY_FAILED.value)
28
+
29
+
30
+ class ActionError(ValidationError):
31
+ """a run cannot do what was asked of it in its current state."""
32
+
33
+
34
+ def _save(run) -> None:
35
+ # nautobot validates on save through `validated_save`; netbox has only `save`.
36
+ getattr(run, "validated_save", run.save)()
37
+
38
+
39
+ def move(run, new: RunStatus | str, **fields: Any):
40
+ """move a locked run to `new`, setting `fields`, or raise ActionError."""
41
+ try:
42
+ check_transition(run.status, new)
43
+ except TransitionError as e:
44
+ raise ActionError(str(e)) from e
45
+ run.status = RunStatus(new).value
46
+ for name, value in fields.items():
47
+ setattr(run, name, value)
48
+ _save(run)
49
+ return run
50
+
51
+
52
+ def locked(run):
53
+ return type(run).objects.select_for_update().get(pk=run.pk)
54
+
55
+
56
+ def expired(run, ttl_hours: int | None) -> bool:
57
+ return bool(
58
+ ttl_hours
59
+ and run.planned_at
60
+ and timezone.now() > run.planned_at + timedelta(hours=ttl_hours)
61
+ )
62
+
63
+
64
+ def create_run(model, flow, user, kind: str, supersede: bool, retire: Callable = lambda run: None):
65
+ """create a pending run of `flow`. a plan (`supersede`) takes the place of the
66
+ flow's waiting runs, each passed to `retire` first. call inside a transaction."""
67
+ if supersede:
68
+ for run in model.objects.select_for_update().filter(flow=flow, status__in=SUPERSEDABLE):
69
+ retire(run)
70
+ move(run, S.SUPERSEDED, finished_at=timezone.now())
71
+ try:
72
+ with transaction.atomic():
73
+ return model.objects.create(flow=flow, kind=kind, requested_by=user)
74
+ except IntegrityError:
75
+ raise ActionError("This flow already has a run in progress.") from None
76
+
77
+
78
+ def _approve_perm(run) -> str:
79
+ return f"{run._meta.app_label}.approve_run"
80
+
81
+
82
+ def refusal(run, user, distinct: bool, ttl_hours: int | None = None) -> str | None:
83
+ """why `user` may not approve or reject `run`, or None if they may."""
84
+ if not user.has_perm(_approve_perm(run), run):
85
+ return "You may not approve or reject this run."
86
+ if distinct and run.requested_by_id == user.pk:
87
+ return "Someone other than the requester approves this run."
88
+ if expired(run, ttl_hours):
89
+ return "This plan is too old to approve. Plan again."
90
+ return None
91
+
92
+
93
+ def check_approver(run, user, distinct: bool) -> None:
94
+ reason = refusal(run, user, distinct)
95
+ if reason:
96
+ raise PermissionDenied(reason)
97
+
98
+
99
+ def approve(run, user, ttl_hours: int | None) -> bool:
100
+ """move a locked, waiting run to approved, or to expired when its plan is too
101
+ old. returns whether it was approved; the caller queues the apply."""
102
+ if expired(run, ttl_hours):
103
+ move(run, S.EXPIRED, finished_at=timezone.now())
104
+ return False
105
+ move(
106
+ run,
107
+ S.APPROVED,
108
+ decided_by=user,
109
+ decided_at=timezone.now(),
110
+ approved_sha256=run.plan_sha256,
111
+ )
112
+ return True
113
+
114
+
115
+ def reject(run, user):
116
+ now = timezone.now()
117
+ return move(run, S.REJECTED, decided_by=user, decided_at=now, finished_at=now)
118
+
119
+
120
+ def check_resumable(run, user) -> None:
121
+ if not user.has_perm(_approve_perm(run), run):
122
+ raise PermissionDenied("You may not resume this run.")
123
+ if run.status != S.APPLY_FAILED.value:
124
+ raise ActionError("Only a failed apply can be resumed.")
125
+
126
+
127
+ # jobs
128
+
129
+
130
+ def transition(run, new: RunStatus | str, **fields: Any) -> None:
131
+ with transaction.atomic():
132
+ move(locked(run), new, **fields)
133
+ run.refresh_from_db()
134
+
135
+
136
+ def settle(run, outcome: Outcome, **fields: Any) -> None:
137
+ """store what a pipeline step decided on the run."""
138
+ if outcome.finished:
139
+ fields["finished_at"] = timezone.now()
140
+ transition(run, outcome.status, **outcome.fields, **fields)
141
+
142
+
143
+ def fail(run, message: str, output: str = "") -> None:
144
+ """record a failure on the run, in the status its phase fails into."""
145
+ fields: dict[str, Any] = {"error": message}
146
+ if output:
147
+ fields["output"] = output
148
+ with transaction.atomic():
149
+ run = locked(run)
150
+ new = failure_status(run.status)
151
+ if new is S.FAILED:
152
+ fields["finished_at"] = timezone.now()
153
+ if new is not None:
154
+ move(run, new, **fields)
155
+
156
+
157
+ def run_guarded(run, execute: Callable, logger, known: Iterable[type[Exception]] = ()) -> None:
158
+ """run a job's body. a failure it expects becomes the run's status; anything
159
+ else does too, and is raised again so the host marks its job as errored."""
160
+ try:
161
+ execute(run)
162
+ except RunFailed as e:
163
+ logger.error(str(e))
164
+ fail(run, str(e), e.output)
165
+ except ActionError as e:
166
+ # the run is not in a state this job can act on; leave it as it is.
167
+ logger.error(" ".join(e.messages))
168
+ except tuple(known) as e:
169
+ logger.error(str(e))
170
+ fail(run, str(e))
171
+ except Exception as e:
172
+ fail(run, f"{type(e).__name__}: {e}")
173
+ raise
174
+
175
+
176
+ def start_apply(run) -> tuple[Plan, bool]:
177
+ """take the run into applying: one apply per target at a time, from an
178
+ approved run or a failed apply, and only with the plan that was approved.
179
+ returns the plan and whether this resumes a failed apply."""
180
+ model = type(run)
181
+ with transaction.atomic():
182
+ current = locked(run)
183
+ busy = (
184
+ model.objects.filter(flow__target=run.flow.target, status=S.APPLYING.value)
185
+ .exclude(pk=run.pk)
186
+ .select_for_update()
187
+ )
188
+ if busy.exists():
189
+ raise RunFailed(f"Another run is applying to {run.flow.target}; try again later.")
190
+ if current.status not in (S.APPROVED.value, S.APPLY_FAILED.value):
191
+ raise ActionError(f"a {current.status} run is not applied")
192
+ resuming = current.status == S.APPLY_FAILED.value
193
+ # checked before the move, so a plan that fails it never becomes resumable.
194
+ plan = approved_plan(current.plan, current.approved_sha256)
195
+ move(current, S.APPLYING, apply_started_at=current.apply_started_at or timezone.now())
196
+ run.refresh_from_db()
197
+ return plan, resuming
@@ -0,0 +1,224 @@
1
+ """what a plan job and an apply job do, apart from where the host keeps the run.
2
+
3
+ the host snapshots the flow's files, locks and moves the run, and stores the
4
+ fields these functions return; everything alembic-shaped happens here."""
5
+
6
+ from __future__ import annotations
7
+
8
+ import hashlib
9
+ import json
10
+ import logging
11
+ from collections.abc import Iterable
12
+ from dataclasses import dataclass, field
13
+ from pathlib import Path
14
+ from typing import Any
15
+
16
+ from .plan import Plan
17
+ from .runner import Completed, Flow, Runner
18
+ from .status import RunStatus
19
+
20
+ S = RunStatus
21
+
22
+ SNAPSHOT_LIMIT = 20 << 20
23
+
24
+ # what a run keeps when an import or a map produced its inventory.
25
+ DERIVED_ENTRY = "inventory.json"
26
+
27
+
28
+ class RunFailed(Exception):
29
+ """a run that cannot go on, with the alembic output that led there."""
30
+
31
+ def __init__(self, message: str, output: str = ""):
32
+ super().__init__(message)
33
+ self.output = output
34
+
35
+
36
+ def failed(outcome: Any, output: str) -> RunFailed:
37
+ return RunFailed(f"{outcome.error}\n{outcome.completed.stderr}".strip(), output)
38
+
39
+
40
+ def output_text(*completed: Completed, before: str = "") -> str:
41
+ """alembic's output, one block per command, after what earlier jobs recorded."""
42
+ blocks = [before.rstrip()] if before else []
43
+ for c in completed:
44
+ mode = next((f" {flag}" for flag in ("--dry-run", "--report") if flag in c.argv), "")
45
+ head = f"$ alembic {c.argv[1]}{mode}"
46
+ if mode == " --dry-run":
47
+ head += " (stale check)"
48
+ body = f"{c.stdout}{c.stderr}".rstrip()
49
+ blocks.append(f"{head}\n{body}" if body else head)
50
+ return "\n\n".join(blocks)
51
+
52
+
53
+ def input_sha256(files: dict[str, str]) -> str:
54
+ return hashlib.sha256(json.dumps(files, sort_keys=True).encode()).hexdigest()
55
+
56
+
57
+ def snapshot(
58
+ items: Iterable[tuple[str, bytes]],
59
+ entry: str,
60
+ where: str,
61
+ logger: logging.Logger,
62
+ ) -> dict[str, str]:
63
+ """a flow's files, from `(path under the root, bytes)` pairs. binary files are
64
+ skipped, the total is capped, and `entry` must be among them."""
65
+ files, total = {}, 0
66
+ for path, data in items:
67
+ total += len(data)
68
+ if total > SNAPSHOT_LIMIT:
69
+ raise RunFailed(f"The files in {where} exceed {SNAPSHOT_LIMIT} bytes.")
70
+ try:
71
+ files[path] = data.decode("utf-8")
72
+ except UnicodeDecodeError:
73
+ logger.info(f"skipping {path}: not utf-8 text")
74
+ if entry not in files:
75
+ raise RunFailed(f"{entry} is not in {where}.")
76
+ return files
77
+
78
+
79
+ def derive(
80
+ runner: Runner,
81
+ flow: Flow,
82
+ run: str,
83
+ files: dict[str, str],
84
+ entry: str,
85
+ map_spec: str,
86
+ where: str,
87
+ logger: logging.Logger,
88
+ ) -> tuple[dict[str, str], str, list[Completed]]:
89
+ """import from the flow's source and apply its map spec, when it has them.
90
+ returns the files to plan from, the entry among them, and what ran."""
91
+ inventory = runner.workspace(flow).write_input(run, files, entry)
92
+ completed: list[Completed] = []
93
+ current = inventory
94
+ if flow.source is not None:
95
+ logger.info("importing from the flow's source")
96
+ imported = runner.import_(flow, run, inventory)
97
+ completed.append(imported.completed)
98
+ if imported.error:
99
+ raise failed(imported, output_text(*completed))
100
+ current = imported.inventory
101
+ if map_spec:
102
+ if map_spec not in files:
103
+ raise RunFailed(f"map spec {map_spec} is not in {where}")
104
+ spec = runner.workspace(flow).input_dir(run) / map_spec
105
+ mapped = runner.map(flow, run, current, spec)
106
+ completed.append(mapped.completed)
107
+ if mapped.error:
108
+ raise failed(mapped, output_text(*completed))
109
+ current = mapped.inventory
110
+ if current == inventory:
111
+ return files, entry, completed
112
+ # the plan, the stale check and the apply all work from the derived
113
+ # inventory, so that is what the run keeps.
114
+ return {DERIVED_ENTRY: Path(current).read_text(encoding="utf-8")}, DERIVED_ENTRY, completed
115
+
116
+
117
+ @dataclass(frozen=True)
118
+ class Outcome:
119
+ """where a job leaves the run, and what the host stores on it."""
120
+
121
+ status: RunStatus
122
+ fields: dict[str, Any] = field(default_factory=dict)
123
+ finished: bool = True
124
+
125
+
126
+ def plan_run(
127
+ runner: Runner,
128
+ flow: Flow,
129
+ run: str,
130
+ files: dict[str, str],
131
+ entry: str,
132
+ map_spec: str,
133
+ where: str,
134
+ drift: bool,
135
+ logger: logging.Logger,
136
+ ) -> Outcome:
137
+ """plan (or, with `drift`, report drift) from the flow's snapshot."""
138
+ files, entry, derived = derive(runner, flow, run, files, entry, map_spec, where, logger)
139
+ inventory = runner.workspace(flow).write_input(run, files, entry)
140
+ common = {"input": files, "input_entry": entry, "input_sha256": input_sha256(files)}
141
+
142
+ if drift:
143
+ outcome = runner.drift(flow, run, inventory)
144
+ output = output_text(*derived, outcome.completed)
145
+ if outcome.error:
146
+ raise failed(outcome, output)
147
+ report = outcome.report
148
+ logger.info(f"drift: {report.counts()}")
149
+ return Outcome(
150
+ S.DRIFTED if report.has_drift else S.NO_CHANGES,
151
+ {"drift_report": report.doc, "summary": report.counts(), "output": output, **common},
152
+ )
153
+
154
+ outcome = runner.plan(flow, run, inventory)
155
+ output = output_text(*derived, outcome.completed)
156
+ if outcome.error:
157
+ raise failed(outcome, output)
158
+ plan = outcome.plan
159
+ logger.info(f"plan: {plan.summary()}")
160
+ fields = {
161
+ "plan": plan.raw.decode("utf-8"),
162
+ "plan_sha256": plan.sha256,
163
+ "summary": {**plan.summary(), "schema_preview": plan.schema_preview},
164
+ "output": output,
165
+ **common,
166
+ }
167
+ if plan.is_empty:
168
+ return Outcome(S.NO_CHANGES, fields)
169
+ return Outcome(S.AWAITING_APPROVAL, fields, finished=False)
170
+
171
+
172
+ def approved_plan(plan_text: str, approved_sha256: str) -> Plan:
173
+ """the stored plan, refused unless it is the one that was approved."""
174
+ plan = Plan.from_bytes(plan_text.encode("utf-8"))
175
+ if not approved_sha256 or plan.sha256 != approved_sha256:
176
+ raise RunFailed("The stored plan is not the one that was approved.")
177
+ return plan
178
+
179
+
180
+ def apply_run(
181
+ runner: Runner,
182
+ flow: Flow,
183
+ run: str,
184
+ files: dict[str, str],
185
+ entry: str,
186
+ plan: Plan,
187
+ approved_sha256: str,
188
+ resuming: bool,
189
+ before: str,
190
+ logger: logging.Logger,
191
+ ) -> Outcome:
192
+ """check the approved plan against the target, then apply it."""
193
+ inventory = runner.workspace(flow).write_input(run, files, entry)
194
+ outputs: list[Completed] = []
195
+ if not resuming:
196
+ # a resumed apply re-runs its plan on top of what already landed, so a
197
+ # re-plan would differ by construction. the journal carries it instead.
198
+ check = runner.check_stale(flow, run, inventory, plan)
199
+ outputs.append(check.completed)
200
+ if check.error:
201
+ raise failed(check, output_text(*outputs, before=before))
202
+ logger.info(f"stale check: the target {'changed' if check.stale else 'is unchanged'}")
203
+ if check.stale:
204
+ return Outcome(
205
+ S.STALE,
206
+ {
207
+ "output": output_text(*outputs, before=before),
208
+ "error": "The target changed since this plan was made; plan again.",
209
+ },
210
+ )
211
+
212
+ outcome = runner.apply(flow, run, plan, approved_sha256)
213
+ outputs.append(outcome.completed)
214
+ if outcome.error:
215
+ raise failed(outcome, output_text(*outputs, before=before))
216
+ logger.info(f"applied {outcome.report.applied_count} operations")
217
+ return Outcome(
218
+ S.APPLIED,
219
+ {
220
+ "apply_report": outcome.report.doc,
221
+ "output": output_text(*outputs, before=before),
222
+ "error": "",
223
+ },
224
+ )