pgrecon 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
pgrecon/__init__.py ADDED
@@ -0,0 +1,5 @@
1
+ """pgrecon: migration reconnaissance for PostgreSQL."""
2
+
3
+ from importlib.metadata import version
4
+
5
+ __version__ = version("pgrecon")
pgrecon/cli.py ADDED
@@ -0,0 +1,327 @@
1
+ """Command-line interface."""
2
+
3
+ import json
4
+ import logging
5
+ import sqlite3
6
+ import sys
7
+ import textwrap
8
+ from collections import Counter
9
+ from dataclasses import asdict
10
+ from pathlib import Path
11
+ from typing import Annotated
12
+
13
+ import typer
14
+
15
+ from pgrecon import __version__
16
+ from pgrecon.effort import estimate as build_estimate
17
+ from pgrecon.effort import person_months
18
+ from pgrecon.extract import needs_legacy, script_text
19
+ from pgrecon.inventory import load_dump
20
+ from pgrecon.rules import Rule, all_rules
21
+ from pgrecon.rules.engine import run_rules, summarize
22
+
23
+ # sqlglot logs a warning whenever exotic syntax makes it fall back to an
24
+ # opaque statement; the outcome is already recorded in the inventory, so
25
+ # the console noise helps nobody.
26
+ logging.getLogger("sqlglot").setLevel(logging.ERROR)
27
+
28
+ app = typer.Typer(
29
+ help="Migration reconnaissance for PostgreSQL.",
30
+ no_args_is_help=True,
31
+ add_completion=False,
32
+ )
33
+
34
+
35
+ def _show_version(value: bool) -> None:
36
+ if value:
37
+ typer.echo(f"pgrecon {__version__}")
38
+ raise typer.Exit()
39
+
40
+
41
+ @app.callback()
42
+ def main(
43
+ version: Annotated[
44
+ bool,
45
+ typer.Option(
46
+ "--version",
47
+ callback=_show_version,
48
+ is_eager=True,
49
+ help="Print the version and exit.",
50
+ ),
51
+ ] = False,
52
+ verbose: Annotated[
53
+ int,
54
+ typer.Option(
55
+ "--verbose",
56
+ "-v",
57
+ count=True,
58
+ help="Log progress to stderr; repeat for debug detail.",
59
+ ),
60
+ ] = 0,
61
+ ) -> None:
62
+ # Logs go to stderr and never carry PL/SQL body text, so stdout
63
+ # stays clean for reports and the log stays safe to share.
64
+ logging.basicConfig(
65
+ level=logging.WARNING - 10 * min(verbose, 2),
66
+ stream=sys.stderr,
67
+ format="%(levelname)s: %(message)s",
68
+ )
69
+
70
+
71
+ @app.command()
72
+ def script(
73
+ out: Annotated[
74
+ Path, typer.Option(help="Where to write the extraction script.")
75
+ ] = Path("pgrecon_extract.sql"),
76
+ source_version: Annotated[
77
+ str | None,
78
+ typer.Option(
79
+ "--source-version",
80
+ help="Oracle version of the source, e.g. 9.2, 11.2, 19."
81
+ " Picks the right script variant.",
82
+ ),
83
+ ] = None,
84
+ legacy: Annotated[
85
+ bool,
86
+ typer.Option(
87
+ "--legacy",
88
+ help="Force the variant for Oracle 9.2-11.1 or old clients.",
89
+ ),
90
+ ] = False,
91
+ ) -> None:
92
+ """Write the offline SQL*Plus extraction script for the client DBA."""
93
+ if source_version is not None:
94
+ try:
95
+ legacy = legacy or needs_legacy(source_version)
96
+ except ValueError as exc:
97
+ raise typer.BadParameter(str(exc)) from exc
98
+ if legacy and out.name == "pgrecon_extract.sql":
99
+ out = out.with_name("pgrecon_extract_legacy.sql")
100
+ out.write_text(script_text(legacy), encoding="ascii", newline="\n")
101
+ typer.echo(f"Wrote {out}" + (" (legacy variant)" if legacy else ""))
102
+ typer.echo("Run it as: sqlplus readonly_user@service @" + out.name + " SCHEMA")
103
+
104
+
105
+ @app.command()
106
+ def load(
107
+ dump_dir: Annotated[
108
+ Path,
109
+ typer.Argument(exists=True, file_okay=False, help="Extraction dump folder."),
110
+ ],
111
+ db: Annotated[Path, typer.Option(help="Inventory database to create.")] = Path(
112
+ "inventory.db"
113
+ ),
114
+ encoding: Annotated[
115
+ str,
116
+ typer.Option(
117
+ help="Encoding of the dump files, e.g. cp949 for a Korean"
118
+ " Windows client. Defaults to UTF-8."
119
+ ),
120
+ ] = "utf-8-sig",
121
+ ) -> None:
122
+ """Load an extraction dump into a local SQLite inventory."""
123
+ counts = load_dump(dump_dir, db, encoding)
124
+ width = max(len(name) for name in counts)
125
+ for name, count in sorted(counts.items()):
126
+ typer.echo(f"{name:<{width}} {count}")
127
+ typer.echo(f"Inventory written to {db}")
128
+
129
+
130
+ @app.command()
131
+ def report(
132
+ db: Annotated[
133
+ Path,
134
+ typer.Option(exists=True, dir_okay=False, help="Inventory database."),
135
+ ] = Path("inventory.db"),
136
+ fmt: Annotated[
137
+ str, typer.Option("--format", help="Output format: text or json.")
138
+ ] = "text",
139
+ remedies: Annotated[
140
+ bool,
141
+ typer.Option(
142
+ "--remedies",
143
+ help="Append each fired rule's remedy to the text report.",
144
+ ),
145
+ ] = False,
146
+ ) -> None:
147
+ """Run the assessment rules and print the findings."""
148
+ findings = run_rules(db)
149
+ summary = summarize(findings)
150
+ by_id = {rule.id: rule for rule in all_rules()}
151
+ fired = {f.rule_id: by_id[f.rule_id] for f in findings}
152
+
153
+ if fmt == "json":
154
+ payload = {
155
+ "summary": summary,
156
+ "findings": [asdict(f) for f in findings],
157
+ "rules": {
158
+ rule_id: _rule_meta(rule) for rule_id, rule in sorted(fired.items())
159
+ },
160
+ }
161
+ typer.echo(json.dumps(payload, indent=2))
162
+ return
163
+ if fmt != "text":
164
+ raise typer.BadParameter("format must be text or json")
165
+
166
+ if not findings:
167
+ typer.echo("No findings.")
168
+ return
169
+ widths = (
170
+ max(len(f.severity.value) for f in findings),
171
+ max(len(f.rule_id) for f in findings),
172
+ max(len(f.name) for f in findings),
173
+ )
174
+ for f in findings:
175
+ typer.echo(
176
+ f"{f.severity.value:<{widths[0]}} {f.rule_id:<{widths[1]}}"
177
+ f" {f.name:<{widths[2]}} {f.detail}"
178
+ )
179
+ typer.echo("")
180
+ parts = [f"{n} {sev}" for sev, n in summary["by_severity"].items()]
181
+ typer.echo(
182
+ f"{summary['findings']} findings ({', '.join(parts)});"
183
+ f" effort points {summary['effort_points']}"
184
+ )
185
+
186
+ if remedies:
187
+ counts = Counter(f.rule_id for f in findings)
188
+ typer.echo("")
189
+ typer.echo("Remedies:")
190
+ for rule_id in dict.fromkeys(f.rule_id for f in findings):
191
+ rule = fired[rule_id]
192
+ n = counts[rule_id]
193
+ tags = [rule.severity.value, f"{n} finding" + ("s" if n != 1 else "")]
194
+ if rule.extension:
195
+ tags.append(f"extension {rule.extension}")
196
+ typer.echo("")
197
+ typer.echo(f"{rule.id} {rule.title} [{', '.join(tags)}]")
198
+ typer.echo(_wrap(rule.remedy))
199
+
200
+
201
+ def _rule_meta(rule: Rule) -> dict[str, object]:
202
+ return {
203
+ "title": rule.title,
204
+ "category": rule.category,
205
+ "severity": rule.severity.value,
206
+ "effort": rule.effort,
207
+ "remedy": rule.remedy,
208
+ "extension": rule.extension,
209
+ }
210
+
211
+
212
+ def _wrap(text: str) -> str:
213
+ return textwrap.fill(text, width=76, initial_indent=" ", subsequent_indent=" ")
214
+
215
+
216
+ @app.command()
217
+ def estimate(
218
+ db: Annotated[
219
+ Path,
220
+ typer.Option(exists=True, dir_okay=False, help="Inventory database."),
221
+ ] = Path("inventory.db"),
222
+ fmt: Annotated[
223
+ str, typer.Option("--format", help="Output format: text or json.")
224
+ ] = "text",
225
+ ) -> None:
226
+ """Estimate migration effort in person-days, as a range."""
227
+ findings = run_rules(db)
228
+ result = build_estimate(db, findings)
229
+
230
+ if fmt == "json":
231
+ payload = {
232
+ "components": {c.label: c.person_days for c in result.components},
233
+ "development_person_days": result.development,
234
+ "person_days": {
235
+ "low": result.low,
236
+ "expected": result.expected,
237
+ "high": result.high,
238
+ },
239
+ "person_months": {
240
+ "low": person_months(result.low),
241
+ "expected": person_months(result.expected),
242
+ "high": person_months(result.high),
243
+ },
244
+ "assumptions": list(result.assumptions),
245
+ }
246
+ typer.echo(json.dumps(payload, indent=2))
247
+ return
248
+ if fmt != "text":
249
+ raise typer.BadParameter("format must be text or json")
250
+
251
+ typer.echo("Migration effort estimate (person-days)")
252
+ typer.echo("")
253
+ width = max(len(c.label) for c in result.components)
254
+ for c in result.components:
255
+ typer.echo(f" {c.label:<{width}} {c.person_days:8.1f}")
256
+ typer.echo(f" {'development subtotal':<{width}} {result.development:8.1f}")
257
+ typer.echo("")
258
+ typer.echo("With testing and stabilization:")
259
+ typer.echo(
260
+ f" low {result.low:.0f}, expected {result.expected:.0f},"
261
+ f" high {result.high:.0f} person-days"
262
+ f" ({person_months(result.low):.1f} to"
263
+ f" {person_months(result.high):.1f} person-months)"
264
+ )
265
+ typer.echo("")
266
+ typer.echo("Assumptions:")
267
+ for line in result.assumptions:
268
+ typer.echo(_wrap("- " + line))
269
+
270
+
271
+ @app.command()
272
+ def explain(
273
+ rule_id: Annotated[
274
+ str | None,
275
+ typer.Argument(help="Rule id, e.g. R-SRC-18. Omit to list the catalog."),
276
+ ] = None,
277
+ ) -> None:
278
+ """Show one rule's remedy, or list the whole catalog."""
279
+ rules = all_rules()
280
+ if rule_id is None:
281
+ width = max(len(rule.id) for rule in rules)
282
+ for rule in sorted(rules, key=lambda r: r.id):
283
+ typer.echo(f"{rule.id:<{width}} {rule.severity.value:<7} {rule.title}")
284
+ typer.echo("")
285
+ typer.echo(f"{len(rules)} rules")
286
+ return
287
+ match = {rule.id: rule for rule in rules}.get(rule_id.upper())
288
+ if match is None:
289
+ raise typer.BadParameter(f"unknown rule id: {rule_id}")
290
+ typer.echo(f"{match.id}: {match.title}")
291
+ line = (
292
+ f"severity: {match.severity.value} effort: {match.effort}"
293
+ f" category: {match.category}"
294
+ )
295
+ if match.extension:
296
+ line += f" extension: {match.extension}"
297
+ typer.echo(line)
298
+ typer.echo("")
299
+ typer.echo(_wrap(match.remedy))
300
+
301
+
302
+ @app.command()
303
+ def info(
304
+ db: Annotated[
305
+ Path,
306
+ typer.Option(exists=True, dir_okay=False, help="Inventory database."),
307
+ ] = Path("inventory.db"),
308
+ ) -> None:
309
+ """Summarize an inventory database."""
310
+ conn = sqlite3.connect(db)
311
+ try:
312
+ for key, value in conn.execute("SELECT key, value FROM meta ORDER BY key"):
313
+ typer.echo(f"{key}: {value}")
314
+ rows = conn.execute(
315
+ "SELECT type, COUNT(*) FROM objects GROUP BY type ORDER BY type"
316
+ ).fetchall()
317
+ if rows:
318
+ typer.echo("objects:")
319
+ for obj_type, count in rows:
320
+ typer.echo(f" {obj_type}: {count}")
321
+ failed = conn.execute("SELECT COUNT(*) FROM ddl WHERE parse_ok = 0").fetchone()[
322
+ 0
323
+ ]
324
+ total = conn.execute("SELECT COUNT(*) FROM ddl").fetchone()[0]
325
+ typer.echo(f"ddl parsed: {total - failed}/{total}")
326
+ finally:
327
+ conn.close()
pgrecon/effort.py ADDED
@@ -0,0 +1,161 @@
1
+ """Turn findings and inventory scale into a person-day estimate.
2
+
3
+ The estimate is built from named components so it can be argued with
4
+ line by line, which is the point: an opaque total invites either blind
5
+ trust or blind dismissal. Development effort is the sum of a baseline,
6
+ mechanical schema conversion, per-finding remediation, PL/SQL porting
7
+ by volume, and data movement; testing and stabilization is applied as
8
+ a factor range on top, because in field reports it rivals development
9
+ and nobody budgets it.
10
+
11
+ Every rate here is a default calibration from published field
12
+ experience and ora2pg-era rules of thumb. They are deliberately
13
+ visible and deliberately conservative; real engagements calibrate
14
+ them against the client's team and workload.
15
+ """
16
+
17
+ import sqlite3
18
+ from dataclasses import dataclass
19
+ from pathlib import Path
20
+
21
+ from pgrecon.rules import Finding, Rule, Severity, all_rules
22
+
23
+ # Person-days. The baseline covers environments, tooling, data-movement
24
+ # setup, and cutover rehearsal scaffolding that exist even for a clean
25
+ # schema.
26
+ BASELINE_PD = 5.0
27
+ PER_TABLE_PD = 0.2
28
+ PER_INDEX_PD = 0.05
29
+
30
+ # Mechanical conversion rates for objects that mostly translate.
31
+ CONVERT_TABLE_PD = 0.05
32
+ CONVERT_COLUMN_PD = 0.02
33
+ CONVERT_VIEW_PD = 0.1
34
+ CONVERT_SEQUENCE_PD = 0.02
35
+
36
+ # Porting rate for stored code, including its unit tests. Findings
37
+ # price the hotspots; this prices the bulk that merely needs careful
38
+ # transcription.
39
+ PLSQL_LINES_PER_DAY = 200.0
40
+
41
+ # Data movement is mostly machine time; the human share scales weakly
42
+ # with volume.
43
+ BYTES_PER_PD = 50e9
44
+
45
+ # A rule firing n times does not cost n times the first fix: the
46
+ # pattern is learned once. The marginal share of each further finding
47
+ # depends on how mechanical the fix is, which severity approximates.
48
+ MARGINAL = {
49
+ Severity.INFO: 0.05,
50
+ Severity.LOW: 0.2,
51
+ Severity.MEDIUM: 0.5,
52
+ Severity.HIGH: 0.7,
53
+ Severity.BLOCKER: 1.0,
54
+ }
55
+
56
+ # Testing and stabilization as a share of development. Field reports
57
+ # put it between a third and everything-again-and-more.
58
+ TESTING_LOW = 1.3
59
+ TESTING_EXPECTED = 1.6
60
+ TESTING_HIGH = 2.2
61
+
62
+ WORKDAYS_PER_MONTH = 21.0
63
+
64
+
65
+ @dataclass(frozen=True)
66
+ class Component:
67
+ label: str
68
+ person_days: float
69
+
70
+
71
+ @dataclass(frozen=True)
72
+ class Estimate:
73
+ components: tuple[Component, ...]
74
+ development: float
75
+ low: float
76
+ expected: float
77
+ high: float
78
+ assumptions: tuple[str, ...]
79
+
80
+
81
+ def _scalar(conn: sqlite3.Connection, query: str) -> float:
82
+ value = conn.execute(query).fetchone()[0]
83
+ return float(value or 0)
84
+
85
+
86
+ def _remediation(findings: list[Finding], rules: list[Rule]) -> float:
87
+ by_id = {rule.id: rule for rule in rules}
88
+ counts: dict[str, int] = {}
89
+ for finding in findings:
90
+ counts[finding.rule_id] = counts.get(finding.rule_id, 0) + 1
91
+ total = 0.0
92
+ for rule_id, n in counts.items():
93
+ rule = by_id.get(rule_id)
94
+ if rule is None:
95
+ continue
96
+ marginal = MARGINAL[rule.severity]
97
+ total += rule.effort * (1 + marginal * (n - 1))
98
+ return total
99
+
100
+
101
+ def estimate(
102
+ db_path: Path, findings: list[Finding], rules: list[Rule] | None = None
103
+ ) -> Estimate:
104
+ if rules is None:
105
+ rules = all_rules()
106
+ conn = sqlite3.connect(db_path)
107
+ try:
108
+ tables = _scalar(conn, "SELECT COUNT(*) FROM tables")
109
+ columns = _scalar(conn, "SELECT COUNT(*) FROM columns")
110
+ indexes = _scalar(conn, "SELECT COUNT(*) FROM indexes")
111
+ views = _scalar(conn, "SELECT COUNT(*) FROM objects WHERE type = 'VIEW'")
112
+ sequences = _scalar(
113
+ conn, "SELECT COUNT(*) FROM objects WHERE type = 'SEQUENCE'"
114
+ )
115
+ source_lines = _scalar(conn, "SELECT COUNT(*) FROM source")
116
+ data_bytes = _scalar(
117
+ conn,
118
+ "SELECT COALESCE(SUM(num_rows * avg_row_len), 0) FROM tables",
119
+ )
120
+ finally:
121
+ conn.close()
122
+
123
+ components = (
124
+ Component(
125
+ "baseline and environment",
126
+ BASELINE_PD + PER_TABLE_PD * tables + PER_INDEX_PD * indexes,
127
+ ),
128
+ Component(
129
+ "schema conversion",
130
+ CONVERT_TABLE_PD * tables
131
+ + CONVERT_COLUMN_PD * columns
132
+ + CONVERT_VIEW_PD * views
133
+ + CONVERT_SEQUENCE_PD * sequences,
134
+ ),
135
+ Component("finding remediation", _remediation(findings, rules)),
136
+ Component("PL/SQL porting by volume", source_lines / PLSQL_LINES_PER_DAY),
137
+ Component("data movement", data_bytes / BYTES_PER_PD),
138
+ )
139
+ development = sum(c.person_days for c in components)
140
+ assumptions = (
141
+ f"stored code ports at {PLSQL_LINES_PER_DAY:.0f} lines per person-day"
142
+ " including unit tests",
143
+ "repeated findings of one rule cost a severity-dependent fraction"
144
+ " of the first fix",
145
+ f"testing and stabilization runs {TESTING_LOW:.1f}x to"
146
+ f" {TESTING_HIGH:.1f}x development, {TESTING_EXPECTED:.1f}x expected",
147
+ "static estimate from schema facts alone: workload, data quality,"
148
+ " and application coupling are not visible here",
149
+ )
150
+ return Estimate(
151
+ components=components,
152
+ development=round(development, 1),
153
+ low=round(development * TESTING_LOW, 1),
154
+ expected=round(development * TESTING_EXPECTED, 1),
155
+ high=round(development * TESTING_HIGH, 1),
156
+ assumptions=assumptions,
157
+ )
158
+
159
+
160
+ def person_months(person_days: float) -> float:
161
+ return round(person_days / WORKDAYS_PER_MONTH, 1)
@@ -0,0 +1,32 @@
1
+ """Offline extraction: the reviewable SQL*Plus scripts and their helpers."""
2
+
3
+ from importlib import resources
4
+
5
+
6
+ def script_text(legacy: bool = False) -> str:
7
+ """Return the offline extraction script shipped with the package.
8
+
9
+ The standard script needs a SQL*Plus 12.2+ client and targets Oracle
10
+ 11.2 or newer. The legacy variant runs on any old on-box SQL*Plus
11
+ against Oracle 9.2 through 11.1: hand-built CSV, no DBMS_METADATA.
12
+ """
13
+ name = "pgrecon_extract_legacy.sql" if legacy else "pgrecon_extract.sql"
14
+ path = resources.files("pgrecon.extract").joinpath(name)
15
+ return path.read_text(encoding="ascii")
16
+
17
+
18
+ def needs_legacy(source_version: str) -> bool:
19
+ """Decide the script variant from an Oracle version like 9.2 or 19.
20
+
21
+ Anything below 11.2 takes the legacy variant. Raises ValueError for
22
+ input that does not look like an Oracle version.
23
+ """
24
+ parts = source_version.strip().split(".")
25
+ try:
26
+ major = int(parts[0])
27
+ minor = int(parts[1]) if len(parts) > 1 else 0
28
+ except (ValueError, IndexError) as exc:
29
+ raise ValueError(f"not an Oracle version: {source_version!r}") from exc
30
+ if not 8 <= major <= 30:
31
+ raise ValueError(f"not a supported Oracle version: {source_version!r}")
32
+ return (major, minor) < (11, 2)