ora2pg-gap-report 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- ora2pg_gap_report/__init__.py +0 -0
- ora2pg_gap_report/cli.py +188 -0
- ora2pg_gap_report/detectors/__init__.py +0 -0
- ora2pg_gap_report/detectors/autonomous_tx.py +94 -0
- ora2pg_gap_report/detectors/compound_triggers.py +68 -0
- ora2pg_gap_report/detectors/connect_by.py +110 -0
- ora2pg_gap_report/detectors/dbms_utl_calls.py +67 -0
- ora2pg_gap_report/effort_estimator.py +54 -0
- ora2pg_gap_report/models.py +12 -0
- ora2pg_gap_report/ora2pg_wrapper.py +140 -0
- ora2pg_gap_report/oracle_connector.py +166 -0
- ora2pg_gap_report/oracle_export.py +82 -0
- ora2pg_gap_report/plsql_lex.py +213 -0
- ora2pg_gap_report/report_generator.py +26 -0
- ora2pg_gap_report/terminal_report.py +106 -0
- ora2pg_gap_report-0.1.0.dist-info/METADATA +261 -0
- ora2pg_gap_report-0.1.0.dist-info/RECORD +21 -0
- ora2pg_gap_report-0.1.0.dist-info/WHEEL +5 -0
- ora2pg_gap_report-0.1.0.dist-info/entry_points.txt +3 -0
- ora2pg_gap_report-0.1.0.dist-info/licenses/LICENSE +21 -0
- ora2pg_gap_report-0.1.0.dist-info/top_level.txt +1 -0
|
File without changes
|
ora2pg_gap_report/cli.py
ADDED
|
@@ -0,0 +1,188 @@
|
|
|
1
|
+
import argparse
|
|
2
|
+
import dataclasses
|
|
3
|
+
import sys
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
|
|
6
|
+
from rich.console import Console
|
|
7
|
+
from rich.markup import escape
|
|
8
|
+
|
|
9
|
+
from .detectors.autonomous_tx import find_autonomous_transactions
|
|
10
|
+
from .detectors.compound_triggers import find_compound_triggers
|
|
11
|
+
from .detectors.connect_by import find_connect_by_risks, guess_object_type, has_connect_by
|
|
12
|
+
from .detectors.dbms_utl_calls import find_dbms_utl_calls
|
|
13
|
+
from .effort_estimator import estimate_hours, ordered_counts, summarize_by_severity
|
|
14
|
+
from .models import Finding
|
|
15
|
+
from .ora2pg_wrapper import Ora2PgNotFoundError, Ora2PgRunError, run_estimate_cost
|
|
16
|
+
from .report_generator import to_json, to_markdown
|
|
17
|
+
from .terminal_report import render as render_terminal
|
|
18
|
+
|
|
19
|
+
_DETECTORS = (
|
|
20
|
+
find_autonomous_transactions,
|
|
21
|
+
find_compound_triggers,
|
|
22
|
+
find_dbms_utl_calls,
|
|
23
|
+
)
|
|
24
|
+
_SEVERITY_ORDER = {"high": 0, "medium": 1, "low": 2}
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def _sort_findings(findings: list[Finding]) -> None:
|
|
28
|
+
findings.sort(key=lambda f: (_SEVERITY_ORDER.get(f.severity, 99), f.object_name, f.line))
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def scan_source(source: str) -> list[Finding]:
|
|
32
|
+
findings: list[Finding] = []
|
|
33
|
+
for detector in _DETECTORS:
|
|
34
|
+
findings.extend(detector(source))
|
|
35
|
+
_sort_findings(findings)
|
|
36
|
+
return findings
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def _build_arg_parser() -> argparse.ArgumentParser:
|
|
40
|
+
parser = argparse.ArgumentParser(
|
|
41
|
+
prog="ora2pg-gap-report",
|
|
42
|
+
description=(
|
|
43
|
+
"Сканирует выгруженный Oracle DDL (PACKAGE BODY / TRIGGER) и "
|
|
44
|
+
"показывает конкретные объекты, которые ora2pg не перенесёт "
|
|
45
|
+
"корректно, и почему."
|
|
46
|
+
),
|
|
47
|
+
)
|
|
48
|
+
parser.add_argument(
|
|
49
|
+
"paths", nargs="+", type=Path, help="Файлы с DDL для анализа (.sql/.pks/.pkb)"
|
|
50
|
+
)
|
|
51
|
+
parser.add_argument(
|
|
52
|
+
"--format",
|
|
53
|
+
choices=("terminal", "markdown", "json"),
|
|
54
|
+
default=None,
|
|
55
|
+
help=(
|
|
56
|
+
"Формат отчёта. По умолчанию — цветной вывод в терминал, если "
|
|
57
|
+
"stdout это tty и не указан --output; иначе markdown."
|
|
58
|
+
),
|
|
59
|
+
)
|
|
60
|
+
parser.add_argument(
|
|
61
|
+
"--output", type=Path, default=None, help="Куда сохранить отчёт (по умолчанию — stdout)"
|
|
62
|
+
)
|
|
63
|
+
parser.add_argument(
|
|
64
|
+
"--check-connect-by",
|
|
65
|
+
action="store_true",
|
|
66
|
+
help=(
|
|
67
|
+
"Дополнительно: для файлов с CONNECT BY реально прогнать ora2pg и "
|
|
68
|
+
"проверить сгенерированный WITH RECURSIVE на известный баг с LEVEL. "
|
|
69
|
+
"Требует установленный ora2pg (не ставится через pip — это "
|
|
70
|
+
"отдельный Perl-инструмент, см. README)."
|
|
71
|
+
),
|
|
72
|
+
)
|
|
73
|
+
parser.add_argument(
|
|
74
|
+
"--ora2pg-bin",
|
|
75
|
+
default="ora2pg",
|
|
76
|
+
help="Путь к исполняемому файлу ora2pg (по умолчанию ищется в PATH)",
|
|
77
|
+
)
|
|
78
|
+
return parser
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def _connect_by_check(path: Path, source: str, ora2pg_bin: str) -> tuple[list[Finding], str | None]:
|
|
82
|
+
"""Returns (findings, warning) — warning is set instead of raising when
|
|
83
|
+
ora2pg isn't available or fails, since this check is opt-in/best-effort
|
|
84
|
+
by design (see docs/research/step0-show-report-baseline.md section 3:
|
|
85
|
+
low priority for MVP)."""
|
|
86
|
+
if not has_connect_by(source):
|
|
87
|
+
return [], None
|
|
88
|
+
try:
|
|
89
|
+
output = run_estimate_cost(path, guess_object_type(source), ora2pg_bin=ora2pg_bin)
|
|
90
|
+
except Ora2PgNotFoundError:
|
|
91
|
+
return [], f"{path}: содержит CONNECT BY, но ora2pg не найден — проверка пропущена"
|
|
92
|
+
except Ora2PgRunError as exc:
|
|
93
|
+
return [], f"{path}: содержит CONNECT BY, но запуск ora2pg завершился ошибкой ({exc})"
|
|
94
|
+
return [
|
|
95
|
+
dataclasses.replace(f, source_file=str(path)) for f in find_connect_by_risks(output)
|
|
96
|
+
], None
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
def _render(findings: list[Finding], fmt: str) -> str:
|
|
100
|
+
if fmt == "json":
|
|
101
|
+
return to_json(findings)
|
|
102
|
+
|
|
103
|
+
counts = summarize_by_severity(findings)
|
|
104
|
+
counts_text = ", ".join(f"{name}: {n}" for name, n in ordered_counts(counts))
|
|
105
|
+
lo, hi = estimate_hours(findings)
|
|
106
|
+
header = (
|
|
107
|
+
"# Отчёт ora2pg-gap-report\n\n"
|
|
108
|
+
f"Найдено проблемных объектов: {len(findings)} ({counts_text})\n\n"
|
|
109
|
+
f"Грубая оценка ручной доработки: {lo:g}–{hi:g} ч. "
|
|
110
|
+
"— неоткалиброванная эвристика по severity, не измерение "
|
|
111
|
+
"(см. PROJECT_BRIEF.md).\n\n"
|
|
112
|
+
)
|
|
113
|
+
return header + to_markdown(findings)
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
def resolve_format(explicit_format: str | None, output: Path | None, stdout_is_tty: bool) -> str:
|
|
117
|
+
"""Pure resolution logic, kept separate from main() so the
|
|
118
|
+
default-format behaviour is testable without a real terminal."""
|
|
119
|
+
if explicit_format is not None:
|
|
120
|
+
return explicit_format
|
|
121
|
+
return "terminal" if (output is None and stdout_is_tty) else "markdown"
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def main(argv: list[str] | None = None) -> int:
|
|
125
|
+
args = _build_arg_parser().parse_args(argv)
|
|
126
|
+
err_console = Console(stderr=True)
|
|
127
|
+
|
|
128
|
+
fmt = resolve_format(args.format, args.output, sys.stdout.isatty())
|
|
129
|
+
|
|
130
|
+
all_findings: list[Finding] = []
|
|
131
|
+
had_error = False
|
|
132
|
+
for path in args.paths:
|
|
133
|
+
if not path.is_file():
|
|
134
|
+
err_console.print(f"[yellow]Пропущен (не найден):[/yellow] {escape(str(path))}")
|
|
135
|
+
had_error = True
|
|
136
|
+
continue
|
|
137
|
+
try:
|
|
138
|
+
source = path.read_text(errors="replace")
|
|
139
|
+
except OSError as exc:
|
|
140
|
+
err_console.print(
|
|
141
|
+
f"[yellow]Пропущен (не читается: {escape(str(exc))}):[/yellow] {escape(str(path))}"
|
|
142
|
+
)
|
|
143
|
+
had_error = True
|
|
144
|
+
continue
|
|
145
|
+
all_findings.extend(
|
|
146
|
+
dataclasses.replace(f, source_file=str(path)) for f in scan_source(source)
|
|
147
|
+
)
|
|
148
|
+
|
|
149
|
+
if args.check_connect_by:
|
|
150
|
+
findings, warning = _connect_by_check(path, source, args.ora2pg_bin)
|
|
151
|
+
all_findings.extend(findings)
|
|
152
|
+
if warning:
|
|
153
|
+
err_console.print(f"[yellow]{escape(warning)}[/yellow]")
|
|
154
|
+
|
|
155
|
+
_sort_findings(all_findings)
|
|
156
|
+
|
|
157
|
+
if fmt == "terminal":
|
|
158
|
+
if args.output:
|
|
159
|
+
try:
|
|
160
|
+
with args.output.open("w", encoding="utf-8") as fh:
|
|
161
|
+
render_terminal(all_findings, console=Console(file=fh))
|
|
162
|
+
except OSError as exc:
|
|
163
|
+
err_console.print(
|
|
164
|
+
f"[red]Не удалось записать отчёт в {escape(str(args.output))}: "
|
|
165
|
+
f"{escape(str(exc))}[/red]"
|
|
166
|
+
)
|
|
167
|
+
return 2
|
|
168
|
+
else:
|
|
169
|
+
render_terminal(all_findings)
|
|
170
|
+
else:
|
|
171
|
+
report = _render(all_findings, fmt)
|
|
172
|
+
if args.output:
|
|
173
|
+
try:
|
|
174
|
+
args.output.write_text(report, encoding="utf-8")
|
|
175
|
+
except OSError as exc:
|
|
176
|
+
err_console.print(
|
|
177
|
+
f"[red]Не удалось записать отчёт в {escape(str(args.output))}: "
|
|
178
|
+
f"{escape(str(exc))}[/red]"
|
|
179
|
+
)
|
|
180
|
+
return 2
|
|
181
|
+
else:
|
|
182
|
+
print(report)
|
|
183
|
+
|
|
184
|
+
return 2 if had_error else 0
|
|
185
|
+
|
|
186
|
+
|
|
187
|
+
if __name__ == "__main__":
|
|
188
|
+
raise SystemExit(main())
|
|
File without changes
|
|
@@ -0,0 +1,94 @@
|
|
|
1
|
+
import re
|
|
2
|
+
|
|
3
|
+
from ..models import Finding
|
|
4
|
+
from ..plsql_lex import (
|
|
5
|
+
ROUTINE_START_RE,
|
|
6
|
+
declare_and_begin,
|
|
7
|
+
find_matching_end,
|
|
8
|
+
line_at,
|
|
9
|
+
mask_strings_and_comments,
|
|
10
|
+
own_declare_text,
|
|
11
|
+
qualified_name_pattern,
|
|
12
|
+
)
|
|
13
|
+
|
|
14
|
+
_PACKAGE_BODY_NAME_RE = re.compile(
|
|
15
|
+
qualified_name_pattern(r"PACKAGE\s+BODY"),
|
|
16
|
+
re.IGNORECASE,
|
|
17
|
+
)
|
|
18
|
+
_PRAGMA_RE = re.compile(r"PRAGMA\s+AUTONOMOUS_TRANSACTION\s*;", re.IGNORECASE)
|
|
19
|
+
|
|
20
|
+
_MESSAGE = (
|
|
21
|
+
"ora2pg перенесёт эту процедуру/функцию через dblink-обёртку "
|
|
22
|
+
"(переименует в *_atx, уберёт COMMIT из тела, добавит функцию-прокси, "
|
|
23
|
+
"вызывающую её через dblink()). Стратегия рабочая, но не бесшовная: "
|
|
24
|
+
"требуется расширение dblink и ручная настройка connection string — "
|
|
25
|
+
"то есть сетевая зависимость между процедурами, которая может быть "
|
|
26
|
+
"неприемлема в контуре с жёсткими требованиями к изоляции. При этом "
|
|
27
|
+
"SHOW_REPORT и --estimate_cost систематически недооценивают стоимость "
|
|
28
|
+
"этой конструкции именно для функций/процедур внутри PACKAGE BODY — "
|
|
29
|
+
"сама PRAGMA стоит в декларативной секции (до BEGIN), которая не "
|
|
30
|
+
"попадает в подсчёт стоимости (declare/code split в "
|
|
31
|
+
"Ora2Pg.pm::_lookup_function)."
|
|
32
|
+
)
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def _package_name_at(package_matches: list, position: int) -> str:
|
|
36
|
+
name = "UNKNOWN"
|
|
37
|
+
for pm in package_matches:
|
|
38
|
+
if pm.start() > position:
|
|
39
|
+
break
|
|
40
|
+
name = pm.group(1).upper()
|
|
41
|
+
return name
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def find_autonomous_transactions(source: str) -> list[Finding]:
|
|
45
|
+
"""Detect PRAGMA AUTONOMOUS_TRANSACTION inside PACKAGE BODY routines.
|
|
46
|
+
|
|
47
|
+
Handles multiple package bodies in one file, string/comment-aware
|
|
48
|
+
scanning, and correctly excludes locally nested subprograms' own
|
|
49
|
+
declare sections from their enclosing routine's — a nested routine's
|
|
50
|
+
PRAGMA is neither dropped nor misattributed to the outer routine, it is
|
|
51
|
+
simply out of scope (detecting *those* is a separate, smaller gap).
|
|
52
|
+
"""
|
|
53
|
+
clean = mask_strings_and_comments(source)
|
|
54
|
+
package_matches = list(_PACKAGE_BODY_NAME_RE.finditer(clean))
|
|
55
|
+
|
|
56
|
+
findings: list[Finding] = []
|
|
57
|
+
cursor = 0
|
|
58
|
+
hard_boundary = len(clean)
|
|
59
|
+
|
|
60
|
+
for match in ROUTINE_START_RE.finditer(clean):
|
|
61
|
+
if match.start() < cursor:
|
|
62
|
+
continue # nested inside a routine already resolved below
|
|
63
|
+
|
|
64
|
+
resolved = declare_and_begin(clean, match.end(), hard_boundary)
|
|
65
|
+
if resolved is None:
|
|
66
|
+
continue
|
|
67
|
+
declare_start, begin_pos, nested_spans = resolved
|
|
68
|
+
|
|
69
|
+
end_pos = find_matching_end(clean, begin_pos, hard_boundary)
|
|
70
|
+
if end_pos is None:
|
|
71
|
+
continue
|
|
72
|
+
cursor = end_pos
|
|
73
|
+
|
|
74
|
+
declare_text = own_declare_text(clean, declare_start, begin_pos, nested_spans)
|
|
75
|
+
pragma_match = _PRAGMA_RE.search(declare_text)
|
|
76
|
+
if not pragma_match:
|
|
77
|
+
continue
|
|
78
|
+
|
|
79
|
+
absolute_pos = declare_start + pragma_match.start()
|
|
80
|
+
line_no = line_at(clean, absolute_pos)
|
|
81
|
+
package_name = _package_name_at(package_matches, match.start())
|
|
82
|
+
|
|
83
|
+
findings.append(
|
|
84
|
+
Finding(
|
|
85
|
+
detector="autonomous_tx",
|
|
86
|
+
severity="high",
|
|
87
|
+
object_name=f"{package_name}.{match.group(1).upper()}",
|
|
88
|
+
line=line_no,
|
|
89
|
+
snippet=pragma_match.group(0).strip(),
|
|
90
|
+
message=_MESSAGE,
|
|
91
|
+
)
|
|
92
|
+
)
|
|
93
|
+
|
|
94
|
+
return findings
|
|
@@ -0,0 +1,68 @@
|
|
|
1
|
+
import re
|
|
2
|
+
|
|
3
|
+
from ..models import Finding
|
|
4
|
+
from ..plsql_lex import line_at, mask_strings_and_comments, qualified_name_pattern
|
|
5
|
+
|
|
6
|
+
_TRIGGER_START_RE = re.compile(
|
|
7
|
+
qualified_name_pattern(
|
|
8
|
+
r"CREATE\s+(?:OR\s+REPLACE\s+)?(?:EDITIONABLE\s+|NONEDITIONABLE\s+)?TRIGGER"
|
|
9
|
+
),
|
|
10
|
+
re.IGNORECASE,
|
|
11
|
+
)
|
|
12
|
+
_COMPOUND_RE = re.compile(r"\bCOMPOUND\s+TRIGGER\b", re.IGNORECASE)
|
|
13
|
+
|
|
14
|
+
_MESSAGE = (
|
|
15
|
+
"COMPOUND TRIGGER: секции BEFORE STATEMENT / BEFORE EACH ROW / "
|
|
16
|
+
"AFTER EACH ROW / AFTER STATEMENT внутри одного триггера. У ora2pg нет "
|
|
17
|
+
"отдельного пути конвертации для этого синтаксиса. В файловом режиме "
|
|
18
|
+
"(-t TRIGGER -i file.sql) его regex-парсер (read_trigger_from_file) "
|
|
19
|
+
"рассчитан на классическую форму 'ON <table> [FOR EACH ROW] "
|
|
20
|
+
"[WHEN (...)] BEGIN...END' и на составном триггере тихо возвращает "
|
|
21
|
+
"0 найденных триггеров — без единой ошибки или предупреждения "
|
|
22
|
+
"(эмпирически подтверждено, docs/research/step0-show-report-baseline.md, "
|
|
23
|
+
"раздел 5). В режиме живого подключения счётчик объектов SHOW_REPORT "
|
|
24
|
+
"покажет этот триггер как обычный валидный (данные берутся из каталога "
|
|
25
|
+
"Oracle, а не из попытки конвертации) — то есть само число объектов "
|
|
26
|
+
"проблему не выдаст. По структуре export_trigger() в Ora2Pg.pm крайне "
|
|
27
|
+
"вероятно, что и в живом режиме конвертация тела COMPOUND TRIGGER даёт "
|
|
28
|
+
"синтаксически неверный или тихо испорченный код. Нужен ручной перенос "
|
|
29
|
+
"— как правило, на несколько независимых обычных триггеров "
|
|
30
|
+
"(BEFORE/AFTER × STATEMENT/ROW) с общим состоянием через пакетную "
|
|
31
|
+
"переменную или временную таблицу вместо секций компаунд-триггера."
|
|
32
|
+
)
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def find_compound_triggers(source: str) -> list[Finding]:
|
|
36
|
+
"""Detect CREATE [OR REPLACE] TRIGGER ... COMPOUND TRIGGER declarations.
|
|
37
|
+
|
|
38
|
+
Bounds each trigger by the next CREATE TRIGGER statement (or end of
|
|
39
|
+
file) rather than full block matching — Oracle does not support nested
|
|
40
|
+
trigger declarations, so this is exact, not an approximation.
|
|
41
|
+
"""
|
|
42
|
+
clean = mask_strings_and_comments(source)
|
|
43
|
+
matches = list(_TRIGGER_START_RE.finditer(clean))
|
|
44
|
+
|
|
45
|
+
findings: list[Finding] = []
|
|
46
|
+
for idx, match in enumerate(matches):
|
|
47
|
+
boundary = matches[idx + 1].start() if idx + 1 < len(matches) else len(clean)
|
|
48
|
+
span = clean[match.end() : boundary]
|
|
49
|
+
|
|
50
|
+
compound_match = _COMPOUND_RE.search(span)
|
|
51
|
+
if not compound_match:
|
|
52
|
+
continue
|
|
53
|
+
|
|
54
|
+
absolute_pos = match.end() + compound_match.start()
|
|
55
|
+
line_no = line_at(clean, absolute_pos)
|
|
56
|
+
|
|
57
|
+
findings.append(
|
|
58
|
+
Finding(
|
|
59
|
+
detector="compound_triggers",
|
|
60
|
+
severity="high",
|
|
61
|
+
object_name=match.group(1).upper(),
|
|
62
|
+
line=line_no,
|
|
63
|
+
snippet=compound_match.group(0).strip(),
|
|
64
|
+
message=_MESSAGE,
|
|
65
|
+
)
|
|
66
|
+
)
|
|
67
|
+
|
|
68
|
+
return findings
|
|
@@ -0,0 +1,110 @@
|
|
|
1
|
+
import re
|
|
2
|
+
|
|
3
|
+
from ..models import Finding
|
|
4
|
+
from ..plsql_lex import line_at, mask_strings_and_comments, skip_balanced_parens
|
|
5
|
+
|
|
6
|
+
_CONNECT_BY_RE = re.compile(r"\bCONNECT\s+BY\b", re.IGNORECASE)
|
|
7
|
+
_WITH_RECURSIVE_NAME_RE = re.compile(r"\bWITH\s+RECURSIVE\s+(\w+)\s+AS\s*\(", re.IGNORECASE)
|
|
8
|
+
_LEVEL_REF_RE = re.compile(r"(?<![A-Za-z0-9_$#])(?:\w+\.)?LEVEL\b", re.IGNORECASE)
|
|
9
|
+
# ora2pg always names the generated CTE "cte" regardless of the source
|
|
10
|
+
# query, so it's useless for identifying *which* function is affected in a
|
|
11
|
+
# report — find the nearest enclosing "CREATE [OR REPLACE] FUNCTION/
|
|
12
|
+
# PROCEDURE name" instead (ora2pg always emits one of these around a
|
|
13
|
+
# CONNECT BY conversion, package-scoped or standalone alike).
|
|
14
|
+
_ENCLOSING_ROUTINE_RE = re.compile(
|
|
15
|
+
r"CREATE\s+(?:OR\s+REPLACE\s+)?(?:FUNCTION|PROCEDURE)\s+(\w+)",
|
|
16
|
+
re.IGNORECASE,
|
|
17
|
+
)
|
|
18
|
+
|
|
19
|
+
# Object-type guess for the *Oracle source*, so the caller can pick the
|
|
20
|
+
# matching `ora2pg -t <TYPE>` mode instead of always assuming PACKAGE —
|
|
21
|
+
# CONNECT BY can just as well live in a standalone function/procedure.
|
|
22
|
+
_OBJECT_TYPE_PATTERNS = (
|
|
23
|
+
(re.compile(r"\bPACKAGE\s+BODY\b", re.IGNORECASE), "PACKAGE"),
|
|
24
|
+
(re.compile(r"\bCREATE\s+(?:OR\s+REPLACE\s+)?TRIGGER\b", re.IGNORECASE), "TRIGGER"),
|
|
25
|
+
(re.compile(r"\bCREATE\s+(?:OR\s+REPLACE\s+)?PROCEDURE\b", re.IGNORECASE), "PROCEDURE"),
|
|
26
|
+
(re.compile(r"\bCREATE\s+(?:OR\s+REPLACE\s+)?FUNCTION\b", re.IGNORECASE), "FUNCTION"),
|
|
27
|
+
(re.compile(r"\bCREATE\s+(?:OR\s+REPLACE\s+)?VIEW\b", re.IGNORECASE), "VIEW"),
|
|
28
|
+
)
|
|
29
|
+
|
|
30
|
+
_MESSAGE = (
|
|
31
|
+
"Сгенерированный ora2pg WITH RECURSIVE ссылается на LEVEL — псевдоколонку "
|
|
32
|
+
"Oracle, которой нет ни в PostgreSQL, ни в самом CTE. ora2pg переименовывает "
|
|
33
|
+
"LEVEL в столбец-счётчик глубины в анкорной ветке CTE, но не везде — "
|
|
34
|
+
"известный баг подстановки его regex-based конвертера CONNECT BY "
|
|
35
|
+
"(docs/research/step0-show-report-baseline.md, раздел 3; воспроизведено "
|
|
36
|
+
"на реальном прогоне ora2pg). Сгенерированный SQL в этом виде не "
|
|
37
|
+
"выполнится в PostgreSQL без ручной правки — LEVEL нужно заменить на "
|
|
38
|
+
"настоящее имя колонки-счётчика."
|
|
39
|
+
)
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def has_connect_by(source: str) -> bool:
|
|
43
|
+
"""Cheap pre-check on the *Oracle source*: is it worth spending an
|
|
44
|
+
ora2pg subprocess call on this file for a CONNECT BY conversion-quality
|
|
45
|
+
check at all?"""
|
|
46
|
+
return bool(_CONNECT_BY_RE.search(mask_strings_and_comments(source)))
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def guess_object_type(source: str) -> str:
|
|
50
|
+
"""Which `ora2pg -t <TYPE>` mode to run against this file. Checked in
|
|
51
|
+
order of specificity — a PACKAGE BODY containing CREATE FUNCTION text
|
|
52
|
+
(unlikely but not impossible in comments/strings, already masked out
|
|
53
|
+
here) must still resolve to PACKAGE, not FUNCTION."""
|
|
54
|
+
clean = mask_strings_and_comments(source)
|
|
55
|
+
for pattern, object_type in _OBJECT_TYPE_PATTERNS:
|
|
56
|
+
if pattern.search(clean):
|
|
57
|
+
return object_type
|
|
58
|
+
return "PACKAGE" # fallback: the most common shape in this project's scope
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def find_connect_by_risks(ora2pg_output: str) -> list[Finding]:
|
|
62
|
+
"""Lint ora2pg's *generated* SQL (not the Oracle source) for a specific,
|
|
63
|
+
confirmed ora2pg bug: a WITH RECURSIVE body that still references the
|
|
64
|
+
Oracle-only LEVEL pseudocolumn instead of the depth counter ora2pg
|
|
65
|
+
itself introduces in the anchor branch. Unlike the other three
|
|
66
|
+
detectors (which analyze Oracle source directly), this one's input is
|
|
67
|
+
ora2pg's own output — see ora2pg_gap_report/ora2pg_wrapper.run_estimate_cost().
|
|
68
|
+
|
|
69
|
+
ora2pg's own cost estimator already counts CONNECT BY correctly (see
|
|
70
|
+
step0-show-report-baseline.md section 3) — the gap this closes isn't
|
|
71
|
+
"was CONNECT BY seen", it's "is the conversion it produced actually
|
|
72
|
+
valid SQL".
|
|
73
|
+
"""
|
|
74
|
+
clean = mask_strings_and_comments(ora2pg_output)
|
|
75
|
+
routine_matches = list(_ENCLOSING_ROUTINE_RE.finditer(clean))
|
|
76
|
+
findings: list[Finding] = []
|
|
77
|
+
|
|
78
|
+
for m in _WITH_RECURSIVE_NAME_RE.finditer(clean):
|
|
79
|
+
cte_name = m.group(1)
|
|
80
|
+
paren_start = m.end() - 1
|
|
81
|
+
paren_end = skip_balanced_parens(clean, paren_start)
|
|
82
|
+
body = clean[paren_start:paren_end]
|
|
83
|
+
|
|
84
|
+
level_match = _LEVEL_REF_RE.search(body)
|
|
85
|
+
if not level_match:
|
|
86
|
+
continue
|
|
87
|
+
|
|
88
|
+
absolute_pos = paren_start + level_match.start()
|
|
89
|
+
object_name = _enclosing_routine_name(routine_matches, m.start()) or cte_name
|
|
90
|
+
findings.append(
|
|
91
|
+
Finding(
|
|
92
|
+
detector="connect_by",
|
|
93
|
+
severity="high",
|
|
94
|
+
object_name=object_name.upper(),
|
|
95
|
+
line=line_at(clean, absolute_pos),
|
|
96
|
+
snippet=level_match.group(0).strip(),
|
|
97
|
+
message=_MESSAGE,
|
|
98
|
+
)
|
|
99
|
+
)
|
|
100
|
+
|
|
101
|
+
return findings
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def _enclosing_routine_name(routine_matches: list, position: int):
|
|
105
|
+
name = None
|
|
106
|
+
for m in routine_matches:
|
|
107
|
+
if m.start() > position:
|
|
108
|
+
break
|
|
109
|
+
name = m.group(1)
|
|
110
|
+
return name
|
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
import re
|
|
2
|
+
|
|
3
|
+
from ..models import Finding
|
|
4
|
+
from ..plsql_lex import IDENTIFIER, line_at, mask_strings_and_comments
|
|
5
|
+
|
|
6
|
+
# A plain \b boundary would treat '$'/'#' as non-word, so e.g.
|
|
7
|
+
# "MY_PKG$UTL_FILE" would be misread as a real UTL_FILE reference — use a
|
|
8
|
+
# lookbehind consistent with plsql_lex.IDENTIFIER's character set instead.
|
|
9
|
+
_CALL_RE = re.compile(
|
|
10
|
+
rf"(?<![A-Za-z0-9_$#])(DBMS_[A-Za-z0-9_$#]*|UTL_[A-Za-z0-9_$#]*)\.({IDENTIFIER})",
|
|
11
|
+
re.IGNORECASE,
|
|
12
|
+
)
|
|
13
|
+
|
|
14
|
+
# Calls ora2pg genuinely rewrites to a working PostgreSQL equivalent
|
|
15
|
+
# (confirmed in docs/research/step0-show-report-baseline.md, section 4).
|
|
16
|
+
# Anything not listed here is treated as unsupported by default — that
|
|
17
|
+
# default is intentional: the research showed the overwhelming majority of
|
|
18
|
+
# DBMS_*/UTL_* usage has no targeted conversion, so "unknown" should read
|
|
19
|
+
# as "needs review", not "probably fine".
|
|
20
|
+
_CONVERTED = {
|
|
21
|
+
"DBMS_OUTPUT.PUT_LINE": "заменяется на вывод через встроенный ora2pg-хелпер (RAISE NOTICE-подобный механизм).",
|
|
22
|
+
"DBMS_OUTPUT.PUT": "заменяется тем же хелпером, что и DBMS_OUTPUT.PUT_LINE.",
|
|
23
|
+
"DBMS_OUTPUT.NEW_LINE": "заменяется тем же хелпером, что и DBMS_OUTPUT.PUT_LINE.",
|
|
24
|
+
"DBMS_OUTPUT.ENABLE": "просто комментируется — поведение теряется, но код не ломается.",
|
|
25
|
+
"DBMS_OUTPUT.DISABLE": "просто комментируется — поведение теряется, но код не ломается.",
|
|
26
|
+
"DBMS_LOB.GETLENGTH": "заменяется на octet_length().",
|
|
27
|
+
"DBMS_LOB.SUBSTR": "заменяется на substr() с перестановкой аргументов.",
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
_UNSUPPORTED_MESSAGE = (
|
|
31
|
+
"Специальной конвертации в ora2pg для этого конкретного вызова не "
|
|
32
|
+
"найдено (проверено по исходникам Ora2Pg/PLSQL.pm на шаге 0) — он "
|
|
33
|
+
"попадёт только в обезличенный счётчик DBMS_/UTL_ (вес 3 в "
|
|
34
|
+
"estimate_cost), а сам код останется как есть и не скомпилируется в "
|
|
35
|
+
"PostgreSQL без ручного переписывания или подключения расширения orafce "
|
|
36
|
+
"(если для этой функции там вообще есть эквивалент)."
|
|
37
|
+
)
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def find_dbms_utl_calls(source: str) -> list[Finding]:
|
|
41
|
+
"""Classify DBMS_*/UTL_* references: flag only the ones ora2pg has no
|
|
42
|
+
targeted conversion for. Calls ora2pg already handles (see _CONVERTED)
|
|
43
|
+
are not reported — they're not a gap, SHOW_REPORT's generic DBMS_/UTL_
|
|
44
|
+
counter already covers "is this package used at all" adequately; the
|
|
45
|
+
value here is telling the two apart.
|
|
46
|
+
"""
|
|
47
|
+
clean = mask_strings_and_comments(source)
|
|
48
|
+
findings: list[Finding] = []
|
|
49
|
+
|
|
50
|
+
for m in _CALL_RE.finditer(clean):
|
|
51
|
+
object_name = f"{m.group(1).upper()}.{m.group(2).upper()}"
|
|
52
|
+
if object_name in _CONVERTED:
|
|
53
|
+
continue
|
|
54
|
+
|
|
55
|
+
line_no = line_at(clean, m.start())
|
|
56
|
+
findings.append(
|
|
57
|
+
Finding(
|
|
58
|
+
detector="dbms_utl_calls",
|
|
59
|
+
severity="medium",
|
|
60
|
+
object_name=object_name,
|
|
61
|
+
line=line_no,
|
|
62
|
+
snippet=m.group(0),
|
|
63
|
+
message=_UNSUPPORTED_MESSAGE,
|
|
64
|
+
)
|
|
65
|
+
)
|
|
66
|
+
|
|
67
|
+
return findings
|
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
from .models import Finding
|
|
2
|
+
|
|
3
|
+
# Deliberately a range per severity, not a single number, and deliberately
|
|
4
|
+
# not lines-of-code-weighted: this is an uncalibrated heuristic, not a
|
|
5
|
+
# measurement. See PROJECT_BRIEF.md — presenting a fake-precise number here
|
|
6
|
+
# is a trust risk with exactly the audience this tool is for. Calibrate
|
|
7
|
+
# against real migration outcomes before treating these as commitments.
|
|
8
|
+
_HOURS_BY_SEVERITY: dict[str, tuple[float, float]] = {
|
|
9
|
+
"high": (2.0, 8.0),
|
|
10
|
+
"medium": (1.0, 4.0),
|
|
11
|
+
"low": (0.25, 1.0),
|
|
12
|
+
}
|
|
13
|
+
_DEFAULT_RANGE = (1.0, 4.0)
|
|
14
|
+
_SEVERITY_ORDER = ("high", "medium", "low")
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def estimate_hours(findings: list[Finding]) -> tuple[float, float]:
|
|
18
|
+
"""Sum of per-finding (low, high) hour ranges. A range, not a point
|
|
19
|
+
estimate — do not collapse it to an average and quote that as a
|
|
20
|
+
number; the spread itself is the honest part of the answer."""
|
|
21
|
+
total_low = total_high = 0.0
|
|
22
|
+
for f in findings:
|
|
23
|
+
lo, hi = _HOURS_BY_SEVERITY.get(f.severity, _DEFAULT_RANGE)
|
|
24
|
+
total_low += lo
|
|
25
|
+
total_high += hi
|
|
26
|
+
return total_low, total_high
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def summarize_by_severity(findings: list[Finding]) -> dict[str, int]:
|
|
30
|
+
"""Counts always sum to len(findings): an unrecognized severity value
|
|
31
|
+
(should not happen with the detectors in this repo today, but nothing
|
|
32
|
+
enforces it at the type level) lands in "other" instead of silently
|
|
33
|
+
vanishing from the displayed total."""
|
|
34
|
+
counts = {"high": 0, "medium": 0, "low": 0}
|
|
35
|
+
other = 0
|
|
36
|
+
for f in findings:
|
|
37
|
+
if f.severity in counts:
|
|
38
|
+
counts[f.severity] += 1
|
|
39
|
+
else:
|
|
40
|
+
other += 1
|
|
41
|
+
if other:
|
|
42
|
+
counts["other"] = other
|
|
43
|
+
return counts
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def ordered_counts(counts: dict[str, int]) -> list[tuple[str, int]]:
|
|
47
|
+
"""(name, count) pairs ordered high/medium/low first, then any other
|
|
48
|
+
bucket — shared so cli.py's Markdown header and terminal_report.py's
|
|
49
|
+
summary panel present the same ordering instead of each composing it
|
|
50
|
+
independently (and, before this, inconsistently: the Markdown header
|
|
51
|
+
used to fall back to plain dict order)."""
|
|
52
|
+
ordered = [(sev, counts[sev]) for sev in _SEVERITY_ORDER if counts.get(sev)]
|
|
53
|
+
ordered += [(name, n) for name, n in counts.items() if name not in _SEVERITY_ORDER and n]
|
|
54
|
+
return ordered
|