mainframe-migration-toolkit 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- mainframe_migration_toolkit-0.2.0.dist-info/METADATA +16 -0
- mainframe_migration_toolkit-0.2.0.dist-info/RECORD +52 -0
- mainframe_migration_toolkit-0.2.0.dist-info/WHEEL +4 -0
- mainframe_migration_toolkit-0.2.0.dist-info/entry_points.txt +3 -0
- mainframe_toolkit/__init__.py +92 -0
- mainframe_toolkit/__main__.py +5 -0
- mainframe_toolkit/_workspace/.claude/skills/analyze-mainframe-similarity/SKILL.md +30 -0
- mainframe_toolkit/_workspace/.claude/skills/migrate-mainframe-job/SKILL.md +63 -0
- mainframe_toolkit/_workspace/.claude/skills/validate-golden-dataset/SKILL.md +12 -0
- mainframe_toolkit/_workspace/AGENTS.md +10 -0
- mainframe_toolkit/_workspace/CLAUDE.md +2 -0
- mainframe_toolkit/_workspace/validator-java/.mvn/wrapper/maven-wrapper.properties +3 -0
- mainframe_toolkit/_workspace/validator-java/mvnw +295 -0
- mainframe_toolkit/_workspace/validator-java/mvnw.cmd +189 -0
- mainframe_toolkit/_workspace/validator-java/pom.xml +116 -0
- mainframe_toolkit/_workspace/validator-java/src/main/java/io/mainframe/migration/validator/AvroValueFormatter.java +152 -0
- mainframe_toolkit/_workspace/validator-java/src/main/java/io/mainframe/migration/validator/CsvTabularReader.java +111 -0
- mainframe_toolkit/_workspace/validator-java/src/main/java/io/mainframe/migration/validator/DataFormat.java +62 -0
- mainframe_toolkit/_workspace/validator-java/src/main/java/io/mainframe/migration/validator/Difference.java +34 -0
- mainframe_toolkit/_workspace/validator-java/src/main/java/io/mainframe/migration/validator/InputFileSet.java +45 -0
- mainframe_toolkit/_workspace/validator-java/src/main/java/io/mainframe/migration/validator/InputOptions.java +23 -0
- mainframe_toolkit/_workspace/validator-java/src/main/java/io/mainframe/migration/validator/JsonReportWriter.java +40 -0
- mainframe_toolkit/_workspace/validator-java/src/main/java/io/mainframe/migration/validator/MultiFileTabularReader.java +80 -0
- mainframe_toolkit/_workspace/validator-java/src/main/java/io/mainframe/migration/validator/Normalization.java +16 -0
- mainframe_toolkit/_workspace/validator-java/src/main/java/io/mainframe/migration/validator/ParquetTabularReader.java +61 -0
- mainframe_toolkit/_workspace/validator-java/src/main/java/io/mainframe/migration/validator/TabularReader.java +13 -0
- mainframe_toolkit/_workspace/validator-java/src/main/java/io/mainframe/migration/validator/TabularReaderFactory.java +29 -0
- mainframe_toolkit/_workspace/validator-java/src/main/java/io/mainframe/migration/validator/ValidationOptions.java +40 -0
- mainframe_toolkit/_workspace/validator-java/src/main/java/io/mainframe/migration/validator/ValidationReport.java +44 -0
- mainframe_toolkit/_workspace/validator-java/src/main/java/io/mainframe/migration/validator/ValidationService.java +395 -0
- mainframe_toolkit/_workspace/validator-java/src/main/java/io/mainframe/migration/validator/ValidatorCli.java +190 -0
- mainframe_toolkit/_workspace/validator-java/src/main/java/io/mainframe/migration/validator/ValueNormalizer.java +27 -0
- mainframe_toolkit/_workspace/validator-java/src/test/java/io/mainframe/migration/validator/DirectoryValidationTest.java +70 -0
- mainframe_toolkit/_workspace/validator-java/src/test/java/io/mainframe/migration/validator/JsonReportWriterTest.java +43 -0
- mainframe_toolkit/_workspace/validator-java/src/test/java/io/mainframe/migration/validator/KeyedValidationTest.java +89 -0
- mainframe_toolkit/_workspace/validator-java/src/test/java/io/mainframe/migration/validator/ParquetValidationTest.java +146 -0
- mainframe_toolkit/_workspace/validator-java/src/test/java/io/mainframe/migration/validator/ValidationServiceTest.java +137 -0
- mainframe_toolkit/_workspace/validator-java/src/test/java/io/mainframe/migration/validator/ValidatorCliTest.java +102 -0
- mainframe_toolkit/cli.py +361 -0
- mainframe_toolkit/cobol.py +106 -0
- mainframe_toolkit/copybook.py +558 -0
- mainframe_toolkit/errors.py +15 -0
- mainframe_toolkit/external.py +48 -0
- mainframe_toolkit/io.py +202 -0
- mainframe_toolkit/jcl.py +126 -0
- mainframe_toolkit/pipeline.py +322 -0
- mainframe_toolkit/sequential.py +259 -0
- mainframe_toolkit/similarity.py +1171 -0
- mainframe_toolkit/sorting.py +60 -0
- mainframe_toolkit/specs.py +312 -0
- mainframe_toolkit/synthetic.py +108 -0
- mainframe_toolkit/workspace.py +133 -0
|
@@ -0,0 +1,558 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
import re
|
|
5
|
+
from dataclasses import asdict, dataclass, field
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
from typing import Any
|
|
8
|
+
|
|
9
|
+
from .sequential import FieldSpec, RecordLayout
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
class CopybookParseError(ValueError):
|
|
13
|
+
pass
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
@dataclass
|
|
17
|
+
class _Entry:
|
|
18
|
+
level: int
|
|
19
|
+
name: str
|
|
20
|
+
body: str
|
|
21
|
+
picture: str | None
|
|
22
|
+
usage: str | None
|
|
23
|
+
occurs: int
|
|
24
|
+
redefines: str | None
|
|
25
|
+
separate_sign: bool
|
|
26
|
+
children: list["_Entry"] = field(default_factory=list)
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
@dataclass(frozen=True)
|
|
30
|
+
class _Picture:
|
|
31
|
+
digits: int
|
|
32
|
+
characters: int
|
|
33
|
+
scale: int
|
|
34
|
+
signed: bool
|
|
35
|
+
textual: bool
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
_ENTRY_RE = re.compile(
|
|
39
|
+
r"(?ims)^[ \t]*(?P<level>\d{1,2})[ \t]+"
|
|
40
|
+
r"(?P<name>[A-Z0-9][A-Z0-9_-]*)\b(?P<body>.*?)"
|
|
41
|
+
r"(?=^[ \t]*\d{1,2}[ \t]+[A-Z0-9]|\Z)"
|
|
42
|
+
)
|
|
43
|
+
_PICTURE_RE = re.compile(
|
|
44
|
+
r"\bPIC(?:TURE)?(?:\s+IS)?\s+"
|
|
45
|
+
r"(?P<picture>(?:[AX9SVP](?:\(\s*\d+\s*\))?)+)",
|
|
46
|
+
re.IGNORECASE,
|
|
47
|
+
)
|
|
48
|
+
_PICTURE_TOKEN_RE = re.compile(r"([AX9SVP])(?:\(\s*(\d+)\s*\))?", re.IGNORECASE)
|
|
49
|
+
_USAGE_RE = re.compile(
|
|
50
|
+
r"\b(?:USAGE\s+(?:IS\s+)?)?"
|
|
51
|
+
r"(PACKED-DECIMAL|COMPUTATIONAL-[1-5]|COMP-[1-5]|"
|
|
52
|
+
r"COMPUTATIONAL|COMP|BINARY|DISPLAY)\b",
|
|
53
|
+
re.IGNORECASE,
|
|
54
|
+
)
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def parse_copybook(
|
|
58
|
+
source: str | Path,
|
|
59
|
+
*,
|
|
60
|
+
source_format: str = "auto",
|
|
61
|
+
record_name: str | None = None,
|
|
62
|
+
encoding: str = "utf-8",
|
|
63
|
+
source_encoding: str = "utf-8",
|
|
64
|
+
include_fillers: bool = True,
|
|
65
|
+
) -> RecordLayout:
|
|
66
|
+
text = _read_source(source, source_encoding)
|
|
67
|
+
return parse_copybook_text(
|
|
68
|
+
text,
|
|
69
|
+
source_format=source_format,
|
|
70
|
+
record_name=record_name,
|
|
71
|
+
encoding=encoding,
|
|
72
|
+
include_fillers=include_fillers,
|
|
73
|
+
)
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def parse_copybook_file(
|
|
77
|
+
path: str | Path,
|
|
78
|
+
**options: Any,
|
|
79
|
+
) -> RecordLayout:
|
|
80
|
+
return parse_copybook(Path(path), **options)
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def parse_copybook_text(
|
|
84
|
+
text: str,
|
|
85
|
+
*,
|
|
86
|
+
source_format: str = "auto",
|
|
87
|
+
record_name: str | None = None,
|
|
88
|
+
encoding: str = "utf-8",
|
|
89
|
+
include_fillers: bool = True,
|
|
90
|
+
) -> RecordLayout:
|
|
91
|
+
entries = _parse_entries(_normalise_source(text, source_format))
|
|
92
|
+
root = _select_root(entries, record_name)
|
|
93
|
+
fields, record_length = _place_node(root, 0, None, ())
|
|
94
|
+
fields = _unique_fields(fields)
|
|
95
|
+
if not include_fillers:
|
|
96
|
+
fields = [item for item in fields if not _is_filler(item.name)]
|
|
97
|
+
return RecordLayout(tuple(fields), record_length=record_length, encoding=encoding)
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def layout_to_dict(layout: RecordLayout) -> dict[str, Any]:
|
|
101
|
+
return {
|
|
102
|
+
"fields": [asdict(item) for item in layout.fields],
|
|
103
|
+
"record_length": layout.record_length,
|
|
104
|
+
"encoding": layout.encoding,
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
def layout_to_json(
|
|
109
|
+
layout: RecordLayout,
|
|
110
|
+
path: str | Path | None = None,
|
|
111
|
+
*,
|
|
112
|
+
indent: int | None = 2,
|
|
113
|
+
) -> str:
|
|
114
|
+
rendered = json.dumps(layout_to_dict(layout), ensure_ascii=False, indent=indent) + "\n"
|
|
115
|
+
if path is not None:
|
|
116
|
+
destination = Path(path)
|
|
117
|
+
destination.parent.mkdir(parents=True, exist_ok=True)
|
|
118
|
+
destination.write_text(rendered, encoding="utf-8")
|
|
119
|
+
return rendered
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def write_layout_json(layout: RecordLayout, path: str | Path) -> None:
|
|
123
|
+
layout_to_json(layout, path)
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
def _read_source(source: str | Path, encoding: str) -> str:
|
|
127
|
+
if isinstance(source, Path):
|
|
128
|
+
return source.read_text(encoding=encoding)
|
|
129
|
+
if "\n" in source or "\r" in source:
|
|
130
|
+
return source
|
|
131
|
+
try:
|
|
132
|
+
path = Path(source)
|
|
133
|
+
if path.is_file():
|
|
134
|
+
return path.read_text(encoding=encoding)
|
|
135
|
+
except OSError:
|
|
136
|
+
pass
|
|
137
|
+
return source
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
def _normalise_source(text: str, source_format: str) -> str:
|
|
141
|
+
selected = source_format.lower().replace("-", "_")
|
|
142
|
+
if selected not in {"auto", "fixed", "free"}:
|
|
143
|
+
raise ValueError("source_format must be auto, fixed, or free")
|
|
144
|
+
lines = text.expandtabs(8).splitlines()
|
|
145
|
+
directive = _source_directive(lines)
|
|
146
|
+
if selected == "auto":
|
|
147
|
+
selected = directive or _detect_source_format(lines)
|
|
148
|
+
|
|
149
|
+
cleaned: list[str] = []
|
|
150
|
+
for raw in lines:
|
|
151
|
+
if re.search(r">>\s*SOURCE\s+FORMAT", raw, re.IGNORECASE):
|
|
152
|
+
continue
|
|
153
|
+
if selected == "fixed":
|
|
154
|
+
padded = raw.ljust(7)
|
|
155
|
+
indicator = padded[6]
|
|
156
|
+
if indicator in {"*", "/", "D", "d"}:
|
|
157
|
+
continue
|
|
158
|
+
content = padded[7:72]
|
|
159
|
+
content = _strip_inline_comment(content)
|
|
160
|
+
if indicator == "-" and cleaned:
|
|
161
|
+
cleaned[-1] = f"{cleaned[-1].rstrip()} {content.lstrip()}"
|
|
162
|
+
continue
|
|
163
|
+
else:
|
|
164
|
+
content = _strip_inline_comment(raw)
|
|
165
|
+
if content.lstrip().startswith(("*", "/")):
|
|
166
|
+
continue
|
|
167
|
+
if content.strip():
|
|
168
|
+
cleaned.append(content.rstrip())
|
|
169
|
+
return _break_inline_entries("\n".join(cleaned))
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
def _source_directive(lines: list[str]) -> str | None:
|
|
173
|
+
for line in lines:
|
|
174
|
+
match = re.search(
|
|
175
|
+
r">>\s*SOURCE\s+FORMAT(?:\s+IS)?\s+(FREE|FIXED)",
|
|
176
|
+
line,
|
|
177
|
+
re.IGNORECASE,
|
|
178
|
+
)
|
|
179
|
+
if match:
|
|
180
|
+
return match.group(1).lower()
|
|
181
|
+
return None
|
|
182
|
+
|
|
183
|
+
|
|
184
|
+
def _detect_source_format(lines: list[str]) -> str:
|
|
185
|
+
if any(re.match(r"^\d{6}.", line) for line in lines):
|
|
186
|
+
return "fixed"
|
|
187
|
+
early_entry = re.compile(r"^\s{0,6}\d{1,2}\s+[A-Z0-9]", re.IGNORECASE)
|
|
188
|
+
if any(early_entry.match(line) for line in lines):
|
|
189
|
+
return "free"
|
|
190
|
+
if any(len(line) > 6 and line[6] in {"*", "/", "-"} for line in lines):
|
|
191
|
+
return "fixed"
|
|
192
|
+
if any(len(line) > 7 and not line[:7].strip() for line in lines):
|
|
193
|
+
return "fixed"
|
|
194
|
+
return "free"
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
def _strip_inline_comment(line: str) -> str:
|
|
198
|
+
quote: str | None = None
|
|
199
|
+
index = 0
|
|
200
|
+
while index < len(line) - 1:
|
|
201
|
+
character = line[index]
|
|
202
|
+
if character in {"'", '"'}:
|
|
203
|
+
if quote == character and index + 1 < len(line) and line[index + 1] == character:
|
|
204
|
+
index += 2
|
|
205
|
+
continue
|
|
206
|
+
quote = None if quote == character else character if quote is None else quote
|
|
207
|
+
if quote is None and line[index : index + 2] == "*>":
|
|
208
|
+
return line[:index]
|
|
209
|
+
index += 1
|
|
210
|
+
return line
|
|
211
|
+
|
|
212
|
+
|
|
213
|
+
def _break_inline_entries(source: str) -> str:
|
|
214
|
+
next_entry = re.compile(
|
|
215
|
+
r"\s+(?:0?[1-9]|[1-4]\d|66|77|78|88)\s+[A-Z0-9]",
|
|
216
|
+
re.IGNORECASE,
|
|
217
|
+
)
|
|
218
|
+
result: list[str] = []
|
|
219
|
+
quote: str | None = None
|
|
220
|
+
index = 0
|
|
221
|
+
while index < len(source):
|
|
222
|
+
character = source[index]
|
|
223
|
+
if character in {"'", '"'}:
|
|
224
|
+
if quote == character and index + 1 < len(source) and source[index + 1] == character:
|
|
225
|
+
result.extend((character, character))
|
|
226
|
+
index += 2
|
|
227
|
+
continue
|
|
228
|
+
quote = None if quote == character else character if quote is None else quote
|
|
229
|
+
result.append(character)
|
|
230
|
+
if character == "." and quote is None:
|
|
231
|
+
match = next_entry.match(source, index + 1)
|
|
232
|
+
if match:
|
|
233
|
+
result.append("\n")
|
|
234
|
+
index += 1
|
|
235
|
+
return "".join(result)
|
|
236
|
+
|
|
237
|
+
|
|
238
|
+
def _parse_entries(source: str) -> list[_Entry]:
|
|
239
|
+
roots: list[_Entry] = []
|
|
240
|
+
stack: list[_Entry] = []
|
|
241
|
+
filler_count = 0
|
|
242
|
+
for match in _ENTRY_RE.finditer(source.upper()):
|
|
243
|
+
level = int(match.group("level"))
|
|
244
|
+
if level in {66, 78, 88}:
|
|
245
|
+
continue
|
|
246
|
+
if level == 0 or level > 49 and level != 77:
|
|
247
|
+
continue
|
|
248
|
+
name = match.group("name")
|
|
249
|
+
if name == "FILLER":
|
|
250
|
+
filler_count += 1
|
|
251
|
+
name = "FILLER" if filler_count == 1 else f"FILLER-{filler_count}"
|
|
252
|
+
body = " ".join(match.group("body").replace(".", " ").split())
|
|
253
|
+
picture_match = _PICTURE_RE.search(body)
|
|
254
|
+
picture = None
|
|
255
|
+
if picture_match:
|
|
256
|
+
picture = re.sub(r"\s+", "", picture_match.group("picture"))
|
|
257
|
+
usage_match = _USAGE_RE.search(body)
|
|
258
|
+
usage = _normalise_usage(usage_match.group(1)) if usage_match else None
|
|
259
|
+
occurs = _fixed_occurs(body)
|
|
260
|
+
redefines_match = re.search(r"\bREDEFINES\s+([A-Z0-9][A-Z0-9_-]*)\b", body)
|
|
261
|
+
entry = _Entry(
|
|
262
|
+
level=level,
|
|
263
|
+
name=name,
|
|
264
|
+
body=body,
|
|
265
|
+
picture=picture,
|
|
266
|
+
usage=usage,
|
|
267
|
+
occurs=occurs,
|
|
268
|
+
redefines=redefines_match.group(1) if redefines_match else None,
|
|
269
|
+
separate_sign=bool(re.search(r"\bSIGN\b.*?\bSEPARATE\b", body)),
|
|
270
|
+
)
|
|
271
|
+
|
|
272
|
+
if level == 77:
|
|
273
|
+
stack.clear()
|
|
274
|
+
else:
|
|
275
|
+
while stack and level <= stack[-1].level:
|
|
276
|
+
stack.pop()
|
|
277
|
+
if stack:
|
|
278
|
+
if stack[-1].picture or stack[-1].usage in {"comp_1", "comp_2"}:
|
|
279
|
+
raise CopybookParseError(f"elementary item {stack[-1].name} cannot contain {name}")
|
|
280
|
+
stack[-1].children.append(entry)
|
|
281
|
+
else:
|
|
282
|
+
roots.append(entry)
|
|
283
|
+
stack.append(entry)
|
|
284
|
+
|
|
285
|
+
if not roots:
|
|
286
|
+
raise CopybookParseError("copybook contains no record entries")
|
|
287
|
+
return roots
|
|
288
|
+
|
|
289
|
+
|
|
290
|
+
def _normalise_usage(value: str) -> str:
|
|
291
|
+
usage = value.upper().replace("COMPUTATIONAL", "COMP")
|
|
292
|
+
return {
|
|
293
|
+
"DISPLAY": "display",
|
|
294
|
+
"PACKED-DECIMAL": "packed_decimal",
|
|
295
|
+
"COMP-3": "packed_decimal",
|
|
296
|
+
"BINARY": "binary",
|
|
297
|
+
"COMP": "binary",
|
|
298
|
+
"COMP-4": "binary",
|
|
299
|
+
"COMP-5": "binary",
|
|
300
|
+
"COMP-1": "comp_1",
|
|
301
|
+
"COMP-2": "comp_2",
|
|
302
|
+
}[usage]
|
|
303
|
+
|
|
304
|
+
|
|
305
|
+
def _fixed_occurs(body: str) -> int:
|
|
306
|
+
match = re.search(r"\bOCCURS\s+(\d+)\b", body)
|
|
307
|
+
if not match:
|
|
308
|
+
return 1
|
|
309
|
+
tail = body[match.end() :]
|
|
310
|
+
if re.match(r"\s+TO\s+\d+", tail) or re.search(r"\bDEPENDING\s+ON\b", tail):
|
|
311
|
+
raise CopybookParseError("variable OCCURS is not supported")
|
|
312
|
+
count = int(match.group(1))
|
|
313
|
+
if count < 1:
|
|
314
|
+
raise CopybookParseError("OCCURS must be greater than zero")
|
|
315
|
+
return count
|
|
316
|
+
|
|
317
|
+
|
|
318
|
+
def _select_root(entries: list[_Entry], record_name: str | None) -> _Entry:
|
|
319
|
+
if record_name is None:
|
|
320
|
+
return entries[0]
|
|
321
|
+
wanted = record_name.upper().replace("_", "-")
|
|
322
|
+
pending = list(entries)
|
|
323
|
+
while pending:
|
|
324
|
+
entry = pending.pop(0)
|
|
325
|
+
if entry.name.replace("_", "-") == wanted:
|
|
326
|
+
return entry
|
|
327
|
+
pending[0:0] = entry.children
|
|
328
|
+
raise CopybookParseError(f"record {record_name!r} was not found")
|
|
329
|
+
|
|
330
|
+
|
|
331
|
+
def _place_node(
|
|
332
|
+
entry: _Entry,
|
|
333
|
+
start: int,
|
|
334
|
+
inherited_usage: str | None,
|
|
335
|
+
occurrence_path: tuple[int, ...],
|
|
336
|
+
) -> tuple[list[FieldSpec], int]:
|
|
337
|
+
usage = entry.usage or inherited_usage
|
|
338
|
+
if entry.children:
|
|
339
|
+
if entry.picture is not None or usage in {"comp_1", "comp_2"}:
|
|
340
|
+
raise CopybookParseError(f"invalid group item {entry.name}")
|
|
341
|
+
all_fields: list[FieldSpec] = []
|
|
342
|
+
one_length: int | None = None
|
|
343
|
+
for occurrence in range(1, entry.occurs + 1):
|
|
344
|
+
suffix = occurrence_path + ((occurrence,) if entry.occurs > 1 else ())
|
|
345
|
+
base = start + (one_length or 0) * (occurrence - 1)
|
|
346
|
+
fields, length = _place_children(entry.children, base, usage, suffix)
|
|
347
|
+
if one_length is None:
|
|
348
|
+
one_length = length
|
|
349
|
+
elif length != one_length:
|
|
350
|
+
raise CopybookParseError(f"non-deterministic length for {entry.name}")
|
|
351
|
+
all_fields.extend(fields)
|
|
352
|
+
return all_fields, (one_length or 0) * entry.occurs
|
|
353
|
+
|
|
354
|
+
if usage in {"comp_1", "comp_2"} and entry.picture is None:
|
|
355
|
+
length = 4 if usage == "comp_1" else 8
|
|
356
|
+
fields = []
|
|
357
|
+
for occurrence in range(1, entry.occurs + 1):
|
|
358
|
+
suffix = occurrence_path + ((occurrence,) if entry.occurs > 1 else ())
|
|
359
|
+
fields.append(
|
|
360
|
+
_field(
|
|
361
|
+
entry,
|
|
362
|
+
start + (occurrence - 1) * length,
|
|
363
|
+
length,
|
|
364
|
+
"bytes",
|
|
365
|
+
0,
|
|
366
|
+
True,
|
|
367
|
+
suffix,
|
|
368
|
+
"latin-1",
|
|
369
|
+
)
|
|
370
|
+
)
|
|
371
|
+
return fields, length * entry.occurs
|
|
372
|
+
if entry.picture is None:
|
|
373
|
+
raise CopybookParseError(f"group item {entry.name} has no subordinate entries")
|
|
374
|
+
|
|
375
|
+
picture = _parse_picture(entry.picture)
|
|
376
|
+
selected_usage = usage or "display"
|
|
377
|
+
length, kind, field_encoding = _physical_type(picture, selected_usage, entry.separate_sign)
|
|
378
|
+
all_fields = []
|
|
379
|
+
for occurrence in range(1, entry.occurs + 1):
|
|
380
|
+
suffix = occurrence_path + ((occurrence,) if entry.occurs > 1 else ())
|
|
381
|
+
all_fields.append(
|
|
382
|
+
_field(
|
|
383
|
+
entry,
|
|
384
|
+
start + (occurrence - 1) * length,
|
|
385
|
+
length,
|
|
386
|
+
kind,
|
|
387
|
+
picture.scale,
|
|
388
|
+
picture.signed,
|
|
389
|
+
suffix,
|
|
390
|
+
field_encoding,
|
|
391
|
+
)
|
|
392
|
+
)
|
|
393
|
+
return all_fields, length * entry.occurs
|
|
394
|
+
|
|
395
|
+
|
|
396
|
+
def _place_children(
|
|
397
|
+
children: list[_Entry],
|
|
398
|
+
start: int,
|
|
399
|
+
inherited_usage: str | None,
|
|
400
|
+
occurrence_path: tuple[int, ...],
|
|
401
|
+
) -> tuple[list[FieldSpec], int]:
|
|
402
|
+
cursor = start
|
|
403
|
+
extent = start
|
|
404
|
+
positions: dict[str, tuple[int, int]] = {}
|
|
405
|
+
fields: list[FieldSpec] = []
|
|
406
|
+
for child in children:
|
|
407
|
+
child_start = cursor
|
|
408
|
+
if child.redefines:
|
|
409
|
+
target = positions.get(child.redefines)
|
|
410
|
+
if target is None:
|
|
411
|
+
raise CopybookParseError(
|
|
412
|
+
f"{child.name} redefines unknown item {child.redefines}"
|
|
413
|
+
)
|
|
414
|
+
child_start = target[0]
|
|
415
|
+
child_fields, child_length = _place_node(
|
|
416
|
+
child, child_start, inherited_usage, occurrence_path
|
|
417
|
+
)
|
|
418
|
+
fields.extend(child_fields)
|
|
419
|
+
positions[child.name] = (child_start, child_length)
|
|
420
|
+
extent = max(extent, child_start + child_length)
|
|
421
|
+
cursor = max(cursor, child_start + child_length)
|
|
422
|
+
return fields, extent - start
|
|
423
|
+
|
|
424
|
+
|
|
425
|
+
def _parse_picture(value: str) -> _Picture:
|
|
426
|
+
position = 0
|
|
427
|
+
tokens: list[tuple[str, int]] = []
|
|
428
|
+
for match in _PICTURE_TOKEN_RE.finditer(value):
|
|
429
|
+
if match.start() != position:
|
|
430
|
+
raise CopybookParseError(f"unsupported PICTURE {value!r}")
|
|
431
|
+
count = int(match.group(2) or 1)
|
|
432
|
+
if count < 1:
|
|
433
|
+
raise CopybookParseError(f"invalid repetition in PICTURE {value!r}")
|
|
434
|
+
tokens.append((match.group(1).upper(), count))
|
|
435
|
+
position = match.end()
|
|
436
|
+
if position != len(value) or not tokens:
|
|
437
|
+
raise CopybookParseError(f"unsupported PICTURE {value!r}")
|
|
438
|
+
|
|
439
|
+
digits = sum(count for symbol, count in tokens if symbol == "9")
|
|
440
|
+
characters = sum(count for symbol, count in tokens if symbol in {"A", "X", "9"})
|
|
441
|
+
textual = any(symbol in {"A", "X"} for symbol, _ in tokens)
|
|
442
|
+
signed = any(symbol == "S" for symbol, _ in tokens)
|
|
443
|
+
if textual:
|
|
444
|
+
if any(symbol in {"S", "V", "P"} for symbol, _ in tokens):
|
|
445
|
+
raise CopybookParseError(f"mixed numeric/alphanumeric PICTURE {value!r}")
|
|
446
|
+
return _Picture(digits, characters, 0, False, True)
|
|
447
|
+
if digits == 0:
|
|
448
|
+
raise CopybookParseError(f"numeric PICTURE {value!r} contains no digits")
|
|
449
|
+
|
|
450
|
+
symbols = [symbol for symbol, _ in tokens]
|
|
451
|
+
first_digit = next(index for index, item in enumerate(tokens) if item[0] == "9")
|
|
452
|
+
last_digit = len(tokens) - 1 - next(
|
|
453
|
+
index for index, item in enumerate(reversed(tokens)) if item[0] == "9"
|
|
454
|
+
)
|
|
455
|
+
leading_p = sum(count for symbol, count in tokens[:first_digit] if symbol == "P")
|
|
456
|
+
trailing_p = sum(count for symbol, count in tokens[last_digit + 1 :] if symbol == "P")
|
|
457
|
+
if leading_p:
|
|
458
|
+
scale = digits + leading_p
|
|
459
|
+
elif trailing_p:
|
|
460
|
+
scale = -trailing_p
|
|
461
|
+
elif "V" in symbols:
|
|
462
|
+
v_index = symbols.index("V")
|
|
463
|
+
scale = sum(
|
|
464
|
+
count for symbol, count in tokens[v_index + 1 :] if symbol in {"9", "P"}
|
|
465
|
+
)
|
|
466
|
+
else:
|
|
467
|
+
scale = 0
|
|
468
|
+
return _Picture(digits, characters, scale, signed, False)
|
|
469
|
+
|
|
470
|
+
|
|
471
|
+
def _physical_type(
|
|
472
|
+
picture: _Picture,
|
|
473
|
+
usage: str,
|
|
474
|
+
separate_sign: bool,
|
|
475
|
+
) -> tuple[int, str, str | None]:
|
|
476
|
+
if picture.textual:
|
|
477
|
+
if usage != "display":
|
|
478
|
+
raise CopybookParseError(f"usage {usage} is invalid for an alphanumeric item")
|
|
479
|
+
return picture.characters, "text", None
|
|
480
|
+
if usage == "display":
|
|
481
|
+
length = picture.digits + (1 if picture.signed and separate_sign else 0)
|
|
482
|
+
if picture.signed and not separate_sign:
|
|
483
|
+
return length, "zoned_decimal", None
|
|
484
|
+
return length, "decimal" if picture.scale else "integer", None
|
|
485
|
+
if usage == "packed_decimal":
|
|
486
|
+
return (picture.digits + 2) // 2, "packed_decimal", None
|
|
487
|
+
if usage == "binary":
|
|
488
|
+
if picture.digits <= 4:
|
|
489
|
+
length = 2
|
|
490
|
+
elif picture.digits <= 9:
|
|
491
|
+
length = 4
|
|
492
|
+
elif picture.digits <= 18:
|
|
493
|
+
length = 8
|
|
494
|
+
else:
|
|
495
|
+
raise CopybookParseError("binary PICTURE exceeds 18 stored digits")
|
|
496
|
+
return length, "binary", None
|
|
497
|
+
if usage == "comp_1":
|
|
498
|
+
return 4, "bytes", "latin-1"
|
|
499
|
+
if usage == "comp_2":
|
|
500
|
+
return 8, "bytes", "latin-1"
|
|
501
|
+
raise CopybookParseError(f"unsupported usage {usage!r}")
|
|
502
|
+
|
|
503
|
+
|
|
504
|
+
def _field(
|
|
505
|
+
entry: _Entry,
|
|
506
|
+
start: int,
|
|
507
|
+
length: int,
|
|
508
|
+
kind: str,
|
|
509
|
+
scale: int,
|
|
510
|
+
signed: bool,
|
|
511
|
+
occurrence_path: tuple[int, ...],
|
|
512
|
+
encoding: str | None,
|
|
513
|
+
) -> FieldSpec:
|
|
514
|
+
suffix = "".join(f"[{index}]" for index in occurrence_path)
|
|
515
|
+
return FieldSpec(
|
|
516
|
+
name=f"{entry.name}{suffix}",
|
|
517
|
+
start=start + 1,
|
|
518
|
+
length=length,
|
|
519
|
+
kind=kind,
|
|
520
|
+
scale=scale,
|
|
521
|
+
encoding=encoding,
|
|
522
|
+
signed=signed,
|
|
523
|
+
)
|
|
524
|
+
|
|
525
|
+
|
|
526
|
+
def _unique_fields(fields: list[FieldSpec]) -> list[FieldSpec]:
|
|
527
|
+
counts: dict[str, int] = {}
|
|
528
|
+
used: set[str] = set()
|
|
529
|
+
result: list[FieldSpec] = []
|
|
530
|
+
for item in fields:
|
|
531
|
+
base = item.name
|
|
532
|
+
counts[base] = counts.get(base, 0) + 1
|
|
533
|
+
name = base if counts[base] == 1 else f"{base}#{counts[base]}"
|
|
534
|
+
while name in used:
|
|
535
|
+
counts[base] += 1
|
|
536
|
+
name = f"{base}#{counts[base]}"
|
|
537
|
+
used.add(name)
|
|
538
|
+
result.append(
|
|
539
|
+
item
|
|
540
|
+
if name == item.name
|
|
541
|
+
else FieldSpec(**{**asdict(item), "name": name})
|
|
542
|
+
)
|
|
543
|
+
return result
|
|
544
|
+
|
|
545
|
+
|
|
546
|
+
def _is_filler(name: str) -> bool:
|
|
547
|
+
return bool(re.match(r"^FILLER(?:-\d+)?(?:\[|#|$)", name))
|
|
548
|
+
|
|
549
|
+
|
|
550
|
+
__all__ = [
|
|
551
|
+
"CopybookParseError",
|
|
552
|
+
"layout_to_dict",
|
|
553
|
+
"layout_to_json",
|
|
554
|
+
"parse_copybook",
|
|
555
|
+
"parse_copybook_file",
|
|
556
|
+
"parse_copybook_text",
|
|
557
|
+
"write_layout_json",
|
|
558
|
+
]
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
class ToolkitError(Exception):
|
|
2
|
+
"""Base error raised by the toolkit."""
|
|
3
|
+
|
|
4
|
+
|
|
5
|
+
class OptionalDependencyError(ToolkitError):
|
|
6
|
+
"""Raised when an operation needs an optional package."""
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
class DataFormatError(ToolkitError):
|
|
10
|
+
"""Raised when a source record cannot be decoded safely."""
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
class SpecificationError(ToolkitError):
|
|
14
|
+
"""Raised when the migration project contract is invalid."""
|
|
15
|
+
|
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
from typing import Any, Callable, Mapping
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
ExternalFunction = Callable[..., Any]
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
class ExternalProgramRegistry:
|
|
12
|
+
def __init__(self, specifications: Mapping[str, str] | None = None):
|
|
13
|
+
self.specifications = {name.upper(): description for name, description in (specifications or {}).items()}
|
|
14
|
+
self._implementations: dict[str, ExternalFunction] = {}
|
|
15
|
+
|
|
16
|
+
@classmethod
|
|
17
|
+
def from_directory(cls, directory: str | Path) -> "ExternalProgramRegistry":
|
|
18
|
+
root = Path(directory)
|
|
19
|
+
combined: dict[str, str] = {}
|
|
20
|
+
for path in sorted(root.glob("*.json")):
|
|
21
|
+
with path.open(encoding="utf-8") as stream:
|
|
22
|
+
value = json.load(stream)
|
|
23
|
+
if not isinstance(value, dict) or not all(isinstance(item, str) for item in value.values()):
|
|
24
|
+
raise ValueError(f"{path} must contain a name-to-description object")
|
|
25
|
+
combined.update({str(name): description for name, description in value.items()})
|
|
26
|
+
return cls(combined)
|
|
27
|
+
|
|
28
|
+
def register(self, name: str, function: ExternalFunction) -> None:
|
|
29
|
+
self._implementations[name.upper()] = function
|
|
30
|
+
|
|
31
|
+
def implementation(self, name: str) -> ExternalFunction:
|
|
32
|
+
key = name.upper()
|
|
33
|
+
if key not in self._implementations:
|
|
34
|
+
description = self.specifications.get(key, "specification unavailable")
|
|
35
|
+
raise LookupError(f"external program {name} has no adapter: {description}")
|
|
36
|
+
return self._implementations[key]
|
|
37
|
+
|
|
38
|
+
def call(self, name: str, *args: Any, **kwargs: Any) -> Any:
|
|
39
|
+
return self.implementation(name)(*args, **kwargs)
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def external_program(name: str, registry: ExternalProgramRegistry) -> Callable[[ExternalFunction], ExternalFunction]:
|
|
43
|
+
def decorator(function: ExternalFunction) -> ExternalFunction:
|
|
44
|
+
registry.register(name, function)
|
|
45
|
+
return function
|
|
46
|
+
|
|
47
|
+
return decorator
|
|
48
|
+
|