mainframe-migration-toolkit 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (52) hide show
  1. mainframe_migration_toolkit-0.2.0.dist-info/METADATA +16 -0
  2. mainframe_migration_toolkit-0.2.0.dist-info/RECORD +52 -0
  3. mainframe_migration_toolkit-0.2.0.dist-info/WHEEL +4 -0
  4. mainframe_migration_toolkit-0.2.0.dist-info/entry_points.txt +3 -0
  5. mainframe_toolkit/__init__.py +92 -0
  6. mainframe_toolkit/__main__.py +5 -0
  7. mainframe_toolkit/_workspace/.claude/skills/analyze-mainframe-similarity/SKILL.md +30 -0
  8. mainframe_toolkit/_workspace/.claude/skills/migrate-mainframe-job/SKILL.md +63 -0
  9. mainframe_toolkit/_workspace/.claude/skills/validate-golden-dataset/SKILL.md +12 -0
  10. mainframe_toolkit/_workspace/AGENTS.md +10 -0
  11. mainframe_toolkit/_workspace/CLAUDE.md +2 -0
  12. mainframe_toolkit/_workspace/validator-java/.mvn/wrapper/maven-wrapper.properties +3 -0
  13. mainframe_toolkit/_workspace/validator-java/mvnw +295 -0
  14. mainframe_toolkit/_workspace/validator-java/mvnw.cmd +189 -0
  15. mainframe_toolkit/_workspace/validator-java/pom.xml +116 -0
  16. mainframe_toolkit/_workspace/validator-java/src/main/java/io/mainframe/migration/validator/AvroValueFormatter.java +152 -0
  17. mainframe_toolkit/_workspace/validator-java/src/main/java/io/mainframe/migration/validator/CsvTabularReader.java +111 -0
  18. mainframe_toolkit/_workspace/validator-java/src/main/java/io/mainframe/migration/validator/DataFormat.java +62 -0
  19. mainframe_toolkit/_workspace/validator-java/src/main/java/io/mainframe/migration/validator/Difference.java +34 -0
  20. mainframe_toolkit/_workspace/validator-java/src/main/java/io/mainframe/migration/validator/InputFileSet.java +45 -0
  21. mainframe_toolkit/_workspace/validator-java/src/main/java/io/mainframe/migration/validator/InputOptions.java +23 -0
  22. mainframe_toolkit/_workspace/validator-java/src/main/java/io/mainframe/migration/validator/JsonReportWriter.java +40 -0
  23. mainframe_toolkit/_workspace/validator-java/src/main/java/io/mainframe/migration/validator/MultiFileTabularReader.java +80 -0
  24. mainframe_toolkit/_workspace/validator-java/src/main/java/io/mainframe/migration/validator/Normalization.java +16 -0
  25. mainframe_toolkit/_workspace/validator-java/src/main/java/io/mainframe/migration/validator/ParquetTabularReader.java +61 -0
  26. mainframe_toolkit/_workspace/validator-java/src/main/java/io/mainframe/migration/validator/TabularReader.java +13 -0
  27. mainframe_toolkit/_workspace/validator-java/src/main/java/io/mainframe/migration/validator/TabularReaderFactory.java +29 -0
  28. mainframe_toolkit/_workspace/validator-java/src/main/java/io/mainframe/migration/validator/ValidationOptions.java +40 -0
  29. mainframe_toolkit/_workspace/validator-java/src/main/java/io/mainframe/migration/validator/ValidationReport.java +44 -0
  30. mainframe_toolkit/_workspace/validator-java/src/main/java/io/mainframe/migration/validator/ValidationService.java +395 -0
  31. mainframe_toolkit/_workspace/validator-java/src/main/java/io/mainframe/migration/validator/ValidatorCli.java +190 -0
  32. mainframe_toolkit/_workspace/validator-java/src/main/java/io/mainframe/migration/validator/ValueNormalizer.java +27 -0
  33. mainframe_toolkit/_workspace/validator-java/src/test/java/io/mainframe/migration/validator/DirectoryValidationTest.java +70 -0
  34. mainframe_toolkit/_workspace/validator-java/src/test/java/io/mainframe/migration/validator/JsonReportWriterTest.java +43 -0
  35. mainframe_toolkit/_workspace/validator-java/src/test/java/io/mainframe/migration/validator/KeyedValidationTest.java +89 -0
  36. mainframe_toolkit/_workspace/validator-java/src/test/java/io/mainframe/migration/validator/ParquetValidationTest.java +146 -0
  37. mainframe_toolkit/_workspace/validator-java/src/test/java/io/mainframe/migration/validator/ValidationServiceTest.java +137 -0
  38. mainframe_toolkit/_workspace/validator-java/src/test/java/io/mainframe/migration/validator/ValidatorCliTest.java +102 -0
  39. mainframe_toolkit/cli.py +361 -0
  40. mainframe_toolkit/cobol.py +106 -0
  41. mainframe_toolkit/copybook.py +558 -0
  42. mainframe_toolkit/errors.py +15 -0
  43. mainframe_toolkit/external.py +48 -0
  44. mainframe_toolkit/io.py +202 -0
  45. mainframe_toolkit/jcl.py +126 -0
  46. mainframe_toolkit/pipeline.py +322 -0
  47. mainframe_toolkit/sequential.py +259 -0
  48. mainframe_toolkit/similarity.py +1171 -0
  49. mainframe_toolkit/sorting.py +60 -0
  50. mainframe_toolkit/specs.py +312 -0
  51. mainframe_toolkit/synthetic.py +108 -0
  52. mainframe_toolkit/workspace.py +133 -0
@@ -0,0 +1,558 @@
1
+ from __future__ import annotations
2
+
3
+ import json
4
+ import re
5
+ from dataclasses import asdict, dataclass, field
6
+ from pathlib import Path
7
+ from typing import Any
8
+
9
+ from .sequential import FieldSpec, RecordLayout
10
+
11
+
12
+ class CopybookParseError(ValueError):
13
+ pass
14
+
15
+
16
+ @dataclass
17
+ class _Entry:
18
+ level: int
19
+ name: str
20
+ body: str
21
+ picture: str | None
22
+ usage: str | None
23
+ occurs: int
24
+ redefines: str | None
25
+ separate_sign: bool
26
+ children: list["_Entry"] = field(default_factory=list)
27
+
28
+
29
+ @dataclass(frozen=True)
30
+ class _Picture:
31
+ digits: int
32
+ characters: int
33
+ scale: int
34
+ signed: bool
35
+ textual: bool
36
+
37
+
38
+ _ENTRY_RE = re.compile(
39
+ r"(?ims)^[ \t]*(?P<level>\d{1,2})[ \t]+"
40
+ r"(?P<name>[A-Z0-9][A-Z0-9_-]*)\b(?P<body>.*?)"
41
+ r"(?=^[ \t]*\d{1,2}[ \t]+[A-Z0-9]|\Z)"
42
+ )
43
+ _PICTURE_RE = re.compile(
44
+ r"\bPIC(?:TURE)?(?:\s+IS)?\s+"
45
+ r"(?P<picture>(?:[AX9SVP](?:\(\s*\d+\s*\))?)+)",
46
+ re.IGNORECASE,
47
+ )
48
+ _PICTURE_TOKEN_RE = re.compile(r"([AX9SVP])(?:\(\s*(\d+)\s*\))?", re.IGNORECASE)
49
+ _USAGE_RE = re.compile(
50
+ r"\b(?:USAGE\s+(?:IS\s+)?)?"
51
+ r"(PACKED-DECIMAL|COMPUTATIONAL-[1-5]|COMP-[1-5]|"
52
+ r"COMPUTATIONAL|COMP|BINARY|DISPLAY)\b",
53
+ re.IGNORECASE,
54
+ )
55
+
56
+
57
+ def parse_copybook(
58
+ source: str | Path,
59
+ *,
60
+ source_format: str = "auto",
61
+ record_name: str | None = None,
62
+ encoding: str = "utf-8",
63
+ source_encoding: str = "utf-8",
64
+ include_fillers: bool = True,
65
+ ) -> RecordLayout:
66
+ text = _read_source(source, source_encoding)
67
+ return parse_copybook_text(
68
+ text,
69
+ source_format=source_format,
70
+ record_name=record_name,
71
+ encoding=encoding,
72
+ include_fillers=include_fillers,
73
+ )
74
+
75
+
76
+ def parse_copybook_file(
77
+ path: str | Path,
78
+ **options: Any,
79
+ ) -> RecordLayout:
80
+ return parse_copybook(Path(path), **options)
81
+
82
+
83
+ def parse_copybook_text(
84
+ text: str,
85
+ *,
86
+ source_format: str = "auto",
87
+ record_name: str | None = None,
88
+ encoding: str = "utf-8",
89
+ include_fillers: bool = True,
90
+ ) -> RecordLayout:
91
+ entries = _parse_entries(_normalise_source(text, source_format))
92
+ root = _select_root(entries, record_name)
93
+ fields, record_length = _place_node(root, 0, None, ())
94
+ fields = _unique_fields(fields)
95
+ if not include_fillers:
96
+ fields = [item for item in fields if not _is_filler(item.name)]
97
+ return RecordLayout(tuple(fields), record_length=record_length, encoding=encoding)
98
+
99
+
100
+ def layout_to_dict(layout: RecordLayout) -> dict[str, Any]:
101
+ return {
102
+ "fields": [asdict(item) for item in layout.fields],
103
+ "record_length": layout.record_length,
104
+ "encoding": layout.encoding,
105
+ }
106
+
107
+
108
+ def layout_to_json(
109
+ layout: RecordLayout,
110
+ path: str | Path | None = None,
111
+ *,
112
+ indent: int | None = 2,
113
+ ) -> str:
114
+ rendered = json.dumps(layout_to_dict(layout), ensure_ascii=False, indent=indent) + "\n"
115
+ if path is not None:
116
+ destination = Path(path)
117
+ destination.parent.mkdir(parents=True, exist_ok=True)
118
+ destination.write_text(rendered, encoding="utf-8")
119
+ return rendered
120
+
121
+
122
+ def write_layout_json(layout: RecordLayout, path: str | Path) -> None:
123
+ layout_to_json(layout, path)
124
+
125
+
126
+ def _read_source(source: str | Path, encoding: str) -> str:
127
+ if isinstance(source, Path):
128
+ return source.read_text(encoding=encoding)
129
+ if "\n" in source or "\r" in source:
130
+ return source
131
+ try:
132
+ path = Path(source)
133
+ if path.is_file():
134
+ return path.read_text(encoding=encoding)
135
+ except OSError:
136
+ pass
137
+ return source
138
+
139
+
140
+ def _normalise_source(text: str, source_format: str) -> str:
141
+ selected = source_format.lower().replace("-", "_")
142
+ if selected not in {"auto", "fixed", "free"}:
143
+ raise ValueError("source_format must be auto, fixed, or free")
144
+ lines = text.expandtabs(8).splitlines()
145
+ directive = _source_directive(lines)
146
+ if selected == "auto":
147
+ selected = directive or _detect_source_format(lines)
148
+
149
+ cleaned: list[str] = []
150
+ for raw in lines:
151
+ if re.search(r">>\s*SOURCE\s+FORMAT", raw, re.IGNORECASE):
152
+ continue
153
+ if selected == "fixed":
154
+ padded = raw.ljust(7)
155
+ indicator = padded[6]
156
+ if indicator in {"*", "/", "D", "d"}:
157
+ continue
158
+ content = padded[7:72]
159
+ content = _strip_inline_comment(content)
160
+ if indicator == "-" and cleaned:
161
+ cleaned[-1] = f"{cleaned[-1].rstrip()} {content.lstrip()}"
162
+ continue
163
+ else:
164
+ content = _strip_inline_comment(raw)
165
+ if content.lstrip().startswith(("*", "/")):
166
+ continue
167
+ if content.strip():
168
+ cleaned.append(content.rstrip())
169
+ return _break_inline_entries("\n".join(cleaned))
170
+
171
+
172
+ def _source_directive(lines: list[str]) -> str | None:
173
+ for line in lines:
174
+ match = re.search(
175
+ r">>\s*SOURCE\s+FORMAT(?:\s+IS)?\s+(FREE|FIXED)",
176
+ line,
177
+ re.IGNORECASE,
178
+ )
179
+ if match:
180
+ return match.group(1).lower()
181
+ return None
182
+
183
+
184
+ def _detect_source_format(lines: list[str]) -> str:
185
+ if any(re.match(r"^\d{6}.", line) for line in lines):
186
+ return "fixed"
187
+ early_entry = re.compile(r"^\s{0,6}\d{1,2}\s+[A-Z0-9]", re.IGNORECASE)
188
+ if any(early_entry.match(line) for line in lines):
189
+ return "free"
190
+ if any(len(line) > 6 and line[6] in {"*", "/", "-"} for line in lines):
191
+ return "fixed"
192
+ if any(len(line) > 7 and not line[:7].strip() for line in lines):
193
+ return "fixed"
194
+ return "free"
195
+
196
+
197
+ def _strip_inline_comment(line: str) -> str:
198
+ quote: str | None = None
199
+ index = 0
200
+ while index < len(line) - 1:
201
+ character = line[index]
202
+ if character in {"'", '"'}:
203
+ if quote == character and index + 1 < len(line) and line[index + 1] == character:
204
+ index += 2
205
+ continue
206
+ quote = None if quote == character else character if quote is None else quote
207
+ if quote is None and line[index : index + 2] == "*>":
208
+ return line[:index]
209
+ index += 1
210
+ return line
211
+
212
+
213
+ def _break_inline_entries(source: str) -> str:
214
+ next_entry = re.compile(
215
+ r"\s+(?:0?[1-9]|[1-4]\d|66|77|78|88)\s+[A-Z0-9]",
216
+ re.IGNORECASE,
217
+ )
218
+ result: list[str] = []
219
+ quote: str | None = None
220
+ index = 0
221
+ while index < len(source):
222
+ character = source[index]
223
+ if character in {"'", '"'}:
224
+ if quote == character and index + 1 < len(source) and source[index + 1] == character:
225
+ result.extend((character, character))
226
+ index += 2
227
+ continue
228
+ quote = None if quote == character else character if quote is None else quote
229
+ result.append(character)
230
+ if character == "." and quote is None:
231
+ match = next_entry.match(source, index + 1)
232
+ if match:
233
+ result.append("\n")
234
+ index += 1
235
+ return "".join(result)
236
+
237
+
238
+ def _parse_entries(source: str) -> list[_Entry]:
239
+ roots: list[_Entry] = []
240
+ stack: list[_Entry] = []
241
+ filler_count = 0
242
+ for match in _ENTRY_RE.finditer(source.upper()):
243
+ level = int(match.group("level"))
244
+ if level in {66, 78, 88}:
245
+ continue
246
+ if level == 0 or level > 49 and level != 77:
247
+ continue
248
+ name = match.group("name")
249
+ if name == "FILLER":
250
+ filler_count += 1
251
+ name = "FILLER" if filler_count == 1 else f"FILLER-{filler_count}"
252
+ body = " ".join(match.group("body").replace(".", " ").split())
253
+ picture_match = _PICTURE_RE.search(body)
254
+ picture = None
255
+ if picture_match:
256
+ picture = re.sub(r"\s+", "", picture_match.group("picture"))
257
+ usage_match = _USAGE_RE.search(body)
258
+ usage = _normalise_usage(usage_match.group(1)) if usage_match else None
259
+ occurs = _fixed_occurs(body)
260
+ redefines_match = re.search(r"\bREDEFINES\s+([A-Z0-9][A-Z0-9_-]*)\b", body)
261
+ entry = _Entry(
262
+ level=level,
263
+ name=name,
264
+ body=body,
265
+ picture=picture,
266
+ usage=usage,
267
+ occurs=occurs,
268
+ redefines=redefines_match.group(1) if redefines_match else None,
269
+ separate_sign=bool(re.search(r"\bSIGN\b.*?\bSEPARATE\b", body)),
270
+ )
271
+
272
+ if level == 77:
273
+ stack.clear()
274
+ else:
275
+ while stack and level <= stack[-1].level:
276
+ stack.pop()
277
+ if stack:
278
+ if stack[-1].picture or stack[-1].usage in {"comp_1", "comp_2"}:
279
+ raise CopybookParseError(f"elementary item {stack[-1].name} cannot contain {name}")
280
+ stack[-1].children.append(entry)
281
+ else:
282
+ roots.append(entry)
283
+ stack.append(entry)
284
+
285
+ if not roots:
286
+ raise CopybookParseError("copybook contains no record entries")
287
+ return roots
288
+
289
+
290
+ def _normalise_usage(value: str) -> str:
291
+ usage = value.upper().replace("COMPUTATIONAL", "COMP")
292
+ return {
293
+ "DISPLAY": "display",
294
+ "PACKED-DECIMAL": "packed_decimal",
295
+ "COMP-3": "packed_decimal",
296
+ "BINARY": "binary",
297
+ "COMP": "binary",
298
+ "COMP-4": "binary",
299
+ "COMP-5": "binary",
300
+ "COMP-1": "comp_1",
301
+ "COMP-2": "comp_2",
302
+ }[usage]
303
+
304
+
305
+ def _fixed_occurs(body: str) -> int:
306
+ match = re.search(r"\bOCCURS\s+(\d+)\b", body)
307
+ if not match:
308
+ return 1
309
+ tail = body[match.end() :]
310
+ if re.match(r"\s+TO\s+\d+", tail) or re.search(r"\bDEPENDING\s+ON\b", tail):
311
+ raise CopybookParseError("variable OCCURS is not supported")
312
+ count = int(match.group(1))
313
+ if count < 1:
314
+ raise CopybookParseError("OCCURS must be greater than zero")
315
+ return count
316
+
317
+
318
+ def _select_root(entries: list[_Entry], record_name: str | None) -> _Entry:
319
+ if record_name is None:
320
+ return entries[0]
321
+ wanted = record_name.upper().replace("_", "-")
322
+ pending = list(entries)
323
+ while pending:
324
+ entry = pending.pop(0)
325
+ if entry.name.replace("_", "-") == wanted:
326
+ return entry
327
+ pending[0:0] = entry.children
328
+ raise CopybookParseError(f"record {record_name!r} was not found")
329
+
330
+
331
+ def _place_node(
332
+ entry: _Entry,
333
+ start: int,
334
+ inherited_usage: str | None,
335
+ occurrence_path: tuple[int, ...],
336
+ ) -> tuple[list[FieldSpec], int]:
337
+ usage = entry.usage or inherited_usage
338
+ if entry.children:
339
+ if entry.picture is not None or usage in {"comp_1", "comp_2"}:
340
+ raise CopybookParseError(f"invalid group item {entry.name}")
341
+ all_fields: list[FieldSpec] = []
342
+ one_length: int | None = None
343
+ for occurrence in range(1, entry.occurs + 1):
344
+ suffix = occurrence_path + ((occurrence,) if entry.occurs > 1 else ())
345
+ base = start + (one_length or 0) * (occurrence - 1)
346
+ fields, length = _place_children(entry.children, base, usage, suffix)
347
+ if one_length is None:
348
+ one_length = length
349
+ elif length != one_length:
350
+ raise CopybookParseError(f"non-deterministic length for {entry.name}")
351
+ all_fields.extend(fields)
352
+ return all_fields, (one_length or 0) * entry.occurs
353
+
354
+ if usage in {"comp_1", "comp_2"} and entry.picture is None:
355
+ length = 4 if usage == "comp_1" else 8
356
+ fields = []
357
+ for occurrence in range(1, entry.occurs + 1):
358
+ suffix = occurrence_path + ((occurrence,) if entry.occurs > 1 else ())
359
+ fields.append(
360
+ _field(
361
+ entry,
362
+ start + (occurrence - 1) * length,
363
+ length,
364
+ "bytes",
365
+ 0,
366
+ True,
367
+ suffix,
368
+ "latin-1",
369
+ )
370
+ )
371
+ return fields, length * entry.occurs
372
+ if entry.picture is None:
373
+ raise CopybookParseError(f"group item {entry.name} has no subordinate entries")
374
+
375
+ picture = _parse_picture(entry.picture)
376
+ selected_usage = usage or "display"
377
+ length, kind, field_encoding = _physical_type(picture, selected_usage, entry.separate_sign)
378
+ all_fields = []
379
+ for occurrence in range(1, entry.occurs + 1):
380
+ suffix = occurrence_path + ((occurrence,) if entry.occurs > 1 else ())
381
+ all_fields.append(
382
+ _field(
383
+ entry,
384
+ start + (occurrence - 1) * length,
385
+ length,
386
+ kind,
387
+ picture.scale,
388
+ picture.signed,
389
+ suffix,
390
+ field_encoding,
391
+ )
392
+ )
393
+ return all_fields, length * entry.occurs
394
+
395
+
396
+ def _place_children(
397
+ children: list[_Entry],
398
+ start: int,
399
+ inherited_usage: str | None,
400
+ occurrence_path: tuple[int, ...],
401
+ ) -> tuple[list[FieldSpec], int]:
402
+ cursor = start
403
+ extent = start
404
+ positions: dict[str, tuple[int, int]] = {}
405
+ fields: list[FieldSpec] = []
406
+ for child in children:
407
+ child_start = cursor
408
+ if child.redefines:
409
+ target = positions.get(child.redefines)
410
+ if target is None:
411
+ raise CopybookParseError(
412
+ f"{child.name} redefines unknown item {child.redefines}"
413
+ )
414
+ child_start = target[0]
415
+ child_fields, child_length = _place_node(
416
+ child, child_start, inherited_usage, occurrence_path
417
+ )
418
+ fields.extend(child_fields)
419
+ positions[child.name] = (child_start, child_length)
420
+ extent = max(extent, child_start + child_length)
421
+ cursor = max(cursor, child_start + child_length)
422
+ return fields, extent - start
423
+
424
+
425
+ def _parse_picture(value: str) -> _Picture:
426
+ position = 0
427
+ tokens: list[tuple[str, int]] = []
428
+ for match in _PICTURE_TOKEN_RE.finditer(value):
429
+ if match.start() != position:
430
+ raise CopybookParseError(f"unsupported PICTURE {value!r}")
431
+ count = int(match.group(2) or 1)
432
+ if count < 1:
433
+ raise CopybookParseError(f"invalid repetition in PICTURE {value!r}")
434
+ tokens.append((match.group(1).upper(), count))
435
+ position = match.end()
436
+ if position != len(value) or not tokens:
437
+ raise CopybookParseError(f"unsupported PICTURE {value!r}")
438
+
439
+ digits = sum(count for symbol, count in tokens if symbol == "9")
440
+ characters = sum(count for symbol, count in tokens if symbol in {"A", "X", "9"})
441
+ textual = any(symbol in {"A", "X"} for symbol, _ in tokens)
442
+ signed = any(symbol == "S" for symbol, _ in tokens)
443
+ if textual:
444
+ if any(symbol in {"S", "V", "P"} for symbol, _ in tokens):
445
+ raise CopybookParseError(f"mixed numeric/alphanumeric PICTURE {value!r}")
446
+ return _Picture(digits, characters, 0, False, True)
447
+ if digits == 0:
448
+ raise CopybookParseError(f"numeric PICTURE {value!r} contains no digits")
449
+
450
+ symbols = [symbol for symbol, _ in tokens]
451
+ first_digit = next(index for index, item in enumerate(tokens) if item[0] == "9")
452
+ last_digit = len(tokens) - 1 - next(
453
+ index for index, item in enumerate(reversed(tokens)) if item[0] == "9"
454
+ )
455
+ leading_p = sum(count for symbol, count in tokens[:first_digit] if symbol == "P")
456
+ trailing_p = sum(count for symbol, count in tokens[last_digit + 1 :] if symbol == "P")
457
+ if leading_p:
458
+ scale = digits + leading_p
459
+ elif trailing_p:
460
+ scale = -trailing_p
461
+ elif "V" in symbols:
462
+ v_index = symbols.index("V")
463
+ scale = sum(
464
+ count for symbol, count in tokens[v_index + 1 :] if symbol in {"9", "P"}
465
+ )
466
+ else:
467
+ scale = 0
468
+ return _Picture(digits, characters, scale, signed, False)
469
+
470
+
471
+ def _physical_type(
472
+ picture: _Picture,
473
+ usage: str,
474
+ separate_sign: bool,
475
+ ) -> tuple[int, str, str | None]:
476
+ if picture.textual:
477
+ if usage != "display":
478
+ raise CopybookParseError(f"usage {usage} is invalid for an alphanumeric item")
479
+ return picture.characters, "text", None
480
+ if usage == "display":
481
+ length = picture.digits + (1 if picture.signed and separate_sign else 0)
482
+ if picture.signed and not separate_sign:
483
+ return length, "zoned_decimal", None
484
+ return length, "decimal" if picture.scale else "integer", None
485
+ if usage == "packed_decimal":
486
+ return (picture.digits + 2) // 2, "packed_decimal", None
487
+ if usage == "binary":
488
+ if picture.digits <= 4:
489
+ length = 2
490
+ elif picture.digits <= 9:
491
+ length = 4
492
+ elif picture.digits <= 18:
493
+ length = 8
494
+ else:
495
+ raise CopybookParseError("binary PICTURE exceeds 18 stored digits")
496
+ return length, "binary", None
497
+ if usage == "comp_1":
498
+ return 4, "bytes", "latin-1"
499
+ if usage == "comp_2":
500
+ return 8, "bytes", "latin-1"
501
+ raise CopybookParseError(f"unsupported usage {usage!r}")
502
+
503
+
504
+ def _field(
505
+ entry: _Entry,
506
+ start: int,
507
+ length: int,
508
+ kind: str,
509
+ scale: int,
510
+ signed: bool,
511
+ occurrence_path: tuple[int, ...],
512
+ encoding: str | None,
513
+ ) -> FieldSpec:
514
+ suffix = "".join(f"[{index}]" for index in occurrence_path)
515
+ return FieldSpec(
516
+ name=f"{entry.name}{suffix}",
517
+ start=start + 1,
518
+ length=length,
519
+ kind=kind,
520
+ scale=scale,
521
+ encoding=encoding,
522
+ signed=signed,
523
+ )
524
+
525
+
526
+ def _unique_fields(fields: list[FieldSpec]) -> list[FieldSpec]:
527
+ counts: dict[str, int] = {}
528
+ used: set[str] = set()
529
+ result: list[FieldSpec] = []
530
+ for item in fields:
531
+ base = item.name
532
+ counts[base] = counts.get(base, 0) + 1
533
+ name = base if counts[base] == 1 else f"{base}#{counts[base]}"
534
+ while name in used:
535
+ counts[base] += 1
536
+ name = f"{base}#{counts[base]}"
537
+ used.add(name)
538
+ result.append(
539
+ item
540
+ if name == item.name
541
+ else FieldSpec(**{**asdict(item), "name": name})
542
+ )
543
+ return result
544
+
545
+
546
+ def _is_filler(name: str) -> bool:
547
+ return bool(re.match(r"^FILLER(?:-\d+)?(?:\[|#|$)", name))
548
+
549
+
550
+ __all__ = [
551
+ "CopybookParseError",
552
+ "layout_to_dict",
553
+ "layout_to_json",
554
+ "parse_copybook",
555
+ "parse_copybook_file",
556
+ "parse_copybook_text",
557
+ "write_layout_json",
558
+ ]
@@ -0,0 +1,15 @@
1
+ class ToolkitError(Exception):
2
+ """Base error raised by the toolkit."""
3
+
4
+
5
+ class OptionalDependencyError(ToolkitError):
6
+ """Raised when an operation needs an optional package."""
7
+
8
+
9
+ class DataFormatError(ToolkitError):
10
+ """Raised when a source record cannot be decoded safely."""
11
+
12
+
13
+ class SpecificationError(ToolkitError):
14
+ """Raised when the migration project contract is invalid."""
15
+
@@ -0,0 +1,48 @@
1
+ from __future__ import annotations
2
+
3
+ import json
4
+ from pathlib import Path
5
+ from typing import Any, Callable, Mapping
6
+
7
+
8
+ ExternalFunction = Callable[..., Any]
9
+
10
+
11
+ class ExternalProgramRegistry:
12
+ def __init__(self, specifications: Mapping[str, str] | None = None):
13
+ self.specifications = {name.upper(): description for name, description in (specifications or {}).items()}
14
+ self._implementations: dict[str, ExternalFunction] = {}
15
+
16
+ @classmethod
17
+ def from_directory(cls, directory: str | Path) -> "ExternalProgramRegistry":
18
+ root = Path(directory)
19
+ combined: dict[str, str] = {}
20
+ for path in sorted(root.glob("*.json")):
21
+ with path.open(encoding="utf-8") as stream:
22
+ value = json.load(stream)
23
+ if not isinstance(value, dict) or not all(isinstance(item, str) for item in value.values()):
24
+ raise ValueError(f"{path} must contain a name-to-description object")
25
+ combined.update({str(name): description for name, description in value.items()})
26
+ return cls(combined)
27
+
28
+ def register(self, name: str, function: ExternalFunction) -> None:
29
+ self._implementations[name.upper()] = function
30
+
31
+ def implementation(self, name: str) -> ExternalFunction:
32
+ key = name.upper()
33
+ if key not in self._implementations:
34
+ description = self.specifications.get(key, "specification unavailable")
35
+ raise LookupError(f"external program {name} has no adapter: {description}")
36
+ return self._implementations[key]
37
+
38
+ def call(self, name: str, *args: Any, **kwargs: Any) -> Any:
39
+ return self.implementation(name)(*args, **kwargs)
40
+
41
+
42
+ def external_program(name: str, registry: ExternalProgramRegistry) -> Callable[[ExternalFunction], ExternalFunction]:
43
+ def decorator(function: ExternalFunction) -> ExternalFunction:
44
+ registry.register(name, function)
45
+ return function
46
+
47
+ return decorator
48
+