mainframe-migration-toolkit 0.2.0__tar.gz → 0.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/.claude/skills/analyze-mainframe-similarity/SKILL.md +3 -1
- {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/PKG-INFO +1 -1
- {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/pyproject.toml +1 -1
- {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/src/mainframe_toolkit/__init__.py +1 -1
- {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/src/mainframe_toolkit/cli.py +8 -0
- {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/src/mainframe_toolkit/similarity.py +284 -27
- {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/.claude/skills/migrate-mainframe-job/SKILL.md +0 -0
- {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/.claude/skills/validate-golden-dataset/SKILL.md +0 -0
- {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/.gitignore +0 -0
- {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/AGENTS.md +0 -0
- {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/CLAUDE.md +0 -0
- {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/src/mainframe_toolkit/__main__.py +0 -0
- {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/src/mainframe_toolkit/cobol.py +0 -0
- {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/src/mainframe_toolkit/copybook.py +0 -0
- {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/src/mainframe_toolkit/errors.py +0 -0
- {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/src/mainframe_toolkit/external.py +0 -0
- {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/src/mainframe_toolkit/io.py +0 -0
- {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/src/mainframe_toolkit/jcl.py +0 -0
- {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/src/mainframe_toolkit/pipeline.py +0 -0
- {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/src/mainframe_toolkit/sequential.py +0 -0
- {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/src/mainframe_toolkit/sorting.py +0 -0
- {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/src/mainframe_toolkit/specs.py +0 -0
- {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/src/mainframe_toolkit/synthetic.py +0 -0
- {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/src/mainframe_toolkit/workspace.py +0 -0
- {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/validator-java/.mvn/wrapper/maven-wrapper.properties +0 -0
- {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/validator-java/mvnw +0 -0
- {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/validator-java/mvnw.cmd +0 -0
- {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/validator-java/pom.xml +0 -0
- {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/validator-java/src/main/java/io/mainframe/migration/validator/AvroValueFormatter.java +0 -0
- {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/validator-java/src/main/java/io/mainframe/migration/validator/CsvTabularReader.java +0 -0
- {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/validator-java/src/main/java/io/mainframe/migration/validator/DataFormat.java +0 -0
- {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/validator-java/src/main/java/io/mainframe/migration/validator/Difference.java +0 -0
- {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/validator-java/src/main/java/io/mainframe/migration/validator/InputFileSet.java +0 -0
- {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/validator-java/src/main/java/io/mainframe/migration/validator/InputOptions.java +0 -0
- {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/validator-java/src/main/java/io/mainframe/migration/validator/JsonReportWriter.java +0 -0
- {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/validator-java/src/main/java/io/mainframe/migration/validator/MultiFileTabularReader.java +0 -0
- {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/validator-java/src/main/java/io/mainframe/migration/validator/Normalization.java +0 -0
- {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/validator-java/src/main/java/io/mainframe/migration/validator/ParquetTabularReader.java +0 -0
- {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/validator-java/src/main/java/io/mainframe/migration/validator/TabularReader.java +0 -0
- {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/validator-java/src/main/java/io/mainframe/migration/validator/TabularReaderFactory.java +0 -0
- {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/validator-java/src/main/java/io/mainframe/migration/validator/ValidationOptions.java +0 -0
- {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/validator-java/src/main/java/io/mainframe/migration/validator/ValidationReport.java +0 -0
- {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/validator-java/src/main/java/io/mainframe/migration/validator/ValidationService.java +0 -0
- {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/validator-java/src/main/java/io/mainframe/migration/validator/ValidatorCli.java +0 -0
- {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/validator-java/src/main/java/io/mainframe/migration/validator/ValueNormalizer.java +0 -0
- {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/validator-java/src/test/java/io/mainframe/migration/validator/DirectoryValidationTest.java +0 -0
- {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/validator-java/src/test/java/io/mainframe/migration/validator/JsonReportWriterTest.java +0 -0
- {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/validator-java/src/test/java/io/mainframe/migration/validator/KeyedValidationTest.java +0 -0
- {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/validator-java/src/test/java/io/mainframe/migration/validator/ParquetValidationTest.java +0 -0
- {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/validator-java/src/test/java/io/mainframe/migration/validator/ValidationServiceTest.java +0 -0
- {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/validator-java/src/test/java/io/mainframe/migration/validator/ValidatorCliTest.java +0 -0
|
@@ -8,11 +8,13 @@ description: Cluster COBOL and JCL source by deterministic textual similarity, i
|
|
|
8
8
|
1. Run the deterministic scan:
|
|
9
9
|
|
|
10
10
|
```text
|
|
11
|
-
python -m mainframe_toolkit similarity scan <source-root> --threshold <threshold> --json similarity.json --csv similarity.csv --review-request similarity-review-request.json
|
|
11
|
+
python -m mainframe_toolkit similarity scan <source-root> --algorithm advanced --threshold <threshold> --json similarity.json --csv similarity.csv --review-request similarity-review-request.json
|
|
12
12
|
```
|
|
13
13
|
|
|
14
14
|
Use the requested threshold, or 80 when absent. Never edit deterministic scores or cluster membership manually.
|
|
15
15
|
|
|
16
|
+
Inspect each pair's `componentes` (`conteudo`, `ordem`, `renomeacao`, `estrutura`, `literais`, and `clone_parcial`). Also inspect `clones_parciais`, which are evidence of shared blocks but do not automatically join clusters. Use `--cluster-policy cohesive` when every pair in a cluster must meet the threshold; keep `connected` for broader codebase discovery.
|
|
17
|
+
|
|
16
18
|
2. For each non-singleton cluster in the review request, read every original file and its normalized evidence. Check all pair relationships; connected-component chaining alone is not proof of equivalence. Ignore comments, sequence columns, copyright blocks, whitespace, generated names, and environment-only identifiers. Compare JCL execution flow/DD semantics or COBOL inputs, outputs, branches, calculations, side effects, error behavior, and record layouts.
|
|
17
19
|
|
|
18
20
|
3. Assign exactly one verdict per requested cluster:
|
|
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "mainframe-migration-toolkit"
|
|
7
|
-
version = "0.
|
|
7
|
+
version = "0.3.0"
|
|
8
8
|
description = "Runtime and deterministic tools for COBOL/JCL to PySpark migrations"
|
|
9
9
|
requires-python = ">=3.10"
|
|
10
10
|
authors = [{ name = "Mainframe Migration Toolkit" }]
|
{mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/src/mainframe_toolkit/cli.py
RENAMED
|
@@ -84,6 +84,10 @@ def _similarity_scan(args: argparse.Namespace) -> int:
|
|
|
84
84
|
threshold=args.threshold,
|
|
85
85
|
shingle_size=args.shingle_size,
|
|
86
86
|
exhaustive_limit=args.exhaustive_limit,
|
|
87
|
+
algorithm=args.algorithm,
|
|
88
|
+
partial_threshold=args.partial_threshold,
|
|
89
|
+
min_clone_tokens=args.min_clone_tokens,
|
|
90
|
+
cluster_policy=args.cluster_policy,
|
|
87
91
|
)
|
|
88
92
|
write_results(report, args.json, args.csv)
|
|
89
93
|
write_ai_review_payload(report, args.review_request)
|
|
@@ -293,6 +297,10 @@ def _parser() -> argparse.ArgumentParser:
|
|
|
293
297
|
scan.add_argument("--threshold", type=float, default=80.0)
|
|
294
298
|
scan.add_argument("--shingle-size", type=int, default=3)
|
|
295
299
|
scan.add_argument("--exhaustive-limit", type=int, default=80)
|
|
300
|
+
scan.add_argument("--algorithm", choices=("advanced", "classic"), default="advanced")
|
|
301
|
+
scan.add_argument("--partial-threshold", type=float, default=75.0)
|
|
302
|
+
scan.add_argument("--min-clone-tokens", type=int, default=30)
|
|
303
|
+
scan.add_argument("--cluster-policy", choices=("cohesive", "connected"), default="connected")
|
|
296
304
|
scan.add_argument("--json", default="similarity.json")
|
|
297
305
|
scan.add_argument("--csv", default="similarity.csv")
|
|
298
306
|
scan.add_argument("--review-request", default="similarity-review-request.json")
|
|
@@ -81,6 +81,45 @@ _LSH_BANDS = 12
|
|
|
81
81
|
_MAX_LSH_BUCKET = 128
|
|
82
82
|
_MAX_REPORTED_PAIRS = 5_000
|
|
83
83
|
|
|
84
|
+
_COBOL_STRUCTURAL_WORDS = frozenset(
|
|
85
|
+
{
|
|
86
|
+
"ACCEPT", "ADD", "CALL", "CANCEL", "CLOSE", "COMPUTE", "CONTINUE",
|
|
87
|
+
"DELETE", "DISPLAY", "DIVIDE", "ELSE", "END-CALL", "END-COMPUTE",
|
|
88
|
+
"END-EVALUATE", "END-IF", "END-PERFORM", "END-READ", "END-SEARCH",
|
|
89
|
+
"END-STRING", "END-WRITE", "ENTRY", "EVALUATE", "EXIT", "GOBACK",
|
|
90
|
+
"GO", "IF", "INITIALIZE", "INSPECT", "MERGE", "MOVE", "MULTIPLY",
|
|
91
|
+
"OPEN", "PERFORM", "READ", "RELEASE", "RETURN", "REWRITE", "SEARCH",
|
|
92
|
+
"SET", "SORT", "START", "STOP", "STRING", "SUBTRACT", "UNSTRING",
|
|
93
|
+
"WHEN", "WRITE", "IDENTIFICATION", "ENVIRONMENT", "DATA", "PROCEDURE",
|
|
94
|
+
"DIVISION", "SECTION", "DECLARATIVES", "END-DECLARATIVES",
|
|
95
|
+
}
|
|
96
|
+
)
|
|
97
|
+
_COBOL_LANGUAGE_WORDS = _COBOL_STRUCTURAL_WORDS | frozenset(
|
|
98
|
+
{
|
|
99
|
+
"PROGRAM-ID", "FILE-CONTROL", "SELECT", "ASSIGN", "FD", "SD", "PIC",
|
|
100
|
+
"PICTURE", "VALUE", "VALUES", "OCCURS", "REDEFINES", "RENAMES", "USAGE",
|
|
101
|
+
"COMP", "COMP-1", "COMP-2", "COMP-3", "BINARY", "PACKED-DECIMAL",
|
|
102
|
+
"DISPLAY-1", "WORKING-STORAGE", "LINKAGE", "FILE", "INPUT-OUTPUT",
|
|
103
|
+
"USING", "GIVING", "THRU", "THROUGH", "UNTIL", "VARYING", "TIMES",
|
|
104
|
+
"FROM", "BY", "TO", "INTO", "OF", "IN", "ON", "AT", "END", "NOT",
|
|
105
|
+
"AND", "OR", "TRUE", "FALSE", "ZERO", "ZEROS", "ZEROES", "SPACE",
|
|
106
|
+
"SPACES", "HIGH-VALUES", "LOW-VALUES", "ALL", "CORRESPONDING", "ROUNDED",
|
|
107
|
+
"SIZE", "ERROR", "INVALID", "KEY", "WITH", "TEST", "AFTER", "BEFORE",
|
|
108
|
+
"ASCENDING", "DESCENDING", "DUPLICATES", "NO", "ADVANCING", "UPON",
|
|
109
|
+
}
|
|
110
|
+
)
|
|
111
|
+
_JCL_LANGUAGE_WORDS = _JCL_OPERATIONS | frozenset(
|
|
112
|
+
{
|
|
113
|
+
"PGM", "PROC", "PARM", "COND", "DSN", "DSNAME", "DISP", "UNIT",
|
|
114
|
+
"SPACE", "DCB", "SYSOUT", "CLASS", "MSGCLASS", "REGION", "TIME",
|
|
115
|
+
"RESTART", "SYMBOLS", "MEMBER", "ORDER", "INCLUDE", "SET", "IF",
|
|
116
|
+
"THEN", "ELSE", "ENDIF", "PASS", "KEEP", "CATLG", "DELETE", "SHR",
|
|
117
|
+
"OLD", "NEW", "MOD", "DUMMY", "DATA", "DLM",
|
|
118
|
+
}
|
|
119
|
+
)
|
|
120
|
+
_IDENTIFIER_RE = re.compile(r"[A-Z_$#@&][A-Z0-9_$#@&.-]*\Z")
|
|
121
|
+
_NUMBER_RE = re.compile(r"[+-]?(?:\d+(?:\.\d*)?|\.\d+)\Z")
|
|
122
|
+
|
|
84
123
|
|
|
85
124
|
@dataclass(frozen=True, slots=True)
|
|
86
125
|
class SourceDocument:
|
|
@@ -91,6 +130,9 @@ class SourceDocument:
|
|
|
91
130
|
content: str = field(repr=False)
|
|
92
131
|
normalized: str = field(repr=False)
|
|
93
132
|
tokens: tuple[str, ...] = field(repr=False)
|
|
133
|
+
abstract_tokens: tuple[str, ...] = field(repr=False)
|
|
134
|
+
structure_tokens: tuple[str, ...] = field(repr=False)
|
|
135
|
+
literal_tokens: tuple[str, ...] = field(repr=False)
|
|
94
136
|
digest: str
|
|
95
137
|
source_id: str
|
|
96
138
|
|
|
@@ -100,22 +142,30 @@ class SimilarityReport:
|
|
|
100
142
|
threshold: float
|
|
101
143
|
shingle_size: int
|
|
102
144
|
exhaustive_limit: int
|
|
145
|
+
algorithm: str
|
|
146
|
+
partial_threshold: float
|
|
147
|
+
cluster_policy: str
|
|
103
148
|
records: list[dict[str, Any]]
|
|
104
149
|
clusters: list[dict[str, Any]]
|
|
105
150
|
statistics: dict[str, Any]
|
|
151
|
+
partial_clones: list[dict[str, Any]] = field(default_factory=list)
|
|
106
152
|
documents: tuple[SourceDocument, ...] = field(default_factory=tuple, repr=False)
|
|
107
153
|
|
|
108
154
|
def to_dict(self) -> dict[str, Any]:
|
|
109
155
|
return {
|
|
110
|
-
"schema_version":
|
|
156
|
+
"schema_version": 2,
|
|
111
157
|
"configuracao": {
|
|
112
158
|
"limiar_percentual": self.threshold,
|
|
113
159
|
"tamanho_shingle": self.shingle_size,
|
|
114
160
|
"limite_comparacao_exaustiva": self.exhaustive_limit,
|
|
161
|
+
"algoritmo": self.algorithm,
|
|
162
|
+
"limiar_clone_parcial": self.partial_threshold,
|
|
163
|
+
"politica_cluster": self.cluster_policy,
|
|
115
164
|
},
|
|
116
165
|
"resumo": deepcopy(self.statistics),
|
|
117
166
|
"arquivos": deepcopy(self.records),
|
|
118
167
|
"clusters": deepcopy(self.clusters),
|
|
168
|
+
"clones_parciais": deepcopy(self.partial_clones),
|
|
119
169
|
}
|
|
120
170
|
|
|
121
171
|
def __getitem__(self, key: str) -> Any:
|
|
@@ -409,6 +459,38 @@ def tokenize_normalized(text: str) -> tuple[str, ...]:
|
|
|
409
459
|
return tuple(_TOKEN_RE.findall(text))
|
|
410
460
|
|
|
411
461
|
|
|
462
|
+
def _is_literal(token: str) -> bool:
|
|
463
|
+
return (
|
|
464
|
+
len(token) >= 2
|
|
465
|
+
and token[0] == token[-1]
|
|
466
|
+
and token[0] in {"'", '"'}
|
|
467
|
+
) or _NUMBER_RE.fullmatch(token) is not None
|
|
468
|
+
|
|
469
|
+
|
|
470
|
+
def _semantic_features(tokens: Sequence[str], tipo: str) -> tuple[tuple[str, ...], tuple[str, ...], tuple[str, ...]]:
|
|
471
|
+
language_words = _COBOL_LANGUAGE_WORDS if tipo == "cobol" else _JCL_LANGUAGE_WORDS
|
|
472
|
+
structural_words = _COBOL_STRUCTURAL_WORDS if tipo == "cobol" else _JCL_LANGUAGE_WORDS
|
|
473
|
+
identifiers: dict[str, str] = {}
|
|
474
|
+
abstract: list[str] = []
|
|
475
|
+
structure: list[str] = []
|
|
476
|
+
literals: list[str] = []
|
|
477
|
+
|
|
478
|
+
for token in tokens:
|
|
479
|
+
upper = token.upper()
|
|
480
|
+
if _is_literal(token):
|
|
481
|
+
literals.append(token)
|
|
482
|
+
abstract.append(token)
|
|
483
|
+
elif upper in language_words or not _IDENTIFIER_RE.fullmatch(upper):
|
|
484
|
+
abstract.append(upper)
|
|
485
|
+
else:
|
|
486
|
+
placeholder = identifiers.setdefault(upper, f"__ID{len(identifiers) + 1}__")
|
|
487
|
+
abstract.append(placeholder)
|
|
488
|
+
if upper in structural_words:
|
|
489
|
+
structure.append(upper)
|
|
490
|
+
|
|
491
|
+
return tuple(abstract), tuple(structure), tuple(literals)
|
|
492
|
+
|
|
493
|
+
|
|
412
494
|
def _stable_digest(value: str, *, size: int = 16) -> str:
|
|
413
495
|
return hashlib.blake2b(value.encode("utf-8"), digest_size=size).hexdigest()
|
|
414
496
|
|
|
@@ -423,6 +505,8 @@ def _make_document(
|
|
|
423
505
|
kind = _canonical_kind(tipo)
|
|
424
506
|
relative = Path(relative_path).as_posix()
|
|
425
507
|
normalized = normalize_source(content, kind)
|
|
508
|
+
tokens = tokenize_normalized(normalized)
|
|
509
|
+
abstract_tokens, structure_tokens, literal_tokens = _semantic_features(tokens, kind)
|
|
426
510
|
return SourceDocument(
|
|
427
511
|
tipo=kind,
|
|
428
512
|
nome=name,
|
|
@@ -430,7 +514,10 @@ def _make_document(
|
|
|
430
514
|
relative_path=relative,
|
|
431
515
|
content=content,
|
|
432
516
|
normalized=normalized,
|
|
433
|
-
tokens=
|
|
517
|
+
tokens=tokens,
|
|
518
|
+
abstract_tokens=abstract_tokens,
|
|
519
|
+
structure_tokens=structure_tokens,
|
|
520
|
+
literal_tokens=literal_tokens,
|
|
434
521
|
digest=_stable_digest(normalized),
|
|
435
522
|
source_id=_stable_digest(f"{kind}\0{relative}", size=10),
|
|
436
523
|
)
|
|
@@ -538,6 +625,34 @@ def _dice(left: Counter[Any], right: Counter[Any]) -> float:
|
|
|
538
625
|
return (2.0 * overlap) / total
|
|
539
626
|
|
|
540
627
|
|
|
628
|
+
def _containment(left: Counter[Any], right: Counter[Any]) -> float:
|
|
629
|
+
smallest = min(sum(left.values()), sum(right.values()))
|
|
630
|
+
if smallest == 0:
|
|
631
|
+
return 1.0 if not left and not right else 0.0
|
|
632
|
+
return sum((left & right).values()) / smallest
|
|
633
|
+
|
|
634
|
+
|
|
635
|
+
@dataclass(frozen=True, slots=True)
|
|
636
|
+
class PairScore:
|
|
637
|
+
percentage: float
|
|
638
|
+
content: float
|
|
639
|
+
order: float
|
|
640
|
+
renamed: float
|
|
641
|
+
structure: float
|
|
642
|
+
literals: float
|
|
643
|
+
partial: float
|
|
644
|
+
|
|
645
|
+
def to_dict(self) -> dict[str, float]:
|
|
646
|
+
return {
|
|
647
|
+
"conteudo": self.content,
|
|
648
|
+
"ordem": self.order,
|
|
649
|
+
"renomeacao": self.renamed,
|
|
650
|
+
"estrutura": self.structure,
|
|
651
|
+
"literais": self.literals,
|
|
652
|
+
"clone_parcial": self.partial,
|
|
653
|
+
}
|
|
654
|
+
|
|
655
|
+
|
|
541
656
|
def _similarity_tokens(
|
|
542
657
|
left: Sequence[str], right: Sequence[str], shingle_size: int
|
|
543
658
|
) -> float:
|
|
@@ -553,21 +668,73 @@ def _similarity_tokens(
|
|
|
553
668
|
return round(100.0 * ((0.25 * ordered) + (0.75 * lexical)), 2)
|
|
554
669
|
|
|
555
670
|
|
|
671
|
+
def _advanced_similarity(left: SourceDocument, right: SourceDocument, shingle_size: int) -> PairScore:
|
|
672
|
+
if left.tokens == right.tokens:
|
|
673
|
+
return PairScore(100.0, 100.0, 100.0, 100.0, 100.0, 100.0, 100.0)
|
|
674
|
+
|
|
675
|
+
width = max(1, min(shingle_size, len(left.tokens), len(right.tokens)))
|
|
676
|
+
left_words = _feature_counter(left.tokens, 1)
|
|
677
|
+
right_words = _feature_counter(right.tokens, 1)
|
|
678
|
+
left_ordered = _feature_counter(left.tokens, width)
|
|
679
|
+
right_ordered = _feature_counter(right.tokens, width)
|
|
680
|
+
content = 100.0 * _dice(left_words, right_words)
|
|
681
|
+
order = 100.0 * _dice(left_ordered, right_ordered)
|
|
682
|
+
renamed = 100.0 * _dice(
|
|
683
|
+
_feature_counter(left.abstract_tokens, width),
|
|
684
|
+
_feature_counter(right.abstract_tokens, width),
|
|
685
|
+
)
|
|
686
|
+
structure = 100.0 * _dice(
|
|
687
|
+
_feature_counter(left.structure_tokens, 2),
|
|
688
|
+
_feature_counter(right.structure_tokens, 2),
|
|
689
|
+
)
|
|
690
|
+
literals = 100.0 * _dice(
|
|
691
|
+
_feature_counter(left.literal_tokens, 1),
|
|
692
|
+
_feature_counter(right.literal_tokens, 1),
|
|
693
|
+
)
|
|
694
|
+
partial = 100.0 * max(
|
|
695
|
+
_containment(left_ordered, right_ordered),
|
|
696
|
+
_containment(
|
|
697
|
+
_feature_counter(left.abstract_tokens, width),
|
|
698
|
+
_feature_counter(right.abstract_tokens, width),
|
|
699
|
+
),
|
|
700
|
+
)
|
|
701
|
+
percentage = (
|
|
702
|
+
(0.15 * content)
|
|
703
|
+
+ (0.15 * order)
|
|
704
|
+
+ (0.35 * renamed)
|
|
705
|
+
+ (0.20 * structure)
|
|
706
|
+
+ (0.15 * literals)
|
|
707
|
+
)
|
|
708
|
+
return PairScore(
|
|
709
|
+
*(
|
|
710
|
+
round(value, 2)
|
|
711
|
+
for value in (percentage, content, order, renamed, structure, literals, partial)
|
|
712
|
+
)
|
|
713
|
+
)
|
|
714
|
+
|
|
715
|
+
|
|
716
|
+
def _classic_pair_score(left: SourceDocument, right: SourceDocument, shingle_size: int) -> PairScore:
|
|
717
|
+
percentage = _similarity_tokens(left.tokens, right.tokens, shingle_size)
|
|
718
|
+
return PairScore(percentage, percentage, percentage, percentage, percentage, percentage, percentage)
|
|
719
|
+
|
|
720
|
+
|
|
556
721
|
def similarity_percent(
|
|
557
722
|
left: str,
|
|
558
723
|
right: str,
|
|
559
724
|
*,
|
|
560
725
|
tipo: str | None = None,
|
|
561
726
|
shingle_size: int = 3,
|
|
727
|
+
algorithm: str = "advanced",
|
|
562
728
|
) -> float:
|
|
563
729
|
if shingle_size < 1:
|
|
564
730
|
raise ValueError("shingle_size deve ser positivo")
|
|
565
|
-
if
|
|
566
|
-
|
|
567
|
-
|
|
568
|
-
|
|
569
|
-
|
|
570
|
-
|
|
731
|
+
if algorithm not in {"advanced", "classic"}:
|
|
732
|
+
raise ValueError("algorithm deve ser advanced ou classic")
|
|
733
|
+
kind = _canonical_kind(tipo or "cobol")
|
|
734
|
+
left_document = _make_document(kind, "LEFT", "LEFT", left)
|
|
735
|
+
right_document = _make_document(kind, "RIGHT", "RIGHT", right)
|
|
736
|
+
scorer = _advanced_similarity if algorithm == "advanced" else _classic_pair_score
|
|
737
|
+
return scorer(left_document, right_document, shingle_size).percentage
|
|
571
738
|
|
|
572
739
|
|
|
573
740
|
def _shingle_hashes(tokens: Sequence[str], width: int) -> set[int]:
|
|
@@ -620,6 +787,7 @@ def _candidate_pairs(
|
|
|
620
787
|
*,
|
|
621
788
|
shingle_size: int,
|
|
622
789
|
exhaustive_limit: int,
|
|
790
|
+
algorithm: str,
|
|
623
791
|
) -> tuple[set[tuple[int, int]], str]:
|
|
624
792
|
candidates: set[tuple[int, int]] = set()
|
|
625
793
|
indices_by_kind: dict[str, list[int]] = defaultdict(list)
|
|
@@ -641,17 +809,21 @@ def _candidate_pairs(
|
|
|
641
809
|
anchor = min(bucket)
|
|
642
810
|
candidates.update((anchor, item) for item in sorted(bucket) if item != anchor)
|
|
643
811
|
|
|
644
|
-
band_buckets: dict[tuple[int, tuple[int, ...]], list[int]] = defaultdict(list)
|
|
812
|
+
band_buckets: dict[tuple[str, int, tuple[int, ...]], list[int]] = defaultdict(list)
|
|
645
813
|
rows = _MINHASH_SIZE // _LSH_BANDS
|
|
646
814
|
# One representative per exact digest keeps duplicate-heavy scans linear.
|
|
647
815
|
representatives = sorted(min(bucket) for bucket in exact_groups.values())
|
|
648
816
|
for index in representatives:
|
|
649
|
-
|
|
650
|
-
|
|
651
|
-
|
|
652
|
-
for
|
|
653
|
-
|
|
654
|
-
|
|
817
|
+
feature_streams = [("raw", documents[index].tokens)]
|
|
818
|
+
if algorithm == "advanced":
|
|
819
|
+
feature_streams.append(("abstract", documents[index].abstract_tokens))
|
|
820
|
+
for feature_name, feature_tokens in feature_streams:
|
|
821
|
+
signature = _minhash_signature(
|
|
822
|
+
_shingle_hashes(feature_tokens, shingle_size)
|
|
823
|
+
)
|
|
824
|
+
for band in range(_LSH_BANDS):
|
|
825
|
+
start = band * rows
|
|
826
|
+
band_buckets[(feature_name, band, signature[start : start + rows])].append(index)
|
|
655
827
|
for bucket in band_buckets.values():
|
|
656
828
|
if len(bucket) > 1:
|
|
657
829
|
_add_bucket_pairs(bucket, candidates)
|
|
@@ -709,6 +881,10 @@ def analyze_documents(
|
|
|
709
881
|
threshold: float = 80.0,
|
|
710
882
|
shingle_size: int = 3,
|
|
711
883
|
exhaustive_limit: int = 80,
|
|
884
|
+
algorithm: str = "advanced",
|
|
885
|
+
partial_threshold: float = 75.0,
|
|
886
|
+
min_clone_tokens: int = 30,
|
|
887
|
+
cluster_policy: str = "connected",
|
|
712
888
|
) -> SimilarityReport:
|
|
713
889
|
if not 0 <= threshold <= 100:
|
|
714
890
|
raise ValueError("threshold deve estar entre 0 e 100")
|
|
@@ -716,6 +892,14 @@ def analyze_documents(
|
|
|
716
892
|
raise ValueError("shingle_size deve ser positivo")
|
|
717
893
|
if exhaustive_limit < 2:
|
|
718
894
|
raise ValueError("exhaustive_limit deve ser pelo menos 2")
|
|
895
|
+
if algorithm not in {"advanced", "classic"}:
|
|
896
|
+
raise ValueError("algorithm deve ser advanced ou classic")
|
|
897
|
+
if not 0 <= partial_threshold <= 100:
|
|
898
|
+
raise ValueError("partial_threshold deve estar entre 0 e 100")
|
|
899
|
+
if min_clone_tokens < 1:
|
|
900
|
+
raise ValueError("min_clone_tokens deve ser positivo")
|
|
901
|
+
if cluster_policy not in {"cohesive", "connected"}:
|
|
902
|
+
raise ValueError("cluster_policy deve ser cohesive ou connected")
|
|
719
903
|
|
|
720
904
|
ordered = sorted(
|
|
721
905
|
documents,
|
|
@@ -733,24 +917,55 @@ def analyze_documents(
|
|
|
733
917
|
ordered,
|
|
734
918
|
shingle_size=shingle_size,
|
|
735
919
|
exhaustive_limit=exhaustive_limit,
|
|
920
|
+
algorithm=algorithm,
|
|
736
921
|
)
|
|
737
|
-
score_cache: dict[tuple[int, int],
|
|
922
|
+
score_cache: dict[tuple[int, int], PairScore] = {}
|
|
738
923
|
|
|
739
|
-
def score(left: int, right: int) ->
|
|
924
|
+
def score(left: int, right: int) -> PairScore:
|
|
740
925
|
key = _pair_key(left, right)
|
|
741
926
|
if key not in score_cache:
|
|
742
|
-
|
|
743
|
-
|
|
744
|
-
)
|
|
927
|
+
scorer = _advanced_similarity if algorithm == "advanced" else _classic_pair_score
|
|
928
|
+
score_cache[key] = scorer(ordered[key[0]], ordered[key[1]], shingle_size)
|
|
745
929
|
return score_cache[key]
|
|
746
930
|
|
|
747
931
|
union_find = _UnionFind(len(ordered))
|
|
748
|
-
|
|
932
|
+
component_members: dict[int, set[int]] = {index: {index} for index in range(len(ordered))}
|
|
933
|
+
qualifying_edges: list[tuple[float, int, int]] = []
|
|
749
934
|
for left, right in sorted(candidates):
|
|
750
|
-
percentage = score(left, right)
|
|
935
|
+
percentage = score(left, right).percentage
|
|
751
936
|
if percentage >= threshold:
|
|
752
|
-
|
|
753
|
-
|
|
937
|
+
qualifying_edges.append((percentage, left, right))
|
|
938
|
+
|
|
939
|
+
rejected_by_cohesion = 0
|
|
940
|
+
cohesion_limit_rejections = 0
|
|
941
|
+
for _, left, right in sorted(qualifying_edges, key=lambda item: (-item[0], item[1], item[2])):
|
|
942
|
+
left_root = union_find.find(left)
|
|
943
|
+
right_root = union_find.find(right)
|
|
944
|
+
if left_root == right_root:
|
|
945
|
+
continue
|
|
946
|
+
left_members = component_members[left_root]
|
|
947
|
+
right_members = component_members[right_root]
|
|
948
|
+
should_merge = True
|
|
949
|
+
if cluster_policy == "cohesive":
|
|
950
|
+
combined = left_members | right_members
|
|
951
|
+
identical = len({ordered[index].digest for index in combined}) == 1
|
|
952
|
+
if len(combined) > exhaustive_limit and not identical:
|
|
953
|
+
should_merge = False
|
|
954
|
+
cohesion_limit_rejections += 1
|
|
955
|
+
elif not identical:
|
|
956
|
+
should_merge = all(
|
|
957
|
+
score(member_left, member_right).percentage >= threshold
|
|
958
|
+
for member_left in sorted(left_members)
|
|
959
|
+
for member_right in sorted(right_members)
|
|
960
|
+
)
|
|
961
|
+
if not should_merge:
|
|
962
|
+
rejected_by_cohesion += 1
|
|
963
|
+
continue
|
|
964
|
+
union_find.union(left_root, right_root)
|
|
965
|
+
merged_root = union_find.find(left_root)
|
|
966
|
+
old_root = right_root if merged_root == left_root else left_root
|
|
967
|
+
component_members[merged_root] = left_members | right_members
|
|
968
|
+
component_members.pop(old_root, None)
|
|
754
969
|
|
|
755
970
|
components: dict[int, list[int]] = defaultdict(list)
|
|
756
971
|
for index in range(len(ordered)):
|
|
@@ -794,13 +1009,15 @@ def analyze_documents(
|
|
|
794
1009
|
if pair_index >= _MAX_REPORTED_PAIRS:
|
|
795
1010
|
exact_interval = False
|
|
796
1011
|
break
|
|
797
|
-
|
|
1012
|
+
pair_score = score(left, right)
|
|
1013
|
+
percentage = pair_score.percentage
|
|
798
1014
|
pair_scores.append(percentage)
|
|
799
1015
|
pair_details.append(
|
|
800
1016
|
{
|
|
801
1017
|
"origem": ordered[left].nome,
|
|
802
1018
|
"destino": ordered[right].nome,
|
|
803
1019
|
"percentual": percentage,
|
|
1020
|
+
"componentes": pair_score.to_dict(),
|
|
804
1021
|
}
|
|
805
1022
|
)
|
|
806
1023
|
percentage_range = _format_range(pair_scores)
|
|
@@ -846,6 +1063,28 @@ def analyze_documents(
|
|
|
846
1063
|
count * (count - 1) // 2
|
|
847
1064
|
for count in Counter(document.tipo for document in ordered).values()
|
|
848
1065
|
)
|
|
1066
|
+
partial_clones: list[dict[str, Any]] = []
|
|
1067
|
+
if algorithm == "advanced":
|
|
1068
|
+
for (left, right), pair_score in sorted(score_cache.items()):
|
|
1069
|
+
minimum_tokens = min(len(ordered[left].tokens), len(ordered[right].tokens))
|
|
1070
|
+
if (
|
|
1071
|
+
pair_score.percentage < threshold
|
|
1072
|
+
and pair_score.partial >= partial_threshold
|
|
1073
|
+
and minimum_tokens >= min_clone_tokens
|
|
1074
|
+
):
|
|
1075
|
+
partial_clones.append(
|
|
1076
|
+
{
|
|
1077
|
+
"tipo": ordered[left].tipo,
|
|
1078
|
+
"origem": ordered[left].nome,
|
|
1079
|
+
"destino": ordered[right].nome,
|
|
1080
|
+
"percentual_global": pair_score.percentage,
|
|
1081
|
+
"percentual_contencao": pair_score.partial,
|
|
1082
|
+
"componentes": pair_score.to_dict(),
|
|
1083
|
+
}
|
|
1084
|
+
)
|
|
1085
|
+
partial_clones.sort(
|
|
1086
|
+
key=lambda item: (-item["percentual_contencao"], item["tipo"], item["origem"], item["destino"])
|
|
1087
|
+
)
|
|
849
1088
|
statistics = {
|
|
850
1089
|
"arquivos": len(ordered),
|
|
851
1090
|
"clusters": len(clusters),
|
|
@@ -853,16 +1092,25 @@ def analyze_documents(
|
|
|
853
1092
|
"pares_possiveis": possible_pairs,
|
|
854
1093
|
"pares_candidatos": len(candidates),
|
|
855
1094
|
"pares_comparados": len(score_cache),
|
|
856
|
-
"pares_acima_limiar":
|
|
1095
|
+
"pares_acima_limiar": len(qualifying_edges),
|
|
857
1096
|
"modo_candidatos": mode,
|
|
1097
|
+
"algoritmo": algorithm,
|
|
1098
|
+
"clones_parciais": len(partial_clones),
|
|
1099
|
+
"politica_cluster": cluster_policy,
|
|
1100
|
+
"unioes_rejeitadas_por_coesao": rejected_by_cohesion,
|
|
1101
|
+
"unioes_rejeitadas_por_limite": cohesion_limit_rejections,
|
|
858
1102
|
}
|
|
859
1103
|
return SimilarityReport(
|
|
860
1104
|
threshold=float(threshold),
|
|
861
1105
|
shingle_size=shingle_size,
|
|
862
1106
|
exhaustive_limit=exhaustive_limit,
|
|
1107
|
+
algorithm=algorithm,
|
|
1108
|
+
partial_threshold=float(partial_threshold),
|
|
1109
|
+
cluster_policy=cluster_policy,
|
|
863
1110
|
records=records,
|
|
864
1111
|
clusters=clusters,
|
|
865
1112
|
statistics=statistics,
|
|
1113
|
+
partial_clones=partial_clones,
|
|
866
1114
|
documents=tuple(ordered),
|
|
867
1115
|
)
|
|
868
1116
|
|
|
@@ -880,6 +1128,10 @@ def analyze_codebase(
|
|
|
880
1128
|
threshold: float = 80.0,
|
|
881
1129
|
shingle_size: int = 3,
|
|
882
1130
|
exhaustive_limit: int = 80,
|
|
1131
|
+
algorithm: str = "advanced",
|
|
1132
|
+
partial_threshold: float = 75.0,
|
|
1133
|
+
min_clone_tokens: int = 30,
|
|
1134
|
+
cluster_policy: str = "connected",
|
|
883
1135
|
ignored_directories: Iterable[str] = _IGNORED_DIRECTORIES,
|
|
884
1136
|
) -> SimilarityReport:
|
|
885
1137
|
return analyze_documents(
|
|
@@ -887,6 +1139,10 @@ def analyze_codebase(
|
|
|
887
1139
|
threshold=threshold,
|
|
888
1140
|
shingle_size=shingle_size,
|
|
889
1141
|
exhaustive_limit=exhaustive_limit,
|
|
1142
|
+
algorithm=algorithm,
|
|
1143
|
+
partial_threshold=partial_threshold,
|
|
1144
|
+
min_clone_tokens=min_clone_tokens,
|
|
1145
|
+
cluster_policy=cluster_policy,
|
|
890
1146
|
)
|
|
891
1147
|
|
|
892
1148
|
|
|
@@ -972,7 +1228,7 @@ def build_ai_review_payload(
|
|
|
972
1228
|
)
|
|
973
1229
|
|
|
974
1230
|
return {
|
|
975
|
-
"schema_version":
|
|
1231
|
+
"schema_version": 2,
|
|
976
1232
|
"tarefa": (
|
|
977
1233
|
"Revise cada cluster pela semantica do codigo e retorne um parecer "
|
|
978
1234
|
"similar, identico ou diferente, com justificativa breve."
|
|
@@ -988,6 +1244,7 @@ def build_ai_review_payload(
|
|
|
988
1244
|
]
|
|
989
1245
|
},
|
|
990
1246
|
"clusters": clusters,
|
|
1247
|
+
"clones_parciais": deepcopy(payload.get("clones_parciais", [])),
|
|
991
1248
|
}
|
|
992
1249
|
|
|
993
1250
|
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/src/mainframe_toolkit/io.py
RENAMED
|
File without changes
|
{mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/src/mainframe_toolkit/jcl.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/validator-java/mvnw.cmd
RENAMED
|
File without changes
|
{mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/validator-java/pom.xml
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|