mainframe-migration-toolkit 0.2.0__tar.gz → 0.3.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (51) hide show
  1. {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/.claude/skills/analyze-mainframe-similarity/SKILL.md +3 -1
  2. {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/PKG-INFO +1 -1
  3. {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/pyproject.toml +1 -1
  4. {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/src/mainframe_toolkit/__init__.py +1 -1
  5. {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/src/mainframe_toolkit/cli.py +8 -0
  6. {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/src/mainframe_toolkit/similarity.py +284 -27
  7. {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/.claude/skills/migrate-mainframe-job/SKILL.md +0 -0
  8. {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/.claude/skills/validate-golden-dataset/SKILL.md +0 -0
  9. {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/.gitignore +0 -0
  10. {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/AGENTS.md +0 -0
  11. {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/CLAUDE.md +0 -0
  12. {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/src/mainframe_toolkit/__main__.py +0 -0
  13. {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/src/mainframe_toolkit/cobol.py +0 -0
  14. {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/src/mainframe_toolkit/copybook.py +0 -0
  15. {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/src/mainframe_toolkit/errors.py +0 -0
  16. {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/src/mainframe_toolkit/external.py +0 -0
  17. {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/src/mainframe_toolkit/io.py +0 -0
  18. {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/src/mainframe_toolkit/jcl.py +0 -0
  19. {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/src/mainframe_toolkit/pipeline.py +0 -0
  20. {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/src/mainframe_toolkit/sequential.py +0 -0
  21. {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/src/mainframe_toolkit/sorting.py +0 -0
  22. {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/src/mainframe_toolkit/specs.py +0 -0
  23. {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/src/mainframe_toolkit/synthetic.py +0 -0
  24. {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/src/mainframe_toolkit/workspace.py +0 -0
  25. {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/validator-java/.mvn/wrapper/maven-wrapper.properties +0 -0
  26. {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/validator-java/mvnw +0 -0
  27. {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/validator-java/mvnw.cmd +0 -0
  28. {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/validator-java/pom.xml +0 -0
  29. {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/validator-java/src/main/java/io/mainframe/migration/validator/AvroValueFormatter.java +0 -0
  30. {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/validator-java/src/main/java/io/mainframe/migration/validator/CsvTabularReader.java +0 -0
  31. {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/validator-java/src/main/java/io/mainframe/migration/validator/DataFormat.java +0 -0
  32. {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/validator-java/src/main/java/io/mainframe/migration/validator/Difference.java +0 -0
  33. {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/validator-java/src/main/java/io/mainframe/migration/validator/InputFileSet.java +0 -0
  34. {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/validator-java/src/main/java/io/mainframe/migration/validator/InputOptions.java +0 -0
  35. {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/validator-java/src/main/java/io/mainframe/migration/validator/JsonReportWriter.java +0 -0
  36. {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/validator-java/src/main/java/io/mainframe/migration/validator/MultiFileTabularReader.java +0 -0
  37. {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/validator-java/src/main/java/io/mainframe/migration/validator/Normalization.java +0 -0
  38. {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/validator-java/src/main/java/io/mainframe/migration/validator/ParquetTabularReader.java +0 -0
  39. {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/validator-java/src/main/java/io/mainframe/migration/validator/TabularReader.java +0 -0
  40. {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/validator-java/src/main/java/io/mainframe/migration/validator/TabularReaderFactory.java +0 -0
  41. {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/validator-java/src/main/java/io/mainframe/migration/validator/ValidationOptions.java +0 -0
  42. {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/validator-java/src/main/java/io/mainframe/migration/validator/ValidationReport.java +0 -0
  43. {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/validator-java/src/main/java/io/mainframe/migration/validator/ValidationService.java +0 -0
  44. {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/validator-java/src/main/java/io/mainframe/migration/validator/ValidatorCli.java +0 -0
  45. {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/validator-java/src/main/java/io/mainframe/migration/validator/ValueNormalizer.java +0 -0
  46. {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/validator-java/src/test/java/io/mainframe/migration/validator/DirectoryValidationTest.java +0 -0
  47. {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/validator-java/src/test/java/io/mainframe/migration/validator/JsonReportWriterTest.java +0 -0
  48. {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/validator-java/src/test/java/io/mainframe/migration/validator/KeyedValidationTest.java +0 -0
  49. {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/validator-java/src/test/java/io/mainframe/migration/validator/ParquetValidationTest.java +0 -0
  50. {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/validator-java/src/test/java/io/mainframe/migration/validator/ValidationServiceTest.java +0 -0
  51. {mainframe_migration_toolkit-0.2.0 → mainframe_migration_toolkit-0.3.0}/validator-java/src/test/java/io/mainframe/migration/validator/ValidatorCliTest.java +0 -0
@@ -8,11 +8,13 @@ description: Cluster COBOL and JCL source by deterministic textual similarity, i
8
8
  1. Run the deterministic scan:
9
9
 
10
10
  ```text
11
- python -m mainframe_toolkit similarity scan <source-root> --threshold <threshold> --json similarity.json --csv similarity.csv --review-request similarity-review-request.json
11
+ python -m mainframe_toolkit similarity scan <source-root> --algorithm advanced --threshold <threshold> --json similarity.json --csv similarity.csv --review-request similarity-review-request.json
12
12
  ```
13
13
 
14
14
  Use the requested threshold, or 80 when absent. Never edit deterministic scores or cluster membership manually.
15
15
 
16
+ Inspect each pair's `componentes` (`conteudo`, `ordem`, `renomeacao`, `estrutura`, `literais`, and `clone_parcial`). Also inspect `clones_parciais`, which are evidence of shared blocks but do not automatically join clusters. Use `--cluster-policy cohesive` when every pair in a cluster must meet the threshold; keep `connected` for broader codebase discovery.
17
+
16
18
  2. For each non-singleton cluster in the review request, read every original file and its normalized evidence. Check all pair relationships; connected-component chaining alone is not proof of equivalence. Ignore comments, sequence columns, copyright blocks, whitespace, generated names, and environment-only identifiers. Compare JCL execution flow/DD semantics or COBOL inputs, outputs, branches, calculations, side effects, error behavior, and record layouts.
17
19
 
18
20
  3. Assign exactly one verdict per requested cluster:
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: mainframe-migration-toolkit
3
- Version: 0.2.0
3
+ Version: 0.3.0
4
4
  Summary: Runtime and deterministic tools for COBOL/JCL to PySpark migrations
5
5
  Author: Mainframe Migration Toolkit
6
6
  Requires-Python: >=3.10
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
4
4
 
5
5
  [project]
6
6
  name = "mainframe-migration-toolkit"
7
- version = "0.2.0"
7
+ version = "0.3.0"
8
8
  description = "Runtime and deterministic tools for COBOL/JCL to PySpark migrations"
9
9
  requires-python = ">=3.10"
10
10
  authors = [{ name = "Mainframe Migration Toolkit" }]
@@ -89,4 +89,4 @@ __all__ = [
89
89
  "write_s3_parquet",
90
90
  ]
91
91
 
92
- __version__ = "0.2.0"
92
+ __version__ = "0.3.0"
@@ -84,6 +84,10 @@ def _similarity_scan(args: argparse.Namespace) -> int:
84
84
  threshold=args.threshold,
85
85
  shingle_size=args.shingle_size,
86
86
  exhaustive_limit=args.exhaustive_limit,
87
+ algorithm=args.algorithm,
88
+ partial_threshold=args.partial_threshold,
89
+ min_clone_tokens=args.min_clone_tokens,
90
+ cluster_policy=args.cluster_policy,
87
91
  )
88
92
  write_results(report, args.json, args.csv)
89
93
  write_ai_review_payload(report, args.review_request)
@@ -293,6 +297,10 @@ def _parser() -> argparse.ArgumentParser:
293
297
  scan.add_argument("--threshold", type=float, default=80.0)
294
298
  scan.add_argument("--shingle-size", type=int, default=3)
295
299
  scan.add_argument("--exhaustive-limit", type=int, default=80)
300
+ scan.add_argument("--algorithm", choices=("advanced", "classic"), default="advanced")
301
+ scan.add_argument("--partial-threshold", type=float, default=75.0)
302
+ scan.add_argument("--min-clone-tokens", type=int, default=30)
303
+ scan.add_argument("--cluster-policy", choices=("cohesive", "connected"), default="connected")
296
304
  scan.add_argument("--json", default="similarity.json")
297
305
  scan.add_argument("--csv", default="similarity.csv")
298
306
  scan.add_argument("--review-request", default="similarity-review-request.json")
@@ -81,6 +81,45 @@ _LSH_BANDS = 12
81
81
  _MAX_LSH_BUCKET = 128
82
82
  _MAX_REPORTED_PAIRS = 5_000
83
83
 
84
+ _COBOL_STRUCTURAL_WORDS = frozenset(
85
+ {
86
+ "ACCEPT", "ADD", "CALL", "CANCEL", "CLOSE", "COMPUTE", "CONTINUE",
87
+ "DELETE", "DISPLAY", "DIVIDE", "ELSE", "END-CALL", "END-COMPUTE",
88
+ "END-EVALUATE", "END-IF", "END-PERFORM", "END-READ", "END-SEARCH",
89
+ "END-STRING", "END-WRITE", "ENTRY", "EVALUATE", "EXIT", "GOBACK",
90
+ "GO", "IF", "INITIALIZE", "INSPECT", "MERGE", "MOVE", "MULTIPLY",
91
+ "OPEN", "PERFORM", "READ", "RELEASE", "RETURN", "REWRITE", "SEARCH",
92
+ "SET", "SORT", "START", "STOP", "STRING", "SUBTRACT", "UNSTRING",
93
+ "WHEN", "WRITE", "IDENTIFICATION", "ENVIRONMENT", "DATA", "PROCEDURE",
94
+ "DIVISION", "SECTION", "DECLARATIVES", "END-DECLARATIVES",
95
+ }
96
+ )
97
+ _COBOL_LANGUAGE_WORDS = _COBOL_STRUCTURAL_WORDS | frozenset(
98
+ {
99
+ "PROGRAM-ID", "FILE-CONTROL", "SELECT", "ASSIGN", "FD", "SD", "PIC",
100
+ "PICTURE", "VALUE", "VALUES", "OCCURS", "REDEFINES", "RENAMES", "USAGE",
101
+ "COMP", "COMP-1", "COMP-2", "COMP-3", "BINARY", "PACKED-DECIMAL",
102
+ "DISPLAY-1", "WORKING-STORAGE", "LINKAGE", "FILE", "INPUT-OUTPUT",
103
+ "USING", "GIVING", "THRU", "THROUGH", "UNTIL", "VARYING", "TIMES",
104
+ "FROM", "BY", "TO", "INTO", "OF", "IN", "ON", "AT", "END", "NOT",
105
+ "AND", "OR", "TRUE", "FALSE", "ZERO", "ZEROS", "ZEROES", "SPACE",
106
+ "SPACES", "HIGH-VALUES", "LOW-VALUES", "ALL", "CORRESPONDING", "ROUNDED",
107
+ "SIZE", "ERROR", "INVALID", "KEY", "WITH", "TEST", "AFTER", "BEFORE",
108
+ "ASCENDING", "DESCENDING", "DUPLICATES", "NO", "ADVANCING", "UPON",
109
+ }
110
+ )
111
+ _JCL_LANGUAGE_WORDS = _JCL_OPERATIONS | frozenset(
112
+ {
113
+ "PGM", "PROC", "PARM", "COND", "DSN", "DSNAME", "DISP", "UNIT",
114
+ "SPACE", "DCB", "SYSOUT", "CLASS", "MSGCLASS", "REGION", "TIME",
115
+ "RESTART", "SYMBOLS", "MEMBER", "ORDER", "INCLUDE", "SET", "IF",
116
+ "THEN", "ELSE", "ENDIF", "PASS", "KEEP", "CATLG", "DELETE", "SHR",
117
+ "OLD", "NEW", "MOD", "DUMMY", "DATA", "DLM",
118
+ }
119
+ )
120
+ _IDENTIFIER_RE = re.compile(r"[A-Z_$#@&][A-Z0-9_$#@&.-]*\Z")
121
+ _NUMBER_RE = re.compile(r"[+-]?(?:\d+(?:\.\d*)?|\.\d+)\Z")
122
+
84
123
 
85
124
  @dataclass(frozen=True, slots=True)
86
125
  class SourceDocument:
@@ -91,6 +130,9 @@ class SourceDocument:
91
130
  content: str = field(repr=False)
92
131
  normalized: str = field(repr=False)
93
132
  tokens: tuple[str, ...] = field(repr=False)
133
+ abstract_tokens: tuple[str, ...] = field(repr=False)
134
+ structure_tokens: tuple[str, ...] = field(repr=False)
135
+ literal_tokens: tuple[str, ...] = field(repr=False)
94
136
  digest: str
95
137
  source_id: str
96
138
 
@@ -100,22 +142,30 @@ class SimilarityReport:
100
142
  threshold: float
101
143
  shingle_size: int
102
144
  exhaustive_limit: int
145
+ algorithm: str
146
+ partial_threshold: float
147
+ cluster_policy: str
103
148
  records: list[dict[str, Any]]
104
149
  clusters: list[dict[str, Any]]
105
150
  statistics: dict[str, Any]
151
+ partial_clones: list[dict[str, Any]] = field(default_factory=list)
106
152
  documents: tuple[SourceDocument, ...] = field(default_factory=tuple, repr=False)
107
153
 
108
154
  def to_dict(self) -> dict[str, Any]:
109
155
  return {
110
- "schema_version": 1,
156
+ "schema_version": 2,
111
157
  "configuracao": {
112
158
  "limiar_percentual": self.threshold,
113
159
  "tamanho_shingle": self.shingle_size,
114
160
  "limite_comparacao_exaustiva": self.exhaustive_limit,
161
+ "algoritmo": self.algorithm,
162
+ "limiar_clone_parcial": self.partial_threshold,
163
+ "politica_cluster": self.cluster_policy,
115
164
  },
116
165
  "resumo": deepcopy(self.statistics),
117
166
  "arquivos": deepcopy(self.records),
118
167
  "clusters": deepcopy(self.clusters),
168
+ "clones_parciais": deepcopy(self.partial_clones),
119
169
  }
120
170
 
121
171
  def __getitem__(self, key: str) -> Any:
@@ -409,6 +459,38 @@ def tokenize_normalized(text: str) -> tuple[str, ...]:
409
459
  return tuple(_TOKEN_RE.findall(text))
410
460
 
411
461
 
462
+ def _is_literal(token: str) -> bool:
463
+ return (
464
+ len(token) >= 2
465
+ and token[0] == token[-1]
466
+ and token[0] in {"'", '"'}
467
+ ) or _NUMBER_RE.fullmatch(token) is not None
468
+
469
+
470
+ def _semantic_features(tokens: Sequence[str], tipo: str) -> tuple[tuple[str, ...], tuple[str, ...], tuple[str, ...]]:
471
+ language_words = _COBOL_LANGUAGE_WORDS if tipo == "cobol" else _JCL_LANGUAGE_WORDS
472
+ structural_words = _COBOL_STRUCTURAL_WORDS if tipo == "cobol" else _JCL_LANGUAGE_WORDS
473
+ identifiers: dict[str, str] = {}
474
+ abstract: list[str] = []
475
+ structure: list[str] = []
476
+ literals: list[str] = []
477
+
478
+ for token in tokens:
479
+ upper = token.upper()
480
+ if _is_literal(token):
481
+ literals.append(token)
482
+ abstract.append(token)
483
+ elif upper in language_words or not _IDENTIFIER_RE.fullmatch(upper):
484
+ abstract.append(upper)
485
+ else:
486
+ placeholder = identifiers.setdefault(upper, f"__ID{len(identifiers) + 1}__")
487
+ abstract.append(placeholder)
488
+ if upper in structural_words:
489
+ structure.append(upper)
490
+
491
+ return tuple(abstract), tuple(structure), tuple(literals)
492
+
493
+
412
494
  def _stable_digest(value: str, *, size: int = 16) -> str:
413
495
  return hashlib.blake2b(value.encode("utf-8"), digest_size=size).hexdigest()
414
496
 
@@ -423,6 +505,8 @@ def _make_document(
423
505
  kind = _canonical_kind(tipo)
424
506
  relative = Path(relative_path).as_posix()
425
507
  normalized = normalize_source(content, kind)
508
+ tokens = tokenize_normalized(normalized)
509
+ abstract_tokens, structure_tokens, literal_tokens = _semantic_features(tokens, kind)
426
510
  return SourceDocument(
427
511
  tipo=kind,
428
512
  nome=name,
@@ -430,7 +514,10 @@ def _make_document(
430
514
  relative_path=relative,
431
515
  content=content,
432
516
  normalized=normalized,
433
- tokens=tokenize_normalized(normalized),
517
+ tokens=tokens,
518
+ abstract_tokens=abstract_tokens,
519
+ structure_tokens=structure_tokens,
520
+ literal_tokens=literal_tokens,
434
521
  digest=_stable_digest(normalized),
435
522
  source_id=_stable_digest(f"{kind}\0{relative}", size=10),
436
523
  )
@@ -538,6 +625,34 @@ def _dice(left: Counter[Any], right: Counter[Any]) -> float:
538
625
  return (2.0 * overlap) / total
539
626
 
540
627
 
628
+ def _containment(left: Counter[Any], right: Counter[Any]) -> float:
629
+ smallest = min(sum(left.values()), sum(right.values()))
630
+ if smallest == 0:
631
+ return 1.0 if not left and not right else 0.0
632
+ return sum((left & right).values()) / smallest
633
+
634
+
635
+ @dataclass(frozen=True, slots=True)
636
+ class PairScore:
637
+ percentage: float
638
+ content: float
639
+ order: float
640
+ renamed: float
641
+ structure: float
642
+ literals: float
643
+ partial: float
644
+
645
+ def to_dict(self) -> dict[str, float]:
646
+ return {
647
+ "conteudo": self.content,
648
+ "ordem": self.order,
649
+ "renomeacao": self.renamed,
650
+ "estrutura": self.structure,
651
+ "literais": self.literals,
652
+ "clone_parcial": self.partial,
653
+ }
654
+
655
+
541
656
  def _similarity_tokens(
542
657
  left: Sequence[str], right: Sequence[str], shingle_size: int
543
658
  ) -> float:
@@ -553,21 +668,73 @@ def _similarity_tokens(
553
668
  return round(100.0 * ((0.25 * ordered) + (0.75 * lexical)), 2)
554
669
 
555
670
 
671
+ def _advanced_similarity(left: SourceDocument, right: SourceDocument, shingle_size: int) -> PairScore:
672
+ if left.tokens == right.tokens:
673
+ return PairScore(100.0, 100.0, 100.0, 100.0, 100.0, 100.0, 100.0)
674
+
675
+ width = max(1, min(shingle_size, len(left.tokens), len(right.tokens)))
676
+ left_words = _feature_counter(left.tokens, 1)
677
+ right_words = _feature_counter(right.tokens, 1)
678
+ left_ordered = _feature_counter(left.tokens, width)
679
+ right_ordered = _feature_counter(right.tokens, width)
680
+ content = 100.0 * _dice(left_words, right_words)
681
+ order = 100.0 * _dice(left_ordered, right_ordered)
682
+ renamed = 100.0 * _dice(
683
+ _feature_counter(left.abstract_tokens, width),
684
+ _feature_counter(right.abstract_tokens, width),
685
+ )
686
+ structure = 100.0 * _dice(
687
+ _feature_counter(left.structure_tokens, 2),
688
+ _feature_counter(right.structure_tokens, 2),
689
+ )
690
+ literals = 100.0 * _dice(
691
+ _feature_counter(left.literal_tokens, 1),
692
+ _feature_counter(right.literal_tokens, 1),
693
+ )
694
+ partial = 100.0 * max(
695
+ _containment(left_ordered, right_ordered),
696
+ _containment(
697
+ _feature_counter(left.abstract_tokens, width),
698
+ _feature_counter(right.abstract_tokens, width),
699
+ ),
700
+ )
701
+ percentage = (
702
+ (0.15 * content)
703
+ + (0.15 * order)
704
+ + (0.35 * renamed)
705
+ + (0.20 * structure)
706
+ + (0.15 * literals)
707
+ )
708
+ return PairScore(
709
+ *(
710
+ round(value, 2)
711
+ for value in (percentage, content, order, renamed, structure, literals, partial)
712
+ )
713
+ )
714
+
715
+
716
+ def _classic_pair_score(left: SourceDocument, right: SourceDocument, shingle_size: int) -> PairScore:
717
+ percentage = _similarity_tokens(left.tokens, right.tokens, shingle_size)
718
+ return PairScore(percentage, percentage, percentage, percentage, percentage, percentage, percentage)
719
+
720
+
556
721
  def similarity_percent(
557
722
  left: str,
558
723
  right: str,
559
724
  *,
560
725
  tipo: str | None = None,
561
726
  shingle_size: int = 3,
727
+ algorithm: str = "advanced",
562
728
  ) -> float:
563
729
  if shingle_size < 1:
564
730
  raise ValueError("shingle_size deve ser positivo")
565
- if tipo is not None:
566
- left = normalize_source(left, tipo)
567
- right = normalize_source(right, tipo)
568
- return _similarity_tokens(
569
- tokenize_normalized(left), tokenize_normalized(right), shingle_size
570
- )
731
+ if algorithm not in {"advanced", "classic"}:
732
+ raise ValueError("algorithm deve ser advanced ou classic")
733
+ kind = _canonical_kind(tipo or "cobol")
734
+ left_document = _make_document(kind, "LEFT", "LEFT", left)
735
+ right_document = _make_document(kind, "RIGHT", "RIGHT", right)
736
+ scorer = _advanced_similarity if algorithm == "advanced" else _classic_pair_score
737
+ return scorer(left_document, right_document, shingle_size).percentage
571
738
 
572
739
 
573
740
  def _shingle_hashes(tokens: Sequence[str], width: int) -> set[int]:
@@ -620,6 +787,7 @@ def _candidate_pairs(
620
787
  *,
621
788
  shingle_size: int,
622
789
  exhaustive_limit: int,
790
+ algorithm: str,
623
791
  ) -> tuple[set[tuple[int, int]], str]:
624
792
  candidates: set[tuple[int, int]] = set()
625
793
  indices_by_kind: dict[str, list[int]] = defaultdict(list)
@@ -641,17 +809,21 @@ def _candidate_pairs(
641
809
  anchor = min(bucket)
642
810
  candidates.update((anchor, item) for item in sorted(bucket) if item != anchor)
643
811
 
644
- band_buckets: dict[tuple[int, tuple[int, ...]], list[int]] = defaultdict(list)
812
+ band_buckets: dict[tuple[str, int, tuple[int, ...]], list[int]] = defaultdict(list)
645
813
  rows = _MINHASH_SIZE // _LSH_BANDS
646
814
  # One representative per exact digest keeps duplicate-heavy scans linear.
647
815
  representatives = sorted(min(bucket) for bucket in exact_groups.values())
648
816
  for index in representatives:
649
- signature = _minhash_signature(
650
- _shingle_hashes(documents[index].tokens, shingle_size)
651
- )
652
- for band in range(_LSH_BANDS):
653
- start = band * rows
654
- band_buckets[(band, signature[start : start + rows])].append(index)
817
+ feature_streams = [("raw", documents[index].tokens)]
818
+ if algorithm == "advanced":
819
+ feature_streams.append(("abstract", documents[index].abstract_tokens))
820
+ for feature_name, feature_tokens in feature_streams:
821
+ signature = _minhash_signature(
822
+ _shingle_hashes(feature_tokens, shingle_size)
823
+ )
824
+ for band in range(_LSH_BANDS):
825
+ start = band * rows
826
+ band_buckets[(feature_name, band, signature[start : start + rows])].append(index)
655
827
  for bucket in band_buckets.values():
656
828
  if len(bucket) > 1:
657
829
  _add_bucket_pairs(bucket, candidates)
@@ -709,6 +881,10 @@ def analyze_documents(
709
881
  threshold: float = 80.0,
710
882
  shingle_size: int = 3,
711
883
  exhaustive_limit: int = 80,
884
+ algorithm: str = "advanced",
885
+ partial_threshold: float = 75.0,
886
+ min_clone_tokens: int = 30,
887
+ cluster_policy: str = "connected",
712
888
  ) -> SimilarityReport:
713
889
  if not 0 <= threshold <= 100:
714
890
  raise ValueError("threshold deve estar entre 0 e 100")
@@ -716,6 +892,14 @@ def analyze_documents(
716
892
  raise ValueError("shingle_size deve ser positivo")
717
893
  if exhaustive_limit < 2:
718
894
  raise ValueError("exhaustive_limit deve ser pelo menos 2")
895
+ if algorithm not in {"advanced", "classic"}:
896
+ raise ValueError("algorithm deve ser advanced ou classic")
897
+ if not 0 <= partial_threshold <= 100:
898
+ raise ValueError("partial_threshold deve estar entre 0 e 100")
899
+ if min_clone_tokens < 1:
900
+ raise ValueError("min_clone_tokens deve ser positivo")
901
+ if cluster_policy not in {"cohesive", "connected"}:
902
+ raise ValueError("cluster_policy deve ser cohesive ou connected")
719
903
 
720
904
  ordered = sorted(
721
905
  documents,
@@ -733,24 +917,55 @@ def analyze_documents(
733
917
  ordered,
734
918
  shingle_size=shingle_size,
735
919
  exhaustive_limit=exhaustive_limit,
920
+ algorithm=algorithm,
736
921
  )
737
- score_cache: dict[tuple[int, int], float] = {}
922
+ score_cache: dict[tuple[int, int], PairScore] = {}
738
923
 
739
- def score(left: int, right: int) -> float:
924
+ def score(left: int, right: int) -> PairScore:
740
925
  key = _pair_key(left, right)
741
926
  if key not in score_cache:
742
- score_cache[key] = _similarity_tokens(
743
- ordered[key[0]].tokens, ordered[key[1]].tokens, shingle_size
744
- )
927
+ scorer = _advanced_similarity if algorithm == "advanced" else _classic_pair_score
928
+ score_cache[key] = scorer(ordered[key[0]], ordered[key[1]], shingle_size)
745
929
  return score_cache[key]
746
930
 
747
931
  union_find = _UnionFind(len(ordered))
748
- qualifying_pairs = 0
932
+ component_members: dict[int, set[int]] = {index: {index} for index in range(len(ordered))}
933
+ qualifying_edges: list[tuple[float, int, int]] = []
749
934
  for left, right in sorted(candidates):
750
- percentage = score(left, right)
935
+ percentage = score(left, right).percentage
751
936
  if percentage >= threshold:
752
- union_find.union(left, right)
753
- qualifying_pairs += 1
937
+ qualifying_edges.append((percentage, left, right))
938
+
939
+ rejected_by_cohesion = 0
940
+ cohesion_limit_rejections = 0
941
+ for _, left, right in sorted(qualifying_edges, key=lambda item: (-item[0], item[1], item[2])):
942
+ left_root = union_find.find(left)
943
+ right_root = union_find.find(right)
944
+ if left_root == right_root:
945
+ continue
946
+ left_members = component_members[left_root]
947
+ right_members = component_members[right_root]
948
+ should_merge = True
949
+ if cluster_policy == "cohesive":
950
+ combined = left_members | right_members
951
+ identical = len({ordered[index].digest for index in combined}) == 1
952
+ if len(combined) > exhaustive_limit and not identical:
953
+ should_merge = False
954
+ cohesion_limit_rejections += 1
955
+ elif not identical:
956
+ should_merge = all(
957
+ score(member_left, member_right).percentage >= threshold
958
+ for member_left in sorted(left_members)
959
+ for member_right in sorted(right_members)
960
+ )
961
+ if not should_merge:
962
+ rejected_by_cohesion += 1
963
+ continue
964
+ union_find.union(left_root, right_root)
965
+ merged_root = union_find.find(left_root)
966
+ old_root = right_root if merged_root == left_root else left_root
967
+ component_members[merged_root] = left_members | right_members
968
+ component_members.pop(old_root, None)
754
969
 
755
970
  components: dict[int, list[int]] = defaultdict(list)
756
971
  for index in range(len(ordered)):
@@ -794,13 +1009,15 @@ def analyze_documents(
794
1009
  if pair_index >= _MAX_REPORTED_PAIRS:
795
1010
  exact_interval = False
796
1011
  break
797
- percentage = score(left, right)
1012
+ pair_score = score(left, right)
1013
+ percentage = pair_score.percentage
798
1014
  pair_scores.append(percentage)
799
1015
  pair_details.append(
800
1016
  {
801
1017
  "origem": ordered[left].nome,
802
1018
  "destino": ordered[right].nome,
803
1019
  "percentual": percentage,
1020
+ "componentes": pair_score.to_dict(),
804
1021
  }
805
1022
  )
806
1023
  percentage_range = _format_range(pair_scores)
@@ -846,6 +1063,28 @@ def analyze_documents(
846
1063
  count * (count - 1) // 2
847
1064
  for count in Counter(document.tipo for document in ordered).values()
848
1065
  )
1066
+ partial_clones: list[dict[str, Any]] = []
1067
+ if algorithm == "advanced":
1068
+ for (left, right), pair_score in sorted(score_cache.items()):
1069
+ minimum_tokens = min(len(ordered[left].tokens), len(ordered[right].tokens))
1070
+ if (
1071
+ pair_score.percentage < threshold
1072
+ and pair_score.partial >= partial_threshold
1073
+ and minimum_tokens >= min_clone_tokens
1074
+ ):
1075
+ partial_clones.append(
1076
+ {
1077
+ "tipo": ordered[left].tipo,
1078
+ "origem": ordered[left].nome,
1079
+ "destino": ordered[right].nome,
1080
+ "percentual_global": pair_score.percentage,
1081
+ "percentual_contencao": pair_score.partial,
1082
+ "componentes": pair_score.to_dict(),
1083
+ }
1084
+ )
1085
+ partial_clones.sort(
1086
+ key=lambda item: (-item["percentual_contencao"], item["tipo"], item["origem"], item["destino"])
1087
+ )
849
1088
  statistics = {
850
1089
  "arquivos": len(ordered),
851
1090
  "clusters": len(clusters),
@@ -853,16 +1092,25 @@ def analyze_documents(
853
1092
  "pares_possiveis": possible_pairs,
854
1093
  "pares_candidatos": len(candidates),
855
1094
  "pares_comparados": len(score_cache),
856
- "pares_acima_limiar": qualifying_pairs,
1095
+ "pares_acima_limiar": len(qualifying_edges),
857
1096
  "modo_candidatos": mode,
1097
+ "algoritmo": algorithm,
1098
+ "clones_parciais": len(partial_clones),
1099
+ "politica_cluster": cluster_policy,
1100
+ "unioes_rejeitadas_por_coesao": rejected_by_cohesion,
1101
+ "unioes_rejeitadas_por_limite": cohesion_limit_rejections,
858
1102
  }
859
1103
  return SimilarityReport(
860
1104
  threshold=float(threshold),
861
1105
  shingle_size=shingle_size,
862
1106
  exhaustive_limit=exhaustive_limit,
1107
+ algorithm=algorithm,
1108
+ partial_threshold=float(partial_threshold),
1109
+ cluster_policy=cluster_policy,
863
1110
  records=records,
864
1111
  clusters=clusters,
865
1112
  statistics=statistics,
1113
+ partial_clones=partial_clones,
866
1114
  documents=tuple(ordered),
867
1115
  )
868
1116
 
@@ -880,6 +1128,10 @@ def analyze_codebase(
880
1128
  threshold: float = 80.0,
881
1129
  shingle_size: int = 3,
882
1130
  exhaustive_limit: int = 80,
1131
+ algorithm: str = "advanced",
1132
+ partial_threshold: float = 75.0,
1133
+ min_clone_tokens: int = 30,
1134
+ cluster_policy: str = "connected",
883
1135
  ignored_directories: Iterable[str] = _IGNORED_DIRECTORIES,
884
1136
  ) -> SimilarityReport:
885
1137
  return analyze_documents(
@@ -887,6 +1139,10 @@ def analyze_codebase(
887
1139
  threshold=threshold,
888
1140
  shingle_size=shingle_size,
889
1141
  exhaustive_limit=exhaustive_limit,
1142
+ algorithm=algorithm,
1143
+ partial_threshold=partial_threshold,
1144
+ min_clone_tokens=min_clone_tokens,
1145
+ cluster_policy=cluster_policy,
890
1146
  )
891
1147
 
892
1148
 
@@ -972,7 +1228,7 @@ def build_ai_review_payload(
972
1228
  )
973
1229
 
974
1230
  return {
975
- "schema_version": 1,
1231
+ "schema_version": 2,
976
1232
  "tarefa": (
977
1233
  "Revise cada cluster pela semantica do codigo e retorne um parecer "
978
1234
  "similar, identico ou diferente, com justificativa breve."
@@ -988,6 +1244,7 @@ def build_ai_review_payload(
988
1244
  ]
989
1245
  },
990
1246
  "clusters": clusters,
1247
+ "clones_parciais": deepcopy(payload.get("clones_parciais", [])),
991
1248
  }
992
1249
 
993
1250