csim 3.4.0__tar.gz → 3.4.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (132) hide show
  1. {csim-3.4.0/csim.egg-info → csim-3.4.2}/PKG-INFO +11 -1
  2. {csim-3.4.0 → csim-3.4.2}/README.md +10 -0
  3. {csim-3.4.0 → csim-3.4.2}/csim/__init__.py +1 -1
  4. {csim-3.4.0 → csim-3.4.2}/csim/main.py +3 -1
  5. {csim-3.4.0 → csim-3.4.2}/csim/processing/distance_metrics.py +43 -21
  6. {csim-3.4.0 → csim-3.4.2}/csim/processing/tree_processing.py +20 -3
  7. {csim-3.4.0 → csim-3.4.2}/csim/python_3/utils.py +4 -0
  8. {csim-3.4.0 → csim-3.4.2}/csim/utils.py +60 -0
  9. {csim-3.4.0 → csim-3.4.2/csim.egg-info}/PKG-INFO +11 -1
  10. {csim-3.4.0 → csim-3.4.2}/setup.py +1 -1
  11. {csim-3.4.0 → csim-3.4.2}/LICENSE +0 -0
  12. {csim-3.4.0 → csim-3.4.2}/MANIFEST.in +0 -0
  13. {csim-3.4.0 → csim-3.4.2}/csim/CodeSimilarity.py +0 -0
  14. {csim-3.4.0 → csim-3.4.2}/csim/DataStructures.py +0 -0
  15. {csim-3.4.0 → csim-3.4.2}/csim/Visitors.py +0 -0
  16. {csim-3.4.0 → csim-3.4.2}/csim/c/CLexer.py +0 -0
  17. {csim-3.4.0 → csim-3.4.2}/csim/c/CLexer.tokens +0 -0
  18. {csim-3.4.0 → csim-3.4.2}/csim/c/CLexerBase.py +0 -0
  19. {csim-3.4.0 → csim-3.4.2}/csim/c/CParser.py +0 -0
  20. {csim-3.4.0 → csim-3.4.2}/csim/c/CParser.tokens +0 -0
  21. {csim-3.4.0 → csim-3.4.2}/csim/c/CParserBase.py +0 -0
  22. {csim-3.4.0 → csim-3.4.2}/csim/c/CParserVisitor.py +0 -0
  23. {csim-3.4.0 → csim-3.4.2}/csim/c/ErrorListener.py +0 -0
  24. {csim-3.4.0 → csim-3.4.2}/csim/c/Symbol.py +0 -0
  25. {csim-3.4.0 → csim-3.4.2}/csim/c/SymbolTable.py +0 -0
  26. {csim-3.4.0 → csim-3.4.2}/csim/c/TypeClassification.py +0 -0
  27. {csim-3.4.0 → csim-3.4.2}/csim/c/__init__.py +0 -0
  28. {csim-3.4.0 → csim-3.4.2}/csim/c/utils.py +0 -0
  29. {csim-3.4.0 → csim-3.4.2}/csim/cpp_14/CPP14Lexer.py +0 -0
  30. {csim-3.4.0 → csim-3.4.2}/csim/cpp_14/CPP14Lexer.tokens +0 -0
  31. {csim-3.4.0 → csim-3.4.2}/csim/cpp_14/CPP14Parser.py +0 -0
  32. {csim-3.4.0 → csim-3.4.2}/csim/cpp_14/CPP14Parser.tokens +0 -0
  33. {csim-3.4.0 → csim-3.4.2}/csim/cpp_14/CPP14ParserBase.py +0 -0
  34. {csim-3.4.0 → csim-3.4.2}/csim/cpp_14/CPP14ParserVisitor.py +0 -0
  35. {csim-3.4.0 → csim-3.4.2}/csim/cpp_14/__init__.py +0 -0
  36. {csim-3.4.0 → csim-3.4.2}/csim/cpp_14/transformGrammar.py +0 -0
  37. {csim-3.4.0 → csim-3.4.2}/csim/cpp_14/utils.py +0 -0
  38. {csim-3.4.0 → csim-3.4.2}/csim/java_20/Java20Lexer.py +0 -0
  39. {csim-3.4.0 → csim-3.4.2}/csim/java_20/Java20Lexer.tokens +0 -0
  40. {csim-3.4.0 → csim-3.4.2}/csim/java_20/Java20Parser.py +0 -0
  41. {csim-3.4.0 → csim-3.4.2}/csim/java_20/Java20Parser.tokens +0 -0
  42. {csim-3.4.0 → csim-3.4.2}/csim/java_20/Java20ParserVisitor.py +0 -0
  43. {csim-3.4.0 → csim-3.4.2}/csim/java_20/__init__.py +0 -0
  44. {csim-3.4.0 → csim-3.4.2}/csim/java_20/utils.py +0 -0
  45. {csim-3.4.0 → csim-3.4.2}/csim/java_24/Java24Lexer.py +0 -0
  46. {csim-3.4.0 → csim-3.4.2}/csim/java_24/Java24Lexer.tokens +0 -0
  47. {csim-3.4.0 → csim-3.4.2}/csim/java_24/Java24LexerBase.py +0 -0
  48. {csim-3.4.0 → csim-3.4.2}/csim/java_24/Java24Parser.py +0 -0
  49. {csim-3.4.0 → csim-3.4.2}/csim/java_24/Java24Parser.tokens +0 -0
  50. {csim-3.4.0 → csim-3.4.2}/csim/java_24/Java24ParserBase.py +0 -0
  51. {csim-3.4.0 → csim-3.4.2}/csim/java_24/Java24ParserVisitor.py +0 -0
  52. {csim-3.4.0 → csim-3.4.2}/csim/java_24/__init__.py +0 -0
  53. {csim-3.4.0 → csim-3.4.2}/csim/java_24/utils.py +0 -0
  54. {csim-3.4.0 → csim-3.4.2}/csim/kotlin/KotlinLexer.py +0 -0
  55. {csim-3.4.0 → csim-3.4.2}/csim/kotlin/KotlinLexer.tokens +0 -0
  56. {csim-3.4.0 → csim-3.4.2}/csim/kotlin/KotlinParser.py +0 -0
  57. {csim-3.4.0 → csim-3.4.2}/csim/kotlin/KotlinParser.tokens +0 -0
  58. {csim-3.4.0 → csim-3.4.2}/csim/kotlin/KotlinParserVisitor.py +0 -0
  59. {csim-3.4.0 → csim-3.4.2}/csim/kotlin/__init__.py +0 -0
  60. {csim-3.4.0 → csim-3.4.2}/csim/kotlin/utils.py +0 -0
  61. {csim-3.4.0 → csim-3.4.2}/csim/language/__init__.py +0 -0
  62. {csim-3.4.0 → csim-3.4.2}/csim/language/lexer.py +0 -0
  63. {csim-3.4.0 → csim-3.4.2}/csim/language/parser.py +0 -0
  64. {csim-3.4.0 → csim-3.4.2}/csim/native/__init__.py +0 -0
  65. {csim-3.4.0 → csim-3.4.2}/csim/native/loader.py +0 -0
  66. {csim-3.4.0 → csim-3.4.2}/csim/native/src/c_bridge.cpp +0 -0
  67. {csim-3.4.0 → csim-3.4.2}/csim/native/src/cpp_14_bridge.cpp +0 -0
  68. {csim-3.4.0 → csim-3.4.2}/csim/native/src/java_20_bridge.cpp +0 -0
  69. {csim-3.4.0 → csim-3.4.2}/csim/native/src/java_24_bridge.cpp +0 -0
  70. {csim-3.4.0 → csim-3.4.2}/csim/native/src/kotlin_bridge.cpp +0 -0
  71. {csim-3.4.0 → csim-3.4.2}/csim/native/src/python_3_bridge.cpp +0 -0
  72. {csim-3.4.0 → csim-3.4.2}/csim/native/tree_builder.py +0 -0
  73. {csim-3.4.0 → csim-3.4.2}/csim/processing/__init__.py +0 -0
  74. {csim-3.4.0 → csim-3.4.2}/csim/python_3/Python3Lexer.py +0 -0
  75. {csim-3.4.0 → csim-3.4.2}/csim/python_3/Python3Lexer.tokens +0 -0
  76. {csim-3.4.0 → csim-3.4.2}/csim/python_3/Python3LexerBase.py +0 -0
  77. {csim-3.4.0 → csim-3.4.2}/csim/python_3/Python3Parser.py +0 -0
  78. {csim-3.4.0 → csim-3.4.2}/csim/python_3/Python3Parser.tokens +0 -0
  79. {csim-3.4.0 → csim-3.4.2}/csim/python_3/Python3ParserBase.py +0 -0
  80. {csim-3.4.0 → csim-3.4.2}/csim/python_3/Python3ParserVisitor.py +0 -0
  81. {csim-3.4.0 → csim-3.4.2}/csim/python_3/__init__.py +0 -0
  82. {csim-3.4.0 → csim-3.4.2}/csim/python_3_13/PythonLexer.py +0 -0
  83. {csim-3.4.0 → csim-3.4.2}/csim/python_3_13/PythonLexer.tokens +0 -0
  84. {csim-3.4.0 → csim-3.4.2}/csim/python_3_13/PythonLexerBase.py +0 -0
  85. {csim-3.4.0 → csim-3.4.2}/csim/python_3_13/PythonParser.py +0 -0
  86. {csim-3.4.0 → csim-3.4.2}/csim/python_3_13/PythonParser.tokens +0 -0
  87. {csim-3.4.0 → csim-3.4.2}/csim/python_3_13/PythonParserVisitor.py +0 -0
  88. {csim-3.4.0 → csim-3.4.2}/csim/python_3_13/__init__.py +0 -0
  89. {csim-3.4.0 → csim-3.4.2}/csim/python_3_13/utils.py +0 -0
  90. {csim-3.4.0 → csim-3.4.2}/csim.egg-info/SOURCES.txt +0 -0
  91. {csim-3.4.0 → csim-3.4.2}/csim.egg-info/dependency_links.txt +0 -0
  92. {csim-3.4.0 → csim-3.4.2}/csim.egg-info/entry_points.txt +0 -0
  93. {csim-3.4.0 → csim-3.4.2}/csim.egg-info/requires.txt +0 -0
  94. {csim-3.4.0 → csim-3.4.2}/csim.egg-info/top_level.txt +0 -0
  95. {csim-3.4.0 → csim-3.4.2}/grammars/CLexer.g4 +0 -0
  96. {csim-3.4.0 → csim-3.4.2}/grammars/CLexerBase.cpp +0 -0
  97. {csim-3.4.0 → csim-3.4.2}/grammars/CLexerBase.h +0 -0
  98. {csim-3.4.0 → csim-3.4.2}/grammars/CPP14Lexer.g4 +0 -0
  99. {csim-3.4.0 → csim-3.4.2}/grammars/CPP14Parser.g4 +0 -0
  100. {csim-3.4.0 → csim-3.4.2}/grammars/CPP14ParserBase.cpp +0 -0
  101. {csim-3.4.0 → csim-3.4.2}/grammars/CPP14ParserBase.h +0 -0
  102. {csim-3.4.0 → csim-3.4.2}/grammars/CParser.g4 +0 -0
  103. {csim-3.4.0 → csim-3.4.2}/grammars/CParserBase.cpp +0 -0
  104. {csim-3.4.0 → csim-3.4.2}/grammars/CParserBase.h +0 -0
  105. {csim-3.4.0 → csim-3.4.2}/grammars/Java20Lexer.g4 +0 -0
  106. {csim-3.4.0 → csim-3.4.2}/grammars/Java20Parser.g4 +0 -0
  107. {csim-3.4.0 → csim-3.4.2}/grammars/Java24Lexer.g4 +0 -0
  108. {csim-3.4.0 → csim-3.4.2}/grammars/Java24Parser.g4 +0 -0
  109. {csim-3.4.0 → csim-3.4.2}/grammars/Java24ParserBase.cpp +0 -0
  110. {csim-3.4.0 → csim-3.4.2}/grammars/Java24ParserBase.h +0 -0
  111. {csim-3.4.0 → csim-3.4.2}/grammars/KotlinLexer.g4 +0 -0
  112. {csim-3.4.0 → csim-3.4.2}/grammars/KotlinParser.g4 +0 -0
  113. {csim-3.4.0 → csim-3.4.2}/grammars/Python3Lexer.g4 +0 -0
  114. {csim-3.4.0 → csim-3.4.2}/grammars/Python3LexerBase.cpp +0 -0
  115. {csim-3.4.0 → csim-3.4.2}/grammars/Python3LexerBase.h +0 -0
  116. {csim-3.4.0 → csim-3.4.2}/grammars/Python3Parser.g4 +0 -0
  117. {csim-3.4.0 → csim-3.4.2}/grammars/Python3ParserBase.cpp +0 -0
  118. {csim-3.4.0 → csim-3.4.2}/grammars/Python3ParserBase.h +0 -0
  119. {csim-3.4.0 → csim-3.4.2}/grammars/PythonLexer.g4 +0 -0
  120. {csim-3.4.0 → csim-3.4.2}/grammars/PythonParser.g4 +0 -0
  121. {csim-3.4.0 → csim-3.4.2}/grammars/Symbol.h +0 -0
  122. {csim-3.4.0 → csim-3.4.2}/grammars/SymbolTable.cpp +0 -0
  123. {csim-3.4.0 → csim-3.4.2}/grammars/SymbolTable.h +0 -0
  124. {csim-3.4.0 → csim-3.4.2}/grammars/TypeClassification.h +0 -0
  125. {csim-3.4.0 → csim-3.4.2}/grammars/UnicodeClasses.g4 +0 -0
  126. {csim-3.4.0 → csim-3.4.2}/grammars/parser_gen_guide.md +0 -0
  127. {csim-3.4.0 → csim-3.4.2}/scripts/build_native_parsers.sh +0 -0
  128. {csim-3.4.0 → csim-3.4.2}/scripts/transform_grammar_for_cpp.py +0 -0
  129. {csim-3.4.0 → csim-3.4.2}/setup.cfg +0 -0
  130. {csim-3.4.0 → csim-3.4.2}/test/test_cli.py +0 -0
  131. {csim-3.4.0 → csim-3.4.2}/test/test_module.py +0 -0
  132. {csim-3.4.0 → csim-3.4.2}/test/test_native_parsers.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: csim
3
- Version: 3.4.0
3
+ Version: 3.4.2
4
4
  Summary: Code Similarity (csim) is a method designed to detect similarity between source codes
5
5
  Home-page: https://github.com/EdsonEddy/csim
6
6
  Author: Eddy Lecoña
@@ -363,6 +363,16 @@ similarity = Compare(name_a='example A', content_a=code_a, name_b='example B', c
363
363
  print(f"Similarity: {similarity}") # Output: Similarity: X.XX
364
364
  ```
365
365
 
366
+ To see how much a program shrinks when it is normalized, pruned and hashed, count its nodes before and after:
367
+
368
+ ```python
369
+ from csim import count_nodes
370
+
371
+ nodes_before, nodes_after = count_nodes("example.py", code, lang="python_3_13")
372
+ ```
373
+
374
+ `nodes_before` is every node of the raw ANTLR parse tree; `nodes_after` is the size of the tree handed to the tree edit distance (the same number `csim tree` prints as "Total nodes after pruning").
375
+
366
376
  ## Documentation
367
377
 
368
378
  - [Getting Started Guide](GETTING_STARTED.md) - Quick tutorial for new users
@@ -324,6 +324,16 @@ similarity = Compare(name_a='example A', content_a=code_a, name_b='example B', c
324
324
  print(f"Similarity: {similarity}") # Output: Similarity: X.XX
325
325
  ```
326
326
 
327
+ To see how much a program shrinks when it is normalized, pruned and hashed, count its nodes before and after:
328
+
329
+ ```python
330
+ from csim import count_nodes
331
+
332
+ nodes_before, nodes_after = count_nodes("example.py", code, lang="python_3_13")
333
+ ```
334
+
335
+ `nodes_before` is every node of the raw ANTLR parse tree; `nodes_after` is the size of the tree handed to the tree edit distance (the same number `csim tree` prints as "Total nodes after pruning").
336
+
327
337
  ## Documentation
328
338
 
329
339
  - [Getting Started Guide](GETTING_STARTED.md) - Quick tutorial for new users
@@ -2,4 +2,4 @@ from .CodeSimilarity import Compare
2
2
  from .language.parser import ANTLR_parse
3
3
  from .processing.tree_processing import Normalize, PruneAndHash
4
4
  from .processing.distance_metrics import SimilarityIndex
5
- from .utils import group_by_exhaustive_search, report_pairwise_similarity
5
+ from .utils import count_nodes, group_by_exhaustive_search, report_pairwise_similarity
@@ -3,6 +3,7 @@ import os
3
3
  from .language.parser import ANTLR_parse
4
4
  from .processing.tree_processing import Normalize, PruneAndHash
5
5
  from .utils import (
6
+ count_tree_nodes,
6
7
  group_by_exhaustive_search,
7
8
  print_antlr_tree,
8
9
  print_tree,
@@ -179,7 +180,8 @@ def main():
179
180
  print()
180
181
 
181
182
  normalized_tree = Normalize(raw_tree, args.lang)
182
- pruned_tree, node_count = PruneAndHash(normalized_tree, args.lang)
183
+ pruned_tree, _ = PruneAndHash(normalized_tree, args.lang)
184
+ node_count = count_tree_nodes(pruned_tree)
183
185
 
184
186
  print("=== Normalized + Pruned Tree (input to Tree Edit Distance) ===")
185
187
  print_tree(pruned_tree, lang=args.lang)
@@ -1,4 +1,4 @@
1
- from zss import simple_distance
1
+ from zss import distance as zss_distance
2
2
 
3
3
 
4
4
  def label_distance(label1, label2):
@@ -36,6 +36,35 @@ def label_distance(label1, label2):
36
36
  return 1.0
37
37
 
38
38
 
39
+ def node_weight(node):
40
+ """Cost of inserting or deleting a node: 1, or the mass a hashed node kept."""
41
+ return node.get("weight", 1)
42
+
43
+
44
+ def rename_cost(node1, node2):
45
+ """Substitution cost between two tree nodes (dicts with a 'label').
46
+
47
+ Same as label_distance, except two weighted hashed nodes of the same rule
48
+ but different content are charged by how much of their content differs
49
+ (multiset overlap of the labels they replaced) instead of a flat 0.5.
50
+ """
51
+ sig1, sig2 = node1.get("sig"), node2.get("sig")
52
+ if sig1 is None or sig2 is None:
53
+ return label_distance(node1["label"], node2["label"])
54
+ if node1["label"] == node2["label"]:
55
+ return 0.0
56
+ rule1, hash1 = str(node1["label"]).split("|", 1)
57
+ rule2, hash2 = str(node2["label"]).split("|", 1)
58
+ heavy = max(node1["weight"], node2["weight"])
59
+ if rule1 != rule2:
60
+ return heavy if hash1 != hash2 else 0.5 * heavy
61
+ common = sum((sig1 & sig2).values())
62
+ total = sum(sig1.values()) + sum(sig2.values())
63
+ # Different digest means different content even if the multisets match
64
+ # (same labels, other order), so it never costs less than the old flat 0.5.
65
+ return max(heavy * (1.0 - 2.0 * common / total), 0.5)
66
+
67
+
39
68
  def TreeEditDistance(N1, N2, ted_algorithm="apted"):
40
69
  """Calculate the tree edit distance between two trees using the specified algorithm.
41
70
  Args:
@@ -46,27 +75,14 @@ def TreeEditDistance(N1, N2, ted_algorithm="apted"):
46
75
  int: The computed tree edit distance between the two trees.
47
76
  """
48
77
  if ted_algorithm == "zss":
49
- # Custom configuration for zss to work with dictionaries
50
- class CustomConfigZss:
51
- @staticmethod
52
- def get_children(node):
53
- return node["children"]
54
-
55
- @staticmethod
56
- def get_label(node):
57
- return node["label"]
58
-
59
- @staticmethod
60
- def label_dist(label1, label2):
61
- """Compares two labels"""
62
- return label_distance(label1, label2)
63
-
64
- d = simple_distance(
78
+ # zss takes per-node cost functions, so weighted hashed nodes work too
79
+ d = zss_distance(
65
80
  N1,
66
81
  N2,
67
- get_children=CustomConfigZss.get_children,
68
- get_label=CustomConfigZss.get_label,
69
- label_dist=CustomConfigZss.label_dist,
82
+ get_children=lambda node: node["children"],
83
+ insert_cost=node_weight,
84
+ remove_cost=node_weight,
85
+ update_cost=rename_cost,
70
86
  )
71
87
  elif ted_algorithm == "apted":
72
88
  from apted import APTED, Config
@@ -75,7 +91,13 @@ def TreeEditDistance(N1, N2, ted_algorithm="apted"):
75
91
  class CustomConfigApted(Config):
76
92
  def rename(self, node1, node2):
77
93
  """Compares attribute .value of trees"""
78
- return label_distance(node1["label"], node2["label"])
94
+ return rename_cost(node1, node2)
95
+
96
+ def delete(self, node):
97
+ return node_weight(node)
98
+
99
+ def insert(self, node):
100
+ return node_weight(node)
79
101
 
80
102
  def children(self, node):
81
103
  """Get childrens of a node"""
@@ -1,4 +1,5 @@
1
1
  import hashlib
2
+ from collections import Counter
2
3
  from antlr4 import TerminalNode
3
4
  from ..Visitors import (
4
5
  Python_3_13_ParserVisitorExtended,
@@ -16,6 +17,7 @@ from ..utils import (
16
17
  get_excluded_token_types,
17
18
  get_hash_rule_indices,
18
19
  get_excluded_rule_types,
20
+ get_hash_mass_alpha,
19
21
  get_relabel_fn,
20
22
  get_structural_rule_indices,
21
23
  )
@@ -121,6 +123,9 @@ def PruneAndHash(tree, lang):
121
123
  control_equivalence_rule_indices = get_control_equivalence_rule_indices(lang)
122
124
  exclude_childrens_from_rule = get_exclude_childrens_from_rule(lang)
123
125
  structural_rule_indices = get_structural_rule_indices(lang)
126
+ # When set, a hashed node keeps the mass of the subtree it replaced (see
127
+ # docs/pruning_fidelity.md, "Weighted hashes"); None keeps 1 node = 1.
128
+ mass_alpha = get_hash_mass_alpha(lang)
124
129
 
125
130
  def traverse_subtree(node):
126
131
  # Collect all labels in the subtree rooted at `node` into a single list
@@ -155,7 +160,8 @@ def PruneAndHash(tree, lang):
155
160
  for c in childrens:
156
161
  flat.extend(traverse_subtree(c))
157
162
  s = "|".join(map(str, flat))
158
- return str(label) + "|" + hashlib.sha256(s.encode("utf-8")).hexdigest()
163
+ digest = str(label) + "|" + hashlib.sha256(s.encode("utf-8")).hexdigest()
164
+ return digest, flat
159
165
 
160
166
  # Ids of nodes whose subtree contains a structural (control-flow/body)
161
167
  # label. A hashed rule that contains one is NOT collapsed: its content is
@@ -184,8 +190,19 @@ def PruneAndHash(tree, lang):
184
190
 
185
191
  # For nodes that are in the hashed rule set, we hash their entire subtree to a single digest
186
192
  if node["label"] in hashed_rule_indices and id(node) not in has_structure:
187
- digest = hash_children(label, node["children"])
188
- return {"label": digest, "children": []}, 1
193
+ digest, flat = hash_children(label, node["children"])
194
+ if mass_alpha is None:
195
+ return {"label": digest, "children": []}, 1
196
+ # Weight and label multiset let the edit distance charge a hashed
197
+ # node in proportion to what it replaced and give partial credit
198
+ # to two hashed nodes of the same kind that share most content.
199
+ weight = (len(flat) + 1) ** mass_alpha
200
+ return {
201
+ "label": digest,
202
+ "children": [],
203
+ "weight": weight,
204
+ "sig": Counter(flat),
205
+ }, weight
189
206
 
190
207
  new_node = {"label": label, "children": []}
191
208
  count = 1
@@ -252,6 +252,10 @@ HASHED_RULE_INDICES = {
252
252
  # without hashing anything.
253
253
  }
254
254
 
255
+ # Exponent for the weight of a hashed node (its subtree size ** alpha) in the
256
+ # tree edit distance; see docs/pruning_fidelity.md, "Weighted hashes".
257
+ HASH_MASS_ALPHA = 0.6
258
+
255
259
  # `for` and `while` are interchangeable ways to write the same loop (the
256
260
  # jv-umsa-dataset/controlled clones rewrite one as the other), so both get the
257
261
  # same label; mirrors python_3_13 and the 2.0.0 behaviour.
@@ -378,6 +378,19 @@ def get_relabel_fn(lang):
378
378
  return None
379
379
 
380
380
 
381
+ def get_hash_mass_alpha(lang):
382
+ """Exponent applied to the size of a hashed subtree to get its weight in
383
+ the edit distance, or None when the language keeps every node at weight 1."""
384
+ import importlib
385
+
386
+ if lang not in (
387
+ "python_3_13", "python_3", "java_20", "java_24", "cpp_14", "kotlin", "c",
388
+ ):
389
+ return None
390
+ module = importlib.import_module(f".{lang}.utils", package=__package__)
391
+ return getattr(module, "HASH_MASS_ALPHA", None)
392
+
393
+
381
394
  def get_structural_rule_indices(lang):
382
395
  """Retrieve the rules that mark a subtree as structural (control flow,
383
396
  bodies, declarations of callables/types) for a language.
@@ -506,6 +519,53 @@ def preprocess_code(file_name, file_content, lang="python_3_13"):
506
519
  return pruned_tree, pruned_count
507
520
 
508
521
 
522
+ def count_tree_nodes(tree):
523
+ """Number of nodes of a normalized/pruned tree (iterative, deep-safe).
524
+
525
+ PruneAndHash's second value is the tree's *mass* when a language weights
526
+ its hashed nodes, so the plain node count has to be taken from the tree.
527
+ """
528
+ count = 0
529
+ stack = [tree]
530
+ while stack:
531
+ node = stack.pop()
532
+ count += 1
533
+ stack.extend(node["children"])
534
+ return count
535
+
536
+
537
+ def count_nodes(file_name, file_content, lang="python_3_13"):
538
+ """Count the nodes of a program before and after pruning.
539
+
540
+ Args:
541
+ file_name (str): Name of the file (used only for syntax-error messages).
542
+ file_content (str): Source code.
543
+ lang (str): Programming language identifier.
544
+
545
+ Returns:
546
+ tuple[int, int]: (nodes_before, nodes_after). `nodes_before` is every
547
+ node of the raw ANTLR parse tree (rules and tokens); `nodes_after` is
548
+ the size of the normalized, pruned and hashed tree that is handed to
549
+ the tree edit distance -- the same number `csim tree` prints as
550
+ "Total nodes after pruning".
551
+ """
552
+ # Local import to avoid circular dependency at module import time
553
+ from .CodeSimilarity import ANTLR_parse, Normalize, PruneAndHash
554
+
555
+ tree = ANTLR_parse(file_name, file_content, lang)
556
+
557
+ # Iterative: real programs can nest deeper than the recursion limit.
558
+ before = 0
559
+ stack = [tree]
560
+ while stack:
561
+ node = stack.pop()
562
+ before += 1
563
+ stack.extend(node.getChild(i) for i in range(node.getChildCount()))
564
+
565
+ pruned, _ = PruneAndHash(Normalize(tree, lang), lang)
566
+ return before, count_tree_nodes(pruned)
567
+
568
+
509
569
  def get_similarity_coefficient(proccesed_code1, proccesed_code2, ted_algorithm):
510
570
  N1, len_N1 = proccesed_code1
511
571
  N2, len_N2 = proccesed_code2
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: csim
3
- Version: 3.4.0
3
+ Version: 3.4.2
4
4
  Summary: Code Similarity (csim) is a method designed to detect similarity between source codes
5
5
  Home-page: https://github.com/EdsonEddy/csim
6
6
  Author: Eddy Lecoña
@@ -363,6 +363,16 @@ similarity = Compare(name_a='example A', content_a=code_a, name_b='example B', c
363
363
  print(f"Similarity: {similarity}") # Output: Similarity: X.XX
364
364
  ```
365
365
 
366
+ To see how much a program shrinks when it is normalized, pruned and hashed, count its nodes before and after:
367
+
368
+ ```python
369
+ from csim import count_nodes
370
+
371
+ nodes_before, nodes_after = count_nodes("example.py", code, lang="python_3_13")
372
+ ```
373
+
374
+ `nodes_before` is every node of the raw ANTLR parse tree; `nodes_after` is the size of the tree handed to the tree edit distance (the same number `csim tree` prints as "Total nodes after pruning").
375
+
366
376
  ## Documentation
367
377
 
368
378
  - [Getting Started Guide](GETTING_STARTED.md) - Quick tutorial for new users
@@ -49,7 +49,7 @@ except ImportError:
49
49
 
50
50
  setup(
51
51
  name="csim",
52
- version="3.4.0",
52
+ version="3.4.2",
53
53
  packages=find_packages(),
54
54
  package_data={
55
55
  # Compiled native parsers, when built (scripts/build_native_parsers.sh).
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes