csim 3.4.0__tar.gz → 3.4.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {csim-3.4.0/csim.egg-info → csim-3.4.2}/PKG-INFO +11 -1
- {csim-3.4.0 → csim-3.4.2}/README.md +10 -0
- {csim-3.4.0 → csim-3.4.2}/csim/__init__.py +1 -1
- {csim-3.4.0 → csim-3.4.2}/csim/main.py +3 -1
- {csim-3.4.0 → csim-3.4.2}/csim/processing/distance_metrics.py +43 -21
- {csim-3.4.0 → csim-3.4.2}/csim/processing/tree_processing.py +20 -3
- {csim-3.4.0 → csim-3.4.2}/csim/python_3/utils.py +4 -0
- {csim-3.4.0 → csim-3.4.2}/csim/utils.py +60 -0
- {csim-3.4.0 → csim-3.4.2/csim.egg-info}/PKG-INFO +11 -1
- {csim-3.4.0 → csim-3.4.2}/setup.py +1 -1
- {csim-3.4.0 → csim-3.4.2}/LICENSE +0 -0
- {csim-3.4.0 → csim-3.4.2}/MANIFEST.in +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim/CodeSimilarity.py +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim/DataStructures.py +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim/Visitors.py +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim/c/CLexer.py +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim/c/CLexer.tokens +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim/c/CLexerBase.py +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim/c/CParser.py +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim/c/CParser.tokens +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim/c/CParserBase.py +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim/c/CParserVisitor.py +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim/c/ErrorListener.py +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim/c/Symbol.py +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim/c/SymbolTable.py +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim/c/TypeClassification.py +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim/c/__init__.py +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim/c/utils.py +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim/cpp_14/CPP14Lexer.py +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim/cpp_14/CPP14Lexer.tokens +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim/cpp_14/CPP14Parser.py +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim/cpp_14/CPP14Parser.tokens +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim/cpp_14/CPP14ParserBase.py +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim/cpp_14/CPP14ParserVisitor.py +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim/cpp_14/__init__.py +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim/cpp_14/transformGrammar.py +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim/cpp_14/utils.py +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim/java_20/Java20Lexer.py +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim/java_20/Java20Lexer.tokens +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim/java_20/Java20Parser.py +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim/java_20/Java20Parser.tokens +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim/java_20/Java20ParserVisitor.py +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim/java_20/__init__.py +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim/java_20/utils.py +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim/java_24/Java24Lexer.py +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim/java_24/Java24Lexer.tokens +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim/java_24/Java24LexerBase.py +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim/java_24/Java24Parser.py +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim/java_24/Java24Parser.tokens +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim/java_24/Java24ParserBase.py +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim/java_24/Java24ParserVisitor.py +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim/java_24/__init__.py +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim/java_24/utils.py +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim/kotlin/KotlinLexer.py +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim/kotlin/KotlinLexer.tokens +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim/kotlin/KotlinParser.py +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim/kotlin/KotlinParser.tokens +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim/kotlin/KotlinParserVisitor.py +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim/kotlin/__init__.py +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim/kotlin/utils.py +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim/language/__init__.py +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim/language/lexer.py +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim/language/parser.py +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim/native/__init__.py +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim/native/loader.py +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim/native/src/c_bridge.cpp +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim/native/src/cpp_14_bridge.cpp +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim/native/src/java_20_bridge.cpp +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim/native/src/java_24_bridge.cpp +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim/native/src/kotlin_bridge.cpp +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim/native/src/python_3_bridge.cpp +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim/native/tree_builder.py +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim/processing/__init__.py +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim/python_3/Python3Lexer.py +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim/python_3/Python3Lexer.tokens +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim/python_3/Python3LexerBase.py +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim/python_3/Python3Parser.py +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim/python_3/Python3Parser.tokens +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim/python_3/Python3ParserBase.py +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim/python_3/Python3ParserVisitor.py +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim/python_3/__init__.py +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim/python_3_13/PythonLexer.py +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim/python_3_13/PythonLexer.tokens +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim/python_3_13/PythonLexerBase.py +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim/python_3_13/PythonParser.py +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim/python_3_13/PythonParser.tokens +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim/python_3_13/PythonParserVisitor.py +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim/python_3_13/__init__.py +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim/python_3_13/utils.py +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim.egg-info/SOURCES.txt +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim.egg-info/dependency_links.txt +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim.egg-info/entry_points.txt +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim.egg-info/requires.txt +0 -0
- {csim-3.4.0 → csim-3.4.2}/csim.egg-info/top_level.txt +0 -0
- {csim-3.4.0 → csim-3.4.2}/grammars/CLexer.g4 +0 -0
- {csim-3.4.0 → csim-3.4.2}/grammars/CLexerBase.cpp +0 -0
- {csim-3.4.0 → csim-3.4.2}/grammars/CLexerBase.h +0 -0
- {csim-3.4.0 → csim-3.4.2}/grammars/CPP14Lexer.g4 +0 -0
- {csim-3.4.0 → csim-3.4.2}/grammars/CPP14Parser.g4 +0 -0
- {csim-3.4.0 → csim-3.4.2}/grammars/CPP14ParserBase.cpp +0 -0
- {csim-3.4.0 → csim-3.4.2}/grammars/CPP14ParserBase.h +0 -0
- {csim-3.4.0 → csim-3.4.2}/grammars/CParser.g4 +0 -0
- {csim-3.4.0 → csim-3.4.2}/grammars/CParserBase.cpp +0 -0
- {csim-3.4.0 → csim-3.4.2}/grammars/CParserBase.h +0 -0
- {csim-3.4.0 → csim-3.4.2}/grammars/Java20Lexer.g4 +0 -0
- {csim-3.4.0 → csim-3.4.2}/grammars/Java20Parser.g4 +0 -0
- {csim-3.4.0 → csim-3.4.2}/grammars/Java24Lexer.g4 +0 -0
- {csim-3.4.0 → csim-3.4.2}/grammars/Java24Parser.g4 +0 -0
- {csim-3.4.0 → csim-3.4.2}/grammars/Java24ParserBase.cpp +0 -0
- {csim-3.4.0 → csim-3.4.2}/grammars/Java24ParserBase.h +0 -0
- {csim-3.4.0 → csim-3.4.2}/grammars/KotlinLexer.g4 +0 -0
- {csim-3.4.0 → csim-3.4.2}/grammars/KotlinParser.g4 +0 -0
- {csim-3.4.0 → csim-3.4.2}/grammars/Python3Lexer.g4 +0 -0
- {csim-3.4.0 → csim-3.4.2}/grammars/Python3LexerBase.cpp +0 -0
- {csim-3.4.0 → csim-3.4.2}/grammars/Python3LexerBase.h +0 -0
- {csim-3.4.0 → csim-3.4.2}/grammars/Python3Parser.g4 +0 -0
- {csim-3.4.0 → csim-3.4.2}/grammars/Python3ParserBase.cpp +0 -0
- {csim-3.4.0 → csim-3.4.2}/grammars/Python3ParserBase.h +0 -0
- {csim-3.4.0 → csim-3.4.2}/grammars/PythonLexer.g4 +0 -0
- {csim-3.4.0 → csim-3.4.2}/grammars/PythonParser.g4 +0 -0
- {csim-3.4.0 → csim-3.4.2}/grammars/Symbol.h +0 -0
- {csim-3.4.0 → csim-3.4.2}/grammars/SymbolTable.cpp +0 -0
- {csim-3.4.0 → csim-3.4.2}/grammars/SymbolTable.h +0 -0
- {csim-3.4.0 → csim-3.4.2}/grammars/TypeClassification.h +0 -0
- {csim-3.4.0 → csim-3.4.2}/grammars/UnicodeClasses.g4 +0 -0
- {csim-3.4.0 → csim-3.4.2}/grammars/parser_gen_guide.md +0 -0
- {csim-3.4.0 → csim-3.4.2}/scripts/build_native_parsers.sh +0 -0
- {csim-3.4.0 → csim-3.4.2}/scripts/transform_grammar_for_cpp.py +0 -0
- {csim-3.4.0 → csim-3.4.2}/setup.cfg +0 -0
- {csim-3.4.0 → csim-3.4.2}/test/test_cli.py +0 -0
- {csim-3.4.0 → csim-3.4.2}/test/test_module.py +0 -0
- {csim-3.4.0 → csim-3.4.2}/test/test_native_parsers.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: csim
|
|
3
|
-
Version: 3.4.
|
|
3
|
+
Version: 3.4.2
|
|
4
4
|
Summary: Code Similarity (csim) is a method designed to detect similarity between source codes
|
|
5
5
|
Home-page: https://github.com/EdsonEddy/csim
|
|
6
6
|
Author: Eddy Lecoña
|
|
@@ -363,6 +363,16 @@ similarity = Compare(name_a='example A', content_a=code_a, name_b='example B', c
|
|
|
363
363
|
print(f"Similarity: {similarity}") # Output: Similarity: X.XX
|
|
364
364
|
```
|
|
365
365
|
|
|
366
|
+
To see how much a program shrinks when it is normalized, pruned and hashed, count its nodes before and after:
|
|
367
|
+
|
|
368
|
+
```python
|
|
369
|
+
from csim import count_nodes
|
|
370
|
+
|
|
371
|
+
nodes_before, nodes_after = count_nodes("example.py", code, lang="python_3_13")
|
|
372
|
+
```
|
|
373
|
+
|
|
374
|
+
`nodes_before` is every node of the raw ANTLR parse tree; `nodes_after` is the size of the tree handed to the tree edit distance (the same number `csim tree` prints as "Total nodes after pruning").
|
|
375
|
+
|
|
366
376
|
## Documentation
|
|
367
377
|
|
|
368
378
|
- [Getting Started Guide](GETTING_STARTED.md) - Quick tutorial for new users
|
|
@@ -324,6 +324,16 @@ similarity = Compare(name_a='example A', content_a=code_a, name_b='example B', c
|
|
|
324
324
|
print(f"Similarity: {similarity}") # Output: Similarity: X.XX
|
|
325
325
|
```
|
|
326
326
|
|
|
327
|
+
To see how much a program shrinks when it is normalized, pruned and hashed, count its nodes before and after:
|
|
328
|
+
|
|
329
|
+
```python
|
|
330
|
+
from csim import count_nodes
|
|
331
|
+
|
|
332
|
+
nodes_before, nodes_after = count_nodes("example.py", code, lang="python_3_13")
|
|
333
|
+
```
|
|
334
|
+
|
|
335
|
+
`nodes_before` is every node of the raw ANTLR parse tree; `nodes_after` is the size of the tree handed to the tree edit distance (the same number `csim tree` prints as "Total nodes after pruning").
|
|
336
|
+
|
|
327
337
|
## Documentation
|
|
328
338
|
|
|
329
339
|
- [Getting Started Guide](GETTING_STARTED.md) - Quick tutorial for new users
|
|
@@ -2,4 +2,4 @@ from .CodeSimilarity import Compare
|
|
|
2
2
|
from .language.parser import ANTLR_parse
|
|
3
3
|
from .processing.tree_processing import Normalize, PruneAndHash
|
|
4
4
|
from .processing.distance_metrics import SimilarityIndex
|
|
5
|
-
from .utils import group_by_exhaustive_search, report_pairwise_similarity
|
|
5
|
+
from .utils import count_nodes, group_by_exhaustive_search, report_pairwise_similarity
|
|
@@ -3,6 +3,7 @@ import os
|
|
|
3
3
|
from .language.parser import ANTLR_parse
|
|
4
4
|
from .processing.tree_processing import Normalize, PruneAndHash
|
|
5
5
|
from .utils import (
|
|
6
|
+
count_tree_nodes,
|
|
6
7
|
group_by_exhaustive_search,
|
|
7
8
|
print_antlr_tree,
|
|
8
9
|
print_tree,
|
|
@@ -179,7 +180,8 @@ def main():
|
|
|
179
180
|
print()
|
|
180
181
|
|
|
181
182
|
normalized_tree = Normalize(raw_tree, args.lang)
|
|
182
|
-
pruned_tree,
|
|
183
|
+
pruned_tree, _ = PruneAndHash(normalized_tree, args.lang)
|
|
184
|
+
node_count = count_tree_nodes(pruned_tree)
|
|
183
185
|
|
|
184
186
|
print("=== Normalized + Pruned Tree (input to Tree Edit Distance) ===")
|
|
185
187
|
print_tree(pruned_tree, lang=args.lang)
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
from zss import
|
|
1
|
+
from zss import distance as zss_distance
|
|
2
2
|
|
|
3
3
|
|
|
4
4
|
def label_distance(label1, label2):
|
|
@@ -36,6 +36,35 @@ def label_distance(label1, label2):
|
|
|
36
36
|
return 1.0
|
|
37
37
|
|
|
38
38
|
|
|
39
|
+
def node_weight(node):
|
|
40
|
+
"""Cost of inserting or deleting a node: 1, or the mass a hashed node kept."""
|
|
41
|
+
return node.get("weight", 1)
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def rename_cost(node1, node2):
|
|
45
|
+
"""Substitution cost between two tree nodes (dicts with a 'label').
|
|
46
|
+
|
|
47
|
+
Same as label_distance, except two weighted hashed nodes of the same rule
|
|
48
|
+
but different content are charged by how much of their content differs
|
|
49
|
+
(multiset overlap of the labels they replaced) instead of a flat 0.5.
|
|
50
|
+
"""
|
|
51
|
+
sig1, sig2 = node1.get("sig"), node2.get("sig")
|
|
52
|
+
if sig1 is None or sig2 is None:
|
|
53
|
+
return label_distance(node1["label"], node2["label"])
|
|
54
|
+
if node1["label"] == node2["label"]:
|
|
55
|
+
return 0.0
|
|
56
|
+
rule1, hash1 = str(node1["label"]).split("|", 1)
|
|
57
|
+
rule2, hash2 = str(node2["label"]).split("|", 1)
|
|
58
|
+
heavy = max(node1["weight"], node2["weight"])
|
|
59
|
+
if rule1 != rule2:
|
|
60
|
+
return heavy if hash1 != hash2 else 0.5 * heavy
|
|
61
|
+
common = sum((sig1 & sig2).values())
|
|
62
|
+
total = sum(sig1.values()) + sum(sig2.values())
|
|
63
|
+
# Different digest means different content even if the multisets match
|
|
64
|
+
# (same labels, other order), so it never costs less than the old flat 0.5.
|
|
65
|
+
return max(heavy * (1.0 - 2.0 * common / total), 0.5)
|
|
66
|
+
|
|
67
|
+
|
|
39
68
|
def TreeEditDistance(N1, N2, ted_algorithm="apted"):
|
|
40
69
|
"""Calculate the tree edit distance between two trees using the specified algorithm.
|
|
41
70
|
Args:
|
|
@@ -46,27 +75,14 @@ def TreeEditDistance(N1, N2, ted_algorithm="apted"):
|
|
|
46
75
|
int: The computed tree edit distance between the two trees.
|
|
47
76
|
"""
|
|
48
77
|
if ted_algorithm == "zss":
|
|
49
|
-
#
|
|
50
|
-
|
|
51
|
-
@staticmethod
|
|
52
|
-
def get_children(node):
|
|
53
|
-
return node["children"]
|
|
54
|
-
|
|
55
|
-
@staticmethod
|
|
56
|
-
def get_label(node):
|
|
57
|
-
return node["label"]
|
|
58
|
-
|
|
59
|
-
@staticmethod
|
|
60
|
-
def label_dist(label1, label2):
|
|
61
|
-
"""Compares two labels"""
|
|
62
|
-
return label_distance(label1, label2)
|
|
63
|
-
|
|
64
|
-
d = simple_distance(
|
|
78
|
+
# zss takes per-node cost functions, so weighted hashed nodes work too
|
|
79
|
+
d = zss_distance(
|
|
65
80
|
N1,
|
|
66
81
|
N2,
|
|
67
|
-
get_children=
|
|
68
|
-
|
|
69
|
-
|
|
82
|
+
get_children=lambda node: node["children"],
|
|
83
|
+
insert_cost=node_weight,
|
|
84
|
+
remove_cost=node_weight,
|
|
85
|
+
update_cost=rename_cost,
|
|
70
86
|
)
|
|
71
87
|
elif ted_algorithm == "apted":
|
|
72
88
|
from apted import APTED, Config
|
|
@@ -75,7 +91,13 @@ def TreeEditDistance(N1, N2, ted_algorithm="apted"):
|
|
|
75
91
|
class CustomConfigApted(Config):
|
|
76
92
|
def rename(self, node1, node2):
|
|
77
93
|
"""Compares attribute .value of trees"""
|
|
78
|
-
return
|
|
94
|
+
return rename_cost(node1, node2)
|
|
95
|
+
|
|
96
|
+
def delete(self, node):
|
|
97
|
+
return node_weight(node)
|
|
98
|
+
|
|
99
|
+
def insert(self, node):
|
|
100
|
+
return node_weight(node)
|
|
79
101
|
|
|
80
102
|
def children(self, node):
|
|
81
103
|
"""Get childrens of a node"""
|
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
import hashlib
|
|
2
|
+
from collections import Counter
|
|
2
3
|
from antlr4 import TerminalNode
|
|
3
4
|
from ..Visitors import (
|
|
4
5
|
Python_3_13_ParserVisitorExtended,
|
|
@@ -16,6 +17,7 @@ from ..utils import (
|
|
|
16
17
|
get_excluded_token_types,
|
|
17
18
|
get_hash_rule_indices,
|
|
18
19
|
get_excluded_rule_types,
|
|
20
|
+
get_hash_mass_alpha,
|
|
19
21
|
get_relabel_fn,
|
|
20
22
|
get_structural_rule_indices,
|
|
21
23
|
)
|
|
@@ -121,6 +123,9 @@ def PruneAndHash(tree, lang):
|
|
|
121
123
|
control_equivalence_rule_indices = get_control_equivalence_rule_indices(lang)
|
|
122
124
|
exclude_childrens_from_rule = get_exclude_childrens_from_rule(lang)
|
|
123
125
|
structural_rule_indices = get_structural_rule_indices(lang)
|
|
126
|
+
# When set, a hashed node keeps the mass of the subtree it replaced (see
|
|
127
|
+
# docs/pruning_fidelity.md, "Weighted hashes"); None keeps 1 node = 1.
|
|
128
|
+
mass_alpha = get_hash_mass_alpha(lang)
|
|
124
129
|
|
|
125
130
|
def traverse_subtree(node):
|
|
126
131
|
# Collect all labels in the subtree rooted at `node` into a single list
|
|
@@ -155,7 +160,8 @@ def PruneAndHash(tree, lang):
|
|
|
155
160
|
for c in childrens:
|
|
156
161
|
flat.extend(traverse_subtree(c))
|
|
157
162
|
s = "|".join(map(str, flat))
|
|
158
|
-
|
|
163
|
+
digest = str(label) + "|" + hashlib.sha256(s.encode("utf-8")).hexdigest()
|
|
164
|
+
return digest, flat
|
|
159
165
|
|
|
160
166
|
# Ids of nodes whose subtree contains a structural (control-flow/body)
|
|
161
167
|
# label. A hashed rule that contains one is NOT collapsed: its content is
|
|
@@ -184,8 +190,19 @@ def PruneAndHash(tree, lang):
|
|
|
184
190
|
|
|
185
191
|
# For nodes that are in the hashed rule set, we hash their entire subtree to a single digest
|
|
186
192
|
if node["label"] in hashed_rule_indices and id(node) not in has_structure:
|
|
187
|
-
digest = hash_children(label, node["children"])
|
|
188
|
-
|
|
193
|
+
digest, flat = hash_children(label, node["children"])
|
|
194
|
+
if mass_alpha is None:
|
|
195
|
+
return {"label": digest, "children": []}, 1
|
|
196
|
+
# Weight and label multiset let the edit distance charge a hashed
|
|
197
|
+
# node in proportion to what it replaced and give partial credit
|
|
198
|
+
# to two hashed nodes of the same kind that share most content.
|
|
199
|
+
weight = (len(flat) + 1) ** mass_alpha
|
|
200
|
+
return {
|
|
201
|
+
"label": digest,
|
|
202
|
+
"children": [],
|
|
203
|
+
"weight": weight,
|
|
204
|
+
"sig": Counter(flat),
|
|
205
|
+
}, weight
|
|
189
206
|
|
|
190
207
|
new_node = {"label": label, "children": []}
|
|
191
208
|
count = 1
|
|
@@ -252,6 +252,10 @@ HASHED_RULE_INDICES = {
|
|
|
252
252
|
# without hashing anything.
|
|
253
253
|
}
|
|
254
254
|
|
|
255
|
+
# Exponent for the weight of a hashed node (its subtree size ** alpha) in the
|
|
256
|
+
# tree edit distance; see docs/pruning_fidelity.md, "Weighted hashes".
|
|
257
|
+
HASH_MASS_ALPHA = 0.6
|
|
258
|
+
|
|
255
259
|
# `for` and `while` are interchangeable ways to write the same loop (the
|
|
256
260
|
# jv-umsa-dataset/controlled clones rewrite one as the other), so both get the
|
|
257
261
|
# same label; mirrors python_3_13 and the 2.0.0 behaviour.
|
|
@@ -378,6 +378,19 @@ def get_relabel_fn(lang):
|
|
|
378
378
|
return None
|
|
379
379
|
|
|
380
380
|
|
|
381
|
+
def get_hash_mass_alpha(lang):
|
|
382
|
+
"""Exponent applied to the size of a hashed subtree to get its weight in
|
|
383
|
+
the edit distance, or None when the language keeps every node at weight 1."""
|
|
384
|
+
import importlib
|
|
385
|
+
|
|
386
|
+
if lang not in (
|
|
387
|
+
"python_3_13", "python_3", "java_20", "java_24", "cpp_14", "kotlin", "c",
|
|
388
|
+
):
|
|
389
|
+
return None
|
|
390
|
+
module = importlib.import_module(f".{lang}.utils", package=__package__)
|
|
391
|
+
return getattr(module, "HASH_MASS_ALPHA", None)
|
|
392
|
+
|
|
393
|
+
|
|
381
394
|
def get_structural_rule_indices(lang):
|
|
382
395
|
"""Retrieve the rules that mark a subtree as structural (control flow,
|
|
383
396
|
bodies, declarations of callables/types) for a language.
|
|
@@ -506,6 +519,53 @@ def preprocess_code(file_name, file_content, lang="python_3_13"):
|
|
|
506
519
|
return pruned_tree, pruned_count
|
|
507
520
|
|
|
508
521
|
|
|
522
|
+
def count_tree_nodes(tree):
|
|
523
|
+
"""Number of nodes of a normalized/pruned tree (iterative, deep-safe).
|
|
524
|
+
|
|
525
|
+
PruneAndHash's second value is the tree's *mass* when a language weights
|
|
526
|
+
its hashed nodes, so the plain node count has to be taken from the tree.
|
|
527
|
+
"""
|
|
528
|
+
count = 0
|
|
529
|
+
stack = [tree]
|
|
530
|
+
while stack:
|
|
531
|
+
node = stack.pop()
|
|
532
|
+
count += 1
|
|
533
|
+
stack.extend(node["children"])
|
|
534
|
+
return count
|
|
535
|
+
|
|
536
|
+
|
|
537
|
+
def count_nodes(file_name, file_content, lang="python_3_13"):
|
|
538
|
+
"""Count the nodes of a program before and after pruning.
|
|
539
|
+
|
|
540
|
+
Args:
|
|
541
|
+
file_name (str): Name of the file (used only for syntax-error messages).
|
|
542
|
+
file_content (str): Source code.
|
|
543
|
+
lang (str): Programming language identifier.
|
|
544
|
+
|
|
545
|
+
Returns:
|
|
546
|
+
tuple[int, int]: (nodes_before, nodes_after). `nodes_before` is every
|
|
547
|
+
node of the raw ANTLR parse tree (rules and tokens); `nodes_after` is
|
|
548
|
+
the size of the normalized, pruned and hashed tree that is handed to
|
|
549
|
+
the tree edit distance -- the same number `csim tree` prints as
|
|
550
|
+
"Total nodes after pruning".
|
|
551
|
+
"""
|
|
552
|
+
# Local import to avoid circular dependency at module import time
|
|
553
|
+
from .CodeSimilarity import ANTLR_parse, Normalize, PruneAndHash
|
|
554
|
+
|
|
555
|
+
tree = ANTLR_parse(file_name, file_content, lang)
|
|
556
|
+
|
|
557
|
+
# Iterative: real programs can nest deeper than the recursion limit.
|
|
558
|
+
before = 0
|
|
559
|
+
stack = [tree]
|
|
560
|
+
while stack:
|
|
561
|
+
node = stack.pop()
|
|
562
|
+
before += 1
|
|
563
|
+
stack.extend(node.getChild(i) for i in range(node.getChildCount()))
|
|
564
|
+
|
|
565
|
+
pruned, _ = PruneAndHash(Normalize(tree, lang), lang)
|
|
566
|
+
return before, count_tree_nodes(pruned)
|
|
567
|
+
|
|
568
|
+
|
|
509
569
|
def get_similarity_coefficient(proccesed_code1, proccesed_code2, ted_algorithm):
|
|
510
570
|
N1, len_N1 = proccesed_code1
|
|
511
571
|
N2, len_N2 = proccesed_code2
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: csim
|
|
3
|
-
Version: 3.4.
|
|
3
|
+
Version: 3.4.2
|
|
4
4
|
Summary: Code Similarity (csim) is a method designed to detect similarity between source codes
|
|
5
5
|
Home-page: https://github.com/EdsonEddy/csim
|
|
6
6
|
Author: Eddy Lecoña
|
|
@@ -363,6 +363,16 @@ similarity = Compare(name_a='example A', content_a=code_a, name_b='example B', c
|
|
|
363
363
|
print(f"Similarity: {similarity}") # Output: Similarity: X.XX
|
|
364
364
|
```
|
|
365
365
|
|
|
366
|
+
To see how much a program shrinks when it is normalized, pruned and hashed, count its nodes before and after:
|
|
367
|
+
|
|
368
|
+
```python
|
|
369
|
+
from csim import count_nodes
|
|
370
|
+
|
|
371
|
+
nodes_before, nodes_after = count_nodes("example.py", code, lang="python_3_13")
|
|
372
|
+
```
|
|
373
|
+
|
|
374
|
+
`nodes_before` is every node of the raw ANTLR parse tree; `nodes_after` is the size of the tree handed to the tree edit distance (the same number `csim tree` prints as "Total nodes after pruning").
|
|
375
|
+
|
|
366
376
|
## Documentation
|
|
367
377
|
|
|
368
378
|
- [Getting Started Guide](GETTING_STARTED.md) - Quick tutorial for new users
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|