csimx 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (137) hide show
  1. csimx-0.1.0/LICENSE +21 -0
  2. csimx-0.1.0/MANIFEST.in +26 -0
  3. csimx-0.1.0/PKG-INFO +492 -0
  4. csimx-0.1.0/README.md +452 -0
  5. csimx-0.1.0/csimx/CodeSimilarity.py +65 -0
  6. csimx-0.1.0/csimx/DataStructures.py +50 -0
  7. csimx-0.1.0/csimx/Visitors.py +516 -0
  8. csimx-0.1.0/csimx/__init__.py +10 -0
  9. csimx-0.1.0/csimx/c/CLexer.py +993 -0
  10. csimx-0.1.0/csimx/c/CLexer.tokens +270 -0
  11. csimx-0.1.0/csimx/c/CLexerBase.py +118 -0
  12. csimx-0.1.0/csimx/c/CParser.py +10663 -0
  13. csimx-0.1.0/csimx/c/CParser.tokens +270 -0
  14. csimx-0.1.0/csimx/c/CParserBase.py +668 -0
  15. csimx-0.1.0/csimx/c/CParserVisitor.py +598 -0
  16. csimx-0.1.0/csimx/c/ErrorListener.py +46 -0
  17. csimx-0.1.0/csimx/c/Symbol.py +20 -0
  18. csimx-0.1.0/csimx/c/SymbolTable.py +197 -0
  19. csimx-0.1.0/csimx/c/TypeClassification.py +14 -0
  20. csimx-0.1.0/csimx/c/__init__.py +0 -0
  21. csimx-0.1.0/csimx/c/utils.py +201 -0
  22. csimx-0.1.0/csimx/cpp_14/CPP14Lexer.py +830 -0
  23. csimx-0.1.0/csimx/cpp_14/CPP14Lexer.tokens +264 -0
  24. csimx-0.1.0/csimx/cpp_14/CPP14Parser.py +15777 -0
  25. csimx-0.1.0/csimx/cpp_14/CPP14Parser.tokens +264 -0
  26. csimx-0.1.0/csimx/cpp_14/CPP14ParserBase.py +29 -0
  27. csimx-0.1.0/csimx/cpp_14/CPP14ParserVisitor.py +968 -0
  28. csimx-0.1.0/csimx/cpp_14/__init__.py +0 -0
  29. csimx-0.1.0/csimx/cpp_14/transformGrammar.py +31 -0
  30. csimx-0.1.0/csimx/cpp_14/utils.py +208 -0
  31. csimx-0.1.0/csimx/java_20/Java20Lexer.py +813 -0
  32. csimx-0.1.0/csimx/java_20/Java20Lexer.tokens +242 -0
  33. csimx-0.1.0/csimx/java_20/Java20Parser.py +20532 -0
  34. csimx-0.1.0/csimx/java_20/Java20Parser.tokens +242 -0
  35. csimx-0.1.0/csimx/java_20/Java20ParserVisitor.py +1258 -0
  36. csimx-0.1.0/csimx/java_20/__init__.py +0 -0
  37. csimx-0.1.0/csimx/java_20/utils.py +240 -0
  38. csimx-0.1.0/csimx/java_24/Java24Lexer.py +658 -0
  39. csimx-0.1.0/csimx/java_24/Java24Lexer.tokens +244 -0
  40. csimx-0.1.0/csimx/java_24/Java24LexerBase.py +5 -0
  41. csimx-0.1.0/csimx/java_24/Java24Parser.py +12499 -0
  42. csimx-0.1.0/csimx/java_24/Java24Parser.tokens +244 -0
  43. csimx-0.1.0/csimx/java_24/Java24ParserBase.py +22 -0
  44. csimx-0.1.0/csimx/java_24/Java24ParserVisitor.py +723 -0
  45. csimx-0.1.0/csimx/java_24/__init__.py +0 -0
  46. csimx-0.1.0/csimx/java_24/utils.py +324 -0
  47. csimx-0.1.0/csimx/kotlin/KotlinLexer.py +1643 -0
  48. csimx-0.1.0/csimx/kotlin/KotlinLexer.tokens +292 -0
  49. csimx-0.1.0/csimx/kotlin/KotlinParser.py +15470 -0
  50. csimx-0.1.0/csimx/kotlin/KotlinParser.tokens +292 -0
  51. csimx-0.1.0/csimx/kotlin/KotlinParserVisitor.py +768 -0
  52. csimx-0.1.0/csimx/kotlin/__init__.py +0 -0
  53. csimx-0.1.0/csimx/kotlin/utils.py +223 -0
  54. csimx-0.1.0/csimx/language/__init__.py +0 -0
  55. csimx-0.1.0/csimx/language/lexer.py +48 -0
  56. csimx-0.1.0/csimx/language/parser.py +143 -0
  57. csimx-0.1.0/csimx/lexical/__init__.py +10 -0
  58. csimx-0.1.0/csimx/lexical/distance.py +66 -0
  59. csimx-0.1.0/csimx/lexical/tokenizer.py +47 -0
  60. csimx-0.1.0/csimx/main.py +283 -0
  61. csimx-0.1.0/csimx/native/__init__.py +15 -0
  62. csimx-0.1.0/csimx/native/loader.py +212 -0
  63. csimx-0.1.0/csimx/native/src/c_bridge.cpp +95 -0
  64. csimx-0.1.0/csimx/native/src/cpp_14_bridge.cpp +85 -0
  65. csimx-0.1.0/csimx/native/src/java_20_bridge.cpp +85 -0
  66. csimx-0.1.0/csimx/native/src/java_24_bridge.cpp +131 -0
  67. csimx-0.1.0/csimx/native/src/kotlin_bridge.cpp +90 -0
  68. csimx-0.1.0/csimx/native/src/python_3_bridge.cpp +169 -0
  69. csimx-0.1.0/csimx/native/tree_builder.py +196 -0
  70. csimx-0.1.0/csimx/processing/__init__.py +0 -0
  71. csimx-0.1.0/csimx/processing/distance_metrics.py +175 -0
  72. csimx-0.1.0/csimx/processing/tree_processing.py +250 -0
  73. csimx-0.1.0/csimx/python_3/Python3Lexer.py +644 -0
  74. csimx-0.1.0/csimx/python_3/Python3Lexer.tokens +185 -0
  75. csimx-0.1.0/csimx/python_3/Python3LexerBase.py +184 -0
  76. csimx-0.1.0/csimx/python_3/Python3Parser.py +6712 -0
  77. csimx-0.1.0/csimx/python_3/Python3Parser.tokens +185 -0
  78. csimx-0.1.0/csimx/python_3/Python3ParserBase.py +31 -0
  79. csimx-0.1.0/csimx/python_3/Python3ParserVisitor.py +403 -0
  80. csimx-0.1.0/csimx/python_3/__init__.py +0 -0
  81. csimx-0.1.0/csimx/python_3/canonical.py +166 -0
  82. csimx-0.1.0/csimx/python_3/utils.py +300 -0
  83. csimx-0.1.0/csimx/python_3_13/PythonLexer.py +1063 -0
  84. csimx-0.1.0/csimx/python_3_13/PythonLexer.tokens +187 -0
  85. csimx-0.1.0/csimx/python_3_13/PythonLexerBase.py +557 -0
  86. csimx-0.1.0/csimx/python_3_13/PythonParser.py +15618 -0
  87. csimx-0.1.0/csimx/python_3_13/PythonParser.tokens +187 -0
  88. csimx-0.1.0/csimx/python_3_13/PythonParserVisitor.py +993 -0
  89. csimx-0.1.0/csimx/python_3_13/__init__.py +0 -0
  90. csimx-0.1.0/csimx/python_3_13/utils.py +240 -0
  91. csimx-0.1.0/csimx/utils.py +770 -0
  92. csimx-0.1.0/csimx.egg-info/PKG-INFO +492 -0
  93. csimx-0.1.0/csimx.egg-info/SOURCES.txt +135 -0
  94. csimx-0.1.0/csimx.egg-info/dependency_links.txt +1 -0
  95. csimx-0.1.0/csimx.egg-info/entry_points.txt +2 -0
  96. csimx-0.1.0/csimx.egg-info/requires.txt +5 -0
  97. csimx-0.1.0/csimx.egg-info/top_level.txt +1 -0
  98. csimx-0.1.0/grammars/CLexer.g4 +752 -0
  99. csimx-0.1.0/grammars/CLexerBase.cpp +157 -0
  100. csimx-0.1.0/grammars/CLexerBase.h +64 -0
  101. csimx-0.1.0/grammars/CPP14Lexer.g4 +398 -0
  102. csimx-0.1.0/grammars/CPP14Parser.g4 +1075 -0
  103. csimx-0.1.0/grammars/CPP14ParserBase.cpp +20 -0
  104. csimx-0.1.0/grammars/CPP14ParserBase.h +9 -0
  105. csimx-0.1.0/grammars/CParser.g4 +942 -0
  106. csimx-0.1.0/grammars/CParserBase.cpp +570 -0
  107. csimx-0.1.0/grammars/CParserBase.h +77 -0
  108. csimx-0.1.0/grammars/Java20Lexer.g4 +941 -0
  109. csimx-0.1.0/grammars/Java20Parser.g4 +1655 -0
  110. csimx-0.1.0/grammars/Java24Lexer.g4 +224 -0
  111. csimx-0.1.0/grammars/Java24Parser.g4 +828 -0
  112. csimx-0.1.0/grammars/Java24ParserBase.cpp +49 -0
  113. csimx-0.1.0/grammars/Java24ParserBase.h +10 -0
  114. csimx-0.1.0/grammars/KotlinLexer.g4 +517 -0
  115. csimx-0.1.0/grammars/KotlinParser.g4 +840 -0
  116. csimx-0.1.0/grammars/Python3Lexer.g4 +743 -0
  117. csimx-0.1.0/grammars/Python3LexerBase.cpp +182 -0
  118. csimx-0.1.0/grammars/Python3LexerBase.h +30 -0
  119. csimx-0.1.0/grammars/Python3Parser.g4 +412 -0
  120. csimx-0.1.0/grammars/Python3ParserBase.cpp +18 -0
  121. csimx-0.1.0/grammars/Python3ParserBase.h +17 -0
  122. csimx-0.1.0/grammars/PythonLexer.g4 +1471 -0
  123. csimx-0.1.0/grammars/PythonParser.g4 +889 -0
  124. csimx-0.1.0/grammars/Symbol.h +76 -0
  125. csimx-0.1.0/grammars/SymbolTable.cpp +209 -0
  126. csimx-0.1.0/grammars/SymbolTable.h +38 -0
  127. csimx-0.1.0/grammars/TypeClassification.h +15 -0
  128. csimx-0.1.0/grammars/UnicodeClasses.g4 +1658 -0
  129. csimx-0.1.0/grammars/parser_gen_guide.md +81 -0
  130. csimx-0.1.0/scripts/build_native_parsers.sh +266 -0
  131. csimx-0.1.0/scripts/transform_grammar_for_cpp.py +82 -0
  132. csimx-0.1.0/setup.cfg +4 -0
  133. csimx-0.1.0/setup.py +105 -0
  134. csimx-0.1.0/test/test_cli.py +154 -0
  135. csimx-0.1.0/test/test_module.py +168 -0
  136. csimx-0.1.0/test/test_native_parsers.py +480 -0
  137. csimx-0.1.0/test/test_python_3_canonical.py +72 -0
csimx-0.1.0/LICENSE ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2025 Edson Eddy
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,26 @@
1
+ # Sources needed to rebuild the native parsers from an sdist.
2
+ include grammars/*.g4
3
+ include grammars/*Base.cpp
4
+ include grammars/*Base.h
5
+ include grammars/parser_gen_guide.md
6
+ include scripts/build_native_parsers.sh
7
+ include scripts/transform_grammar_for_cpp.py
8
+ include csimx/native/src/*.cpp
9
+
10
+ # The C grammar's symbol-table implementation (typedef disambiguation) --
11
+ # doesn't match the *Base.cpp/*Base.h patterns above since it isn't a
12
+ # lexer/parser base class.
13
+ include grammars/SymbolTable.cpp
14
+ include grammars/SymbolTable.h
15
+ include grammars/Symbol.h
16
+ include grammars/TypeClassification.h
17
+
18
+ include README.md
19
+ include LICENSE
20
+
21
+ # Compiled parsers are platform-specific and ride in the platform wheels.
22
+ # Shipping them in the sdist would install a macOS .dylib onto Linux, where it
23
+ # cannot load -- harmless (csimx falls back to Python) but misleading.
24
+ recursive-exclude csimx/native/lib *
25
+ recursive-exclude * __pycache__
26
+ recursive-exclude * *.py[co]
csimx-0.1.0/PKG-INFO ADDED
@@ -0,0 +1,492 @@
1
+ Metadata-Version: 2.4
2
+ Name: csimx
3
+ Version: 0.1.0
4
+ Summary: csimx: code similarity in two stages, a fast lexical filter and a structural comparison (parse trees and tree edit distance)
5
+ Home-page: https://github.com/EdsonEddy/csimx
6
+ Author: Eddy Lecoña
7
+ Author-email: crew0eddy@gmail.com
8
+ License: MIT
9
+ Project-URL: Bug Tracker, https://github.com/EdsonEddy/csimx/issues
10
+ Project-URL: Documentation, https://github.com/EdsonEddy/csimx/wiki
11
+ Project-URL: Source Code, https://github.com/EdsonEddy/csimx
12
+ Keywords: code analysis,similarity detection,tree parser,tree edit distance,code snippets,code comparison
13
+ Platform: All
14
+ Classifier: Programming Language :: Python :: 3
15
+ Classifier: License :: OSI Approved :: MIT License
16
+ Classifier: Operating System :: OS Independent
17
+ Classifier: Development Status :: 4 - Beta
18
+ Requires-Python: >=3.10
19
+ Description-Content-Type: text/markdown
20
+ License-File: LICENSE
21
+ Requires-Dist: antlr4-python3-runtime==4.13.2
22
+ Requires-Dist: zss==1.2.0
23
+ Requires-Dist: numpy==1.26.4
24
+ Requires-Dist: apted==1.0.3
25
+ Requires-Dist: pygments<3,>=2.20
26
+ Dynamic: author
27
+ Dynamic: author-email
28
+ Dynamic: classifier
29
+ Dynamic: description
30
+ Dynamic: description-content-type
31
+ Dynamic: home-page
32
+ Dynamic: keywords
33
+ Dynamic: license
34
+ Dynamic: license-file
35
+ Dynamic: platform
36
+ Dynamic: project-url
37
+ Dynamic: requires-dist
38
+ Dynamic: requires-python
39
+ Dynamic: summary
40
+
41
+ # Code Similarity (csimx)
42
+
43
+ Code Similarity (csimx) provide a module designed to detect similarities between source code files, even when obfuscation techniques have been applied. It is particularly useful for programming instructors and students who need to verify code originality.
44
+
45
+ > **Origin.** csimx started from [csim](https://github.com/EdsonEddy/csim) 4.1.0 and keeps its structural comparison unchanged (same trees, same index). What it adds is a lexical stage that `group` can use to skip pairs of files that are too different. The version history of csim before this fork is in `CHANGELOG.md`; the documents in `docs/` were written for csim and still call the project csim.
46
+
47
+ ## Key Features
48
+
49
+ - **Source Code Similarity Analysis:** Compares source code files to determine their degree of similarity.
50
+ - **Pairwise Reporting:** Generate detailed similarity reports for all file pairs.
51
+ - **File Grouping:** Cluster similar files into groups based on a configurable threshold.
52
+ - **Flexible Search Strategies:**
53
+ - **Exhaustive Search:** All-pairs comparison for maximum precision
54
+ - **Two Stages:** A fast lexical stage (Pygments tokens + Myers diff) can filter the pairs that `group` sends to the structural stage (parse trees + tree edit distance).
55
+ - **Advanced Analysis:** Utilizes parse trees and the tree edit distance algorithm for in-depth analysis.
56
+ - **Parse Trees:** Represents the syntactic structure of source code, enabling detailed comparisons.
57
+ - **Tree Edit Distance:** Measures the similarity between different code structures.
58
+ - **Hash-Based Pruning:** Optimizes the comparison process by reducing tree size while preserving essential structure.
59
+ - **Multi-Language Support:** Supports Python 3.13, Python 3 (universal grammar), Java 20, Java 24, C++14, Kotlin (experimental), and C (experimental) source code analysis.
60
+
61
+ ## Technologies Used
62
+
63
+ - **Python:** The core programming language for the tool.
64
+ - **ANTLR:** A parser generator for creating parse trees from source code.
65
+ - **apted:** A library for computing the tree edit distance (default algorithm).
66
+ - **zss:** A library for calculating the tree edit distance, alternatively to apted.
67
+ - **NumPy:** Used for efficient numerical operations.
68
+ - **Pygments:** Tokenizer of the lexical stage (one for every supported language).
69
+
70
+ ## Installation
71
+ For the installation `pip` is required, you can either clone the repository and install it locally or install it directly from PyPI.
72
+
73
+ 1. Clone the repository:
74
+ ```sh
75
+ git clone https://github.com/EdsonEddy/csimx.git
76
+ ```
77
+ 2. Navigate to the project directory:
78
+ ```sh
79
+ cd csimx
80
+ ```
81
+ 3. Install the package:
82
+ ```sh
83
+ pip install .
84
+ ```
85
+
86
+ Alternatively, you can install it directly from PyPI:
87
+
88
+ ```sh
89
+ pip install csimx
90
+ ```
91
+
92
+
93
+ ### Version Compatibility
94
+ - **Python:** 3.10–3.12 (recommended 3.11)
95
+ - **ANTLR4 Python Runtime:** 4.13.2
96
+ - **zss:** 1.2.0
97
+ - **apted:** 1.0.3
98
+ - **numpy:** 1.26.4
99
+
100
+ ## Quick Start
101
+
102
+ **New to csimx?** Start here: [GETTING_STARTED.md](GETTING_STARTED.md)
103
+
104
+ For detailed information about search strategies, see: [docs/STRATEGIES.md](docs/STRATEGIES.md)
105
+
106
+ csimx supports three main actions: **report** (for pairwise similarity analysis), **group** (for clustering similar files), and **tree**/**view** (for visualizing a file's normalized/pruned parse tree). The tool supports Python 3.13, Python 3, Java 20, Java 24, C++14, Kotlin (experimental), and C (experimental) source code files.
107
+
108
+ ### General Command Structure
109
+ ```sh
110
+ csimx <action> --path <directory> [options]
111
+ ```
112
+
113
+ ### Action 1: `report` - Generate Similarity Report
114
+
115
+ Generates a pairwise similarity report comparing all files in a directory.
116
+
117
+ ```sh
118
+ csimx report --path /path/to/directory
119
+ ```
120
+
121
+ **Example Output:**
122
+ ```
123
+ file1.py is similar to file2.py with similarity index: 0.95
124
+ file1.py is similar to file3.py with similarity index: 0.45
125
+ file2.py is similar to file3.py with similarity index: 0.50
126
+ ```
127
+
128
+ **Options:**
129
+ - `--lang, -l`: Programming language (default: `python_3_13`). Options: `python_3_13`, `python_3`, `java_20`, `java_24`, `cpp_14`, `kotlin`, `c`
130
+ - `--talg, -ta`: Tree edit distance algorithm (default: `apted`). Options: `zss`, `apted`
131
+ - `--index, -ix`: Similarity index formula (default: `legacy`). Options: `legacy`, `ratio`, `metric`
132
+
133
+ **Example with options:**
134
+ ```sh
135
+ csimx report --path /path/to/directory --lang java_20 --talg zss --index ratio
136
+ ```
137
+
138
+ ### Action 2: `group` - Group Files by Similarity
139
+
140
+ Groups files by similarity using a specified threshold and strategy.
141
+
142
+ ```sh
143
+ csimx group --path /path/to/directory --threshold 0.8
144
+ ```
145
+
146
+ **Example Output:**
147
+ ```
148
+ Threshold: 0.8
149
+ Total files processed: 4
150
+ Group 1 (Average Similarity: 0.98):
151
+ ./file1.py
152
+ ./file2.py
153
+ Group 2 (Average Similarity: 0.95):
154
+ ./file3.py
155
+ ./file4.py
156
+ ```
157
+
158
+ #### Strategy Options
159
+
160
+ The `group` action supports two strategies for finding similar files:
161
+
162
+ ##### 1. **exhaustive** (Default)
163
+ Compares every file against every other file (O(n²)). This is the most thorough approach but slower for large datasets.
164
+
165
+ ```sh
166
+ csimx group --path /path/to/directory --threshold 0.8 --strategy exhaustive
167
+ ```
168
+
169
+ **When to use each:**
170
+ - **exhaustive**: Small datasets (< 100 files), when maximum precision is critical
171
+
172
+ #### Group Action Options
173
+
174
+ - `--threshold, -t`: Similarity threshold (0.0 to 1.0). **Required.**
175
+ - `--strategy, -s`: Grouping strategy (default: `exhaustive`). Options: `exhaustive`
176
+ - `--lang, -l`: Programming language (default: `python_3_13`). Options: `python_3_13`, `python_3`, `java_20`, `java_24`, `cpp_14`, `kotlin`, `c`
177
+ - `--talg, -ta`: Tree edit distance algorithm (default: `apted`). Options: `zss`, `apted`
178
+ - `--index, -ix`: Similarity index formula (default: `legacy`). Options: `legacy`, `ratio`, `metric`
179
+ - `--prefilter`: Compare the tokens of each pair first and run the structural comparison only on the pairs whose lexical index reaches `--threshold` minus 0.05 (see below). Off by default.
180
+ - `--prefilter-margin X`: Same, with margin `X` (0.0 to 0.30); it turns the prefilter on by itself.
181
+
182
+ **Complete example:**
183
+ ```sh
184
+ csimx group --path /path/to/directory --threshold 0.9 --strategy exhaustive --lang python_3_13 --talg zss
185
+ ```
186
+
187
+ #### Lexical prefilter (`group` only)
188
+
189
+ csimx has two stages: the **structural** one (parse tree, tree edit distance) gives the index, and a **lexical** one compares the token sequences of two files with the Myers diff algorithm (`1 - d / (len_a + len_b)`, comments and layout dropped, names, numbers and strings generalized to their type). With `--prefilter` the lexical stage goes first and the structural one only runs on the pairs that are not too different in tokens, so the time of a big directory drops by a few times; files that take part in no such pair are not even parsed.
190
+
191
+ ```sh
192
+ csimx group --path /path/to/directory --threshold 0.7 --lang python_3 --prefilter
193
+ csimx group --path /path/to/directory --threshold 0.7 --lang python_3 --prefilter-margin 0.1
194
+ ```
195
+
196
+ The filter is lossy: a pair with the same structure and very different tokens (a moved block, reordered statements) can be skipped even though the structural stage would group it. The margin is how far under `--threshold` the lexical index may be; a larger margin loses fewer pairs and skips fewer. It pays off from a threshold of about 0.6 up. See the 0.1.0 entry of `CHANGELOG.md` for the numbers. `report` has no prefilter.
197
+
198
+ #### Similarity Index Formulas
199
+
200
+ `--index` chooses how the tree edit distance `d` is normalized into the
201
+ similarity index. With `m = max(n1, n2)` and `s = n1 + n2`:
202
+
203
+ | Value | Formula | Notes |
204
+ |---|---|---|
205
+ | `legacy` (default) | `1 - d / m` | The index of csim <= 3.4.2. |
206
+ | `ratio` | `m / (m + d)` | Always in (0, 1]. Ranks pairs exactly as `legacy` does, on a different scale. |
207
+ | `metric` | `(s - d) / (s + d)` | Metric normalization of tree edit distance (Li & Zhang); satisfies the triangle inequality. Ranks by total size instead of by the larger tree. |
208
+
209
+ **Thresholds do not carry over between formulas.** `ratio` is a monotone
210
+ rescaling of `legacy`, so it groups files in exactly the same order, but on a
211
+ different scale. To reproduce a `legacy` threshold under `ratio`, use
212
+ `t_ratio = 1 / (2 - t_legacy)`:
213
+
214
+ | `legacy` | `ratio` |
215
+ |---|---|
216
+ | 0.70 | 0.769 |
217
+ | 0.80 | 0.833 |
218
+ | 0.90 | 0.909 |
219
+
220
+ Because `ratio` never reaches 0 in practice (a pair with nothing in common
221
+ sits near 0.5, where `legacy` puts it near 0), a threshold taken straight from
222
+ the `legacy` scale will be far more lenient than intended. `legacy` is the
223
+ default so that existing thresholds keep their meaning.
224
+
225
+ ### Action 3: `tree` (alias: `view`) - Visualize Parse Trees
226
+
227
+ Prints the normalized/pruned tree for a single file — the exact tree that gets passed to the tree edit distance algorithm. Useful for debugging how the normalization, collapsing, and hashing rules affect a specific file before it's compared against others.
228
+
229
+ ```sh
230
+ csimx tree --path /path/to/file.py --lang python_3_13
231
+ ```
232
+
233
+ **Example Output:**
234
+ ```
235
+ === Normalized + Pruned Tree (input to Tree Edit Distance) ===
236
+ statements
237
+ function_def_raw
238
+ param [hashed:e3b0c442]
239
+ statements
240
+ STRING
241
+ if_stmt
242
+ comparison [hashed:93e10dca]
243
+ return_stmt [hashed:337adaa9]
244
+ assignment [hashed:118045cc]
245
+ primary [hashed:e1b0c7ab]
246
+
247
+ Total nodes after pruning: 24
248
+ ```
249
+
250
+ Rule and token names are resolved for readability, `LOOP` marks nodes collapsed under control-flow equivalence (e.g. `for`/`while`), and `[hashed:xxxxxxxx]` marks subtrees that were hashed into a single node instead of compared structurally.
251
+
252
+ **Options:**
253
+ - `--path, -p`: Path to a single source code file (**required**).
254
+ - `--lang, -l`: Programming language (default: `python_3_13`). Options: `python_3_13`, `python_3`, `java_20`, `java_24`, `cpp_14`, `kotlin`, `c`
255
+ - `--show-raw`: Also print the raw ANTLR parse tree before normalization/pruning, for side-by-side comparison.
256
+
257
+ **Example with `--show-raw`:**
258
+ ```sh
259
+ csimx tree --path /path/to/file.py --lang python_3_13 --show-raw
260
+ ```
261
+
262
+ ### Language Support
263
+
264
+ The tool supports the following programming languages:
265
+
266
+ **Python 3.13:**
267
+ ```sh
268
+ csimx report --path /path/to/python/files --lang python_3_13
269
+ ```
270
+
271
+ **Python 3 (universal grammar):**
272
+ ```sh
273
+ csimx report --path /path/to/python/files --lang python_3
274
+ ```
275
+
276
+ Same `.py` files as `python_3_13`, parsed with grammars-v4's "universal Python 2/3"
277
+ grammar, which publishes a C++ target -- giving a large speedup once the native
278
+ parser is built (see [Native Parsers](#native-parsers) below). Grouping output is
279
+ byte-identical to `python_3_13` on most real-world code (verified against
280
+ hundreds of real judge submissions); a narrow, understood exception remains for
281
+ files with very few top-level statements, where tree-size differences can
282
+ slightly overstate similarity — see `csimx/python_3/utils.py` for the full
283
+ writeup. Does not parse positional-only parameters (`/`, PEP 570), the walrus
284
+ operator (`:=`, PEP 572), or `match`/`case` (PEP 634); csimx falls back to
285
+ `python_3_13` automatically for files using those, so results stay correct,
286
+ just slower for that subset.
287
+
288
+ **Java 20:**
289
+ ```sh
290
+ csimx report --path /path/to/java/files --lang java_20
291
+ ```
292
+
293
+ **Java 24 (experimental):**
294
+ ```sh
295
+ csimx report --path /path/to/java/files --lang java_24
296
+ ```
297
+
298
+ Java 24 uses an optimized grammar (grammars-v4/java/java) that parses much faster
299
+ than `java_20` when the native parser is available (see [Native Parsers](#native-parsers)
300
+ below), and supports the same modern Java syntax (records, sealed classes, pattern
301
+ matching, switch expressions, text blocks). **However, its similarity/grouping
302
+ output is not yet tuned to match `java_20` and can be significantly less accurate**
303
+ (verified on real submissions — see `CHANGELOG.md`). Use `java_20` for `group`/`report`
304
+ until this is resolved; `java_24` is available for parse-speed experimentation only.
305
+
306
+ **C++14:**
307
+ ```sh
308
+ csimx report --path /path/to/cpp/files --lang cpp_14
309
+ ```
310
+
311
+ **Kotlin (experimental):**
312
+ ```sh
313
+ csimx report --path /path/to/kotlin/files --lang kotlin
314
+ ```
315
+
316
+ First-cut integration (grammars-v4/kotlin/kotlin) with a working native parser
317
+ (no C++ base class needed at all -- this grammar declares no `superClass`).
318
+ Unlike `java_24`/`python_3`, there is no real-world Kotlin corpus in this
319
+ project's benchmark set to tune or validate grouping precision against yet, so
320
+ the normalization rules (`csimx/kotlin/utils.py`) follow the same *categories*
321
+ already validated for other languages (structural punctuation, identifier
322
+ text, import/package plumbing, body-wrapping content) but haven't been
323
+ corpus-measured for false-positive/false-negative rates. Treat `group`/`report`
324
+ results as a reasonable starting point, not a tuned config, until a real
325
+ corpus drives the next pass.
326
+
327
+ **C (experimental):**
328
+ ```sh
329
+ csimx report --path /path/to/c/files --lang c
330
+ ```
331
+
332
+ First-cut integration (grammars-v4/c, ISO C23 + GNU/MSVC extensions), with a
333
+ working native parser backed by a real symbol-table implementation for
334
+ typedef disambiguation. Like Kotlin, there's no real-world C corpus in this
335
+ project's benchmark set to tune grouping precision against yet -- same
336
+ caveats apply, see `csimx/c/utils.py`.
337
+
338
+ Runs with preprocessing disabled (`--nopp`) always: a real preprocessor can't
339
+ be assumed present in a production container, and judge submissions have no
340
+ consistent include paths anyway. `#include`/`#define`/etc. lines are
341
+ swallowed as hidden tokens rather than expanded, which means macro-dependent
342
+ code (token-pasting tricks, macros used for control flow) can fail to parse
343
+ or parse differently than a real compiler would see it -- real submissions
344
+ essentially never rely on that, but it's a known, real limitation.
345
+
346
+ ### Native Parsers
347
+
348
+ For `java_20`, `java_24`, `cpp_14`, `python_3`, `kotlin`, and `c`, csimx can
349
+ use a compiled C++ ANTLR parser instead of the pure-Python one, giving a
350
+ large speedup with identical (or, for `java_24`, not-yet-identical -- see
351
+ above) output. `python_3_13` always uses the pure-Python parser (no C++
352
+ target is available for that grammar).
353
+
354
+ Check which backend is active for each language:
355
+
356
+ ```sh
357
+ csimx info
358
+ ```
359
+
360
+ If a native library isn't present for a language, csimx falls back to the
361
+ pure-Python parser automatically — results are unaffected, only speed. Build
362
+ the native parsers from source with:
363
+
364
+ ```sh
365
+ scripts/build_native_parsers.sh
366
+ ```
367
+
368
+ Set `CSIM_DISABLE_NATIVE=1` to force the pure-Python parsers for every
369
+ language, e.g. for debugging or benchmarking.
370
+
371
+ ### Threshold Guidance
372
+
373
+ The similarity threshold represents the structural similarity of the code (based on the Abstract Syntax Tree). Choose appropriate thresholds based on your use case:
374
+
375
+ - **0.95+**: Nearly identical code (likely plagiarism)
376
+ - **0.85-0.95**: Very similar code (probable plagiarism)
377
+ - **0.70-0.85**: Moderately similar code (review recommended)
378
+ - **<0.70**: Low similarity (likely independent work)
379
+
380
+ ### Using csimx as a Python Module
381
+
382
+ You can also use csimx programmatically within your Python code. The library provides low-level functions for advanced use cases:
383
+
384
+ ```python
385
+ from csimx.utils import group_by_exhaustive_search, report_pairwise_similarity
386
+
387
+ # Example: Group files by similarity
388
+ file_names = ["file1.py", "file2.py", "file3.py"]
389
+ file_contents = [code1, code2, code3]
390
+
391
+ results = group_by_exhaustive_search(
392
+ file_names=file_names,
393
+ file_contents=file_contents,
394
+ lang="python_3_13",
395
+ threshold=0.8,
396
+ ted_algorithm="apted"
397
+ )
398
+
399
+ print(results)
400
+ ```
401
+
402
+ Or use the legacy Compare class for simple pairwise comparisons:
403
+
404
+ ```python
405
+ from csimx import Compare
406
+
407
+ code_a = "a = 5"
408
+ code_b = "c = 50"
409
+ similarity = Compare(name_a='example A', content_a=code_a, name_b='example B', content_b=code_b)
410
+ print(f"Similarity: {similarity}") # Output: Similarity: X.XX
411
+ ```
412
+
413
+ To see how much a program shrinks when it is normalized, pruned and hashed, count its nodes before and after:
414
+
415
+ ```python
416
+ from csimx import count_nodes
417
+
418
+ nodes_before, nodes_after = count_nodes("example.py", code, lang="python_3_13")
419
+ ```
420
+
421
+ `nodes_before` is every node of the raw ANTLR parse tree; `nodes_after` is the size of the tree handed to the tree edit distance (the same number `csimx tree` prints as "Total nodes after pruning").
422
+
423
+ ## Documentation
424
+
425
+ - [Getting Started Guide](GETTING_STARTED.md) - Quick tutorial for new users
426
+ - [Search Strategies Guide](docs/STRATEGIES.md) - Detailed explanation of available search strategies
427
+ - [ANTLR Parser Generation](grammars/parser_gen_guide.md) - For grammar customization
428
+
429
+ ## ANTLR4 Installation and Parser/Lexer Generation
430
+
431
+ This installation is not required—the generated files are already included in the project. If you'd like to review the steps to generate them yourself, see [grammars/parser_gen_guide.md](grammars/parser_gen_guide.md).
432
+
433
+ Note: The included generated files were produced by **ANTLR 4.13.2** and are compatible with the pinned runtime listed above.
434
+
435
+ ## Contributing
436
+
437
+ Contributions are welcome! To contribute, please follow these steps:
438
+
439
+ 1. Fork the repository.
440
+ 2. Create a new branch (`git checkout -b feature/new-feature`).
441
+ 3. Make your changes and commit them (`git commit -am 'Add new feature'`).
442
+ 4. Push to the branch (`git push origin feature/new-feature`).
443
+ 5. Open a Pull Request.
444
+
445
+ ## License
446
+
447
+ This project is licensed under the MIT License. See the [LICENSE](LICENSE) file for details.
448
+
449
+ ## Support
450
+
451
+ - **Questions?** Open a [GitHub Discussion](https://github.com/EdsonEddy/csimx/discussions)
452
+ - **Found a bug?** File a [GitHub Issue](https://github.com/EdsonEddy/csimx/issues)
453
+ - **Want to contribute?** See [Contributing](#contributing) section
454
+
455
+ ## References
456
+
457
+ For more information on the techniques and tools used in this project, refer to the following resources:
458
+
459
+ - [ANTLR](https://www.antlr.org/)
460
+ - [Parse Tree (Wikipedia)](https://en.wikipedia.org/wiki/Parse_tree)
461
+ - [Tree Edit Distance (Wikipedia)](https://en.wikipedia.org/wiki/Tree_edit_distance)
462
+ - [Locality Sensitive Hashing (Wikipedia)](https://en.wikipedia.org/wiki/Locality-sensitive_hashing)
463
+ - [MinHash (Wikipedia)](https://en.wikipedia.org/wiki/MinHash)
464
+ - [zss (PyPI)](https://pypi.org/project/zss/)
465
+ - [Hashing (Python Docs)](https://docs.python.org/3/library/hashlib.html)
466
+ - [apted (GitHub)](https://github.com/JoaoFelipe/apted)
467
+
468
+ ## Third-Party Licenses
469
+
470
+ This project utilizes the following third-party libraries:
471
+
472
+ ### ANTLR (ANother Tool for Language Recognition)
473
+ - **Purpose:** A parser generator used to create parse trees from source code.
474
+ - **License:** BSD 3-Clause
475
+ - **Website:** [https://www.antlr.org/](https://www.antlr.org/)
476
+ - **Repository:** [https://github.com/antlr/antlr4](https://github.com/antlr/antlr4)
477
+
478
+ ### ANTLR4-parser-for-Python-3.14 by RobEin
479
+ - **Purpose:** Python 3.14 grammar for ANTLR4
480
+ - **License:** MIT License
481
+ - **Repository:** [https://github.com/RobEin/ANTLR4-parser-for-Python-3.14](https://github.com/RobEin/ANTLR4-parser-for-Python-3.14)
482
+
483
+ ### zss (Zhang-Shasha)
484
+ - **Purpose:** Tree edit distance algorithm implementation for comparing tree structures
485
+ - **License:** MIT License
486
+ - **Repository:** [https://github.com/timtadh/zhang-shasha](https://github.com/timtadh/zhang-shasha)
487
+
488
+ ### apted (All Path Tree Edit Distance)
489
+ - **Purpose:** Python APTED algorithm for the Tree Edit Distance, an alternative to zss
490
+ - **License:** MIT License
491
+ - **Repository:** [https://github.com/JoaoFelipe/apted](https://github.com/JoaoFelipe/apted)
492
+