duplicatecode 0.2.0__tar.gz → 0.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {duplicatecode-0.2.0 → duplicatecode-0.3.0}/Cargo.lock +15 -4
- {duplicatecode-0.2.0 → duplicatecode-0.3.0}/Cargo.toml +1 -1
- {duplicatecode-0.2.0 → duplicatecode-0.3.0}/PKG-INFO +131 -3
- {duplicatecode-0.2.0 → duplicatecode-0.3.0}/README.md +129 -1
- {duplicatecode-0.2.0 → duplicatecode-0.3.0}/crates/duplicatecode-cli/src/bench.rs +4 -1
- duplicatecode-0.3.0/crates/duplicatecode-cli/src/groups.rs +267 -0
- {duplicatecode-0.2.0 → duplicatecode-0.3.0}/crates/duplicatecode-cli/src/main.rs +376 -26
- {duplicatecode-0.2.0 → duplicatecode-0.3.0}/crates/duplicatecode-engine/Cargo.toml +2 -1
- {duplicatecode-0.2.0 → duplicatecode-0.3.0}/crates/duplicatecode-engine/src/embed.rs +259 -35
- duplicatecode-0.3.0/crates/duplicatecode-engine/src/explain.rs +130 -0
- duplicatecode-0.3.0/crates/duplicatecode-engine/src/fragments.rs +201 -0
- {duplicatecode-0.2.0 → duplicatecode-0.3.0}/crates/duplicatecode-engine/src/index.rs +163 -23
- {duplicatecode-0.2.0 → duplicatecode-0.3.0}/crates/duplicatecode-engine/src/lang.rs +6 -2
- {duplicatecode-0.2.0 → duplicatecode-0.3.0}/crates/duplicatecode-engine/src/lib.rs +2 -0
- {duplicatecode-0.2.0 → duplicatecode-0.3.0}/crates/duplicatecode-engine/src/similarity.rs +32 -0
- {duplicatecode-0.2.0 → duplicatecode-0.3.0}/crates/duplicatecode-engine/src/units.rs +566 -6
- {duplicatecode-0.2.0 → duplicatecode-0.3.0}/pyproject.toml +1 -1
- {duplicatecode-0.2.0 → duplicatecode-0.3.0}/crates/duplicatecode-cli/Cargo.toml +0 -0
- {duplicatecode-0.2.0 → duplicatecode-0.3.0}/crates/duplicatecode-cli/src/review.rs +0 -0
- {duplicatecode-0.2.0 → duplicatecode-0.3.0}/crates/duplicatecode-engine/src/diff.rs +0 -0
- {duplicatecode-0.2.0 → duplicatecode-0.3.0}/crates/duplicatecode-engine/src/fingerprint.rs +0 -0
- {duplicatecode-0.2.0 → duplicatecode-0.3.0}/crates/duplicatecode-engine/src/mutate.rs +0 -0
- {duplicatecode-0.2.0 → duplicatecode-0.3.0}/crates/duplicatecode-engine/src/naming.rs +0 -0
|
@@ -91,9 +91,9 @@ dependencies = [
|
|
|
91
91
|
|
|
92
92
|
[[package]]
|
|
93
93
|
name = "cc"
|
|
94
|
-
version = "1.
|
|
94
|
+
version = "1.2.67"
|
|
95
95
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
|
96
|
-
checksum = "
|
|
96
|
+
checksum = "e17dd265a7d0f31ef544e1b20e03add05d3b45b491b633b10d67145d2acc1a38"
|
|
97
97
|
dependencies = [
|
|
98
98
|
"find-msvc-tools",
|
|
99
99
|
"shlex",
|
|
@@ -198,7 +198,7 @@ dependencies = [
|
|
|
198
198
|
|
|
199
199
|
[[package]]
|
|
200
200
|
name = "duplicatecode"
|
|
201
|
-
version = "0.
|
|
201
|
+
version = "0.3.0"
|
|
202
202
|
dependencies = [
|
|
203
203
|
"anyhow",
|
|
204
204
|
"clap",
|
|
@@ -212,7 +212,7 @@ dependencies = [
|
|
|
212
212
|
|
|
213
213
|
[[package]]
|
|
214
214
|
name = "duplicatecode-engine"
|
|
215
|
-
version = "0.
|
|
215
|
+
version = "0.3.0"
|
|
216
216
|
dependencies = [
|
|
217
217
|
"ignore",
|
|
218
218
|
"serde",
|
|
@@ -220,6 +220,7 @@ dependencies = [
|
|
|
220
220
|
"tree-sitter",
|
|
221
221
|
"tree-sitter-c-sharp",
|
|
222
222
|
"tree-sitter-python",
|
|
223
|
+
"tree-sitter-sequel",
|
|
223
224
|
"tree-sitter-typescript",
|
|
224
225
|
"ureq",
|
|
225
226
|
"walkdir",
|
|
@@ -790,6 +791,16 @@ dependencies = [
|
|
|
790
791
|
"tree-sitter-language",
|
|
791
792
|
]
|
|
792
793
|
|
|
794
|
+
[[package]]
|
|
795
|
+
name = "tree-sitter-sequel"
|
|
796
|
+
version = "0.3.11"
|
|
797
|
+
source = "registry+https://github.com/rust-lang/crates.io-index"
|
|
798
|
+
checksum = "9d198ad3c319c02e43c21efa1ec796b837afcb96ffaef1a40c1978fbdcec7d17"
|
|
799
|
+
dependencies = [
|
|
800
|
+
"cc",
|
|
801
|
+
"tree-sitter-language",
|
|
802
|
+
]
|
|
803
|
+
|
|
793
804
|
[[package]]
|
|
794
805
|
name = "tree-sitter-typescript"
|
|
795
806
|
version = "0.23.2"
|
|
@@ -1,16 +1,16 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: duplicatecode
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.3.0
|
|
4
4
|
Classifier: Programming Language :: Rust
|
|
5
5
|
Classifier: Environment :: Console
|
|
6
|
-
Summary: Static (LLM-free) detection of duplicate/similar code in Python, TypeScript and C#
|
|
6
|
+
Summary: Static (LLM-free) detection of duplicate/similar code in Python, TypeScript/JavaScript, SQL and C#
|
|
7
7
|
License: MIT
|
|
8
8
|
Requires-Python: >=3.9
|
|
9
9
|
Description-Content-Type: text/markdown; charset=UTF-8; variant=GFM
|
|
10
10
|
|
|
11
11
|
# duplicatecode
|
|
12
12
|
|
|
13
|
-
Static (LLM-free) detection of duplicate / similar code in Python, TypeScript (TSX) and C#,
|
|
13
|
+
Static (LLM-free) detection of duplicate / similar code in Python, TypeScript/JavaScript (TSX/JSX), SQL and C#,
|
|
14
14
|
aimed at catching an LLM re-implementing something that already exists. Input can be a git diff
|
|
15
15
|
checked against existing source.
|
|
16
16
|
|
|
@@ -42,6 +42,134 @@ duplicatecode scan . --pairs --json # machine-readable pairs in
|
|
|
42
42
|
duplicatecode bench --dataset dataset [--file-level] [--mutations] [--negatives <other repo>]
|
|
43
43
|
```
|
|
44
44
|
|
|
45
|
+
## More ways to look
|
|
46
|
+
|
|
47
|
+
```sh
|
|
48
|
+
duplicatecode fragments packages/ # copied blocks inside different functions (4+ identical statements)
|
|
49
|
+
duplicatecode scan . --pairs --explain # say what differs: literals, calls, lines (what to parameterize)
|
|
50
|
+
duplicatecode find "retry with exponential backoff" src/ # does something like this already exist? (needs embeddings)
|
|
51
|
+
```
|
|
52
|
+
|
|
53
|
+
- `fragments` indexes windows of consecutive normalized statements, so a block pasted into another
|
|
54
|
+
function is found even when the surrounding functions differ. On an injection benchmark (renamed
|
|
55
|
+
blocks of real functions pasted into others) it finds 93-97% of blocks of 6+ statements; see
|
|
56
|
+
`eval/REPORT.md`. Tune with `--min-stmts` / `--min-tokens`; constructors and dunders are ignored.
|
|
57
|
+
- `--explain` adds, per pair, the literals and calls only one side has and the source lines with no
|
|
58
|
+
counterpart ("16 of 18/17 statements shared").
|
|
59
|
+
- `find` embeds the query and every unit and ranks by similarity. With MiniLM, 50 task descriptions
|
|
60
|
+
against 168 implementation units rank the right one first 94% of the time; with 1,379 unrelated real
|
|
61
|
+
units mixed in, 82% first and 96% in the top 3 (every query within the top 10).
|
|
62
|
+
- Default thresholds depend on the profile: `copies` keeps its tuned 0.6 (`scan`), 0.4 (`diff`), 0.45
|
|
63
|
+
(`review`); `reimpl` scores live lower, so it defaults to 0.35 / 0.28 / 0.30, which correspond to roughly
|
|
64
|
+
0.1% / 1% false-positive rates on unrelated code in the benchmarks. With `--embed` the `reimpl`
|
|
65
|
+
defaults rise by 0.07 because blended scores of unrelated code rise too.
|
|
66
|
+
|
|
67
|
+
## Embeddings (bring your own key)
|
|
68
|
+
|
|
69
|
+
Two optional signals, both off by default (the detector stays LLM-free unless you ask):
|
|
70
|
+
|
|
71
|
+
- `--embeddings` embeds identifier *names* (cheap) as an extra name-similarity signal.
|
|
72
|
+
- `--embed <preset>` is the short form (`minilm`, `qwen3`, `potion`, `openai`, `cohere`; `qwen3` is about 5x slower than `minilm` on CPU, so use it with a GPU or prefer `minilm`) and implies
|
|
73
|
+
`--embed-code`. Model ids go into the cache key, so vectors of different models never mix.
|
|
74
|
+
- `--embed-code` embeds the *whole text of every unit* (function, class, file, SQL statement) and blends
|
|
75
|
+
the cosine into the score (`--embed-weight`, default 0.35, others scaled by 1 - weight). It finds
|
|
76
|
+
re-implementations that share no tokens. On function-level LLM re-implementations it lifted Python from
|
|
77
|
+
0.70 to 0.89 on a held-out half; see `eval/REPORT.md` for the numbers and caveats. Vectors are cached per
|
|
78
|
+
unit text in `~/.cache/duplicatecode/embeddings.bin`, so a rescan only embeds what changed.
|
|
79
|
+
|
|
80
|
+
Credentials come from the environment:
|
|
81
|
+
|
|
82
|
+
```sh
|
|
83
|
+
# OpenAI
|
|
84
|
+
export OPENAI_API_KEY=sk-... # optional: OPENAI_EMBEDDING_MODEL (default text-embedding-3-small)
|
|
85
|
+
# Cohere (native /v2/embed)
|
|
86
|
+
export COHERE_API_KEY=... # optional: COHERE_EMBEDDING_MODEL (default embed-v4.0)
|
|
87
|
+
# OpenAI-compatible server (vLLM, Ollama, LiteLLM, ...)
|
|
88
|
+
export OPENAI_BASE_URL=http://localhost:11434/v1 OPENAI_API_KEY=anything OPENAI_EMBEDDING_MODEL=nomic-embed-text
|
|
89
|
+
# Azure AI Foundry (key, or `az login` if no key is set)
|
|
90
|
+
export AZURE_AI_FOUNDRY_ENDPOINT=https://<res>.services.ai.azure.com
|
|
91
|
+
export AZURE_AI_FOUNDRY_API_KEY=... AZURE_AI_FOUNDRY_EMBEDDING_DEPLOYMENT=text-embedding-3-small
|
|
92
|
+
# Azure OpenAI: AZURE_OPENAI_ENDPOINT / AZURE_OPENAI_API_KEY / AZURE_OPENAI_EMBEDDING_DEPLOYMENT
|
|
93
|
+
# Fully local, no key: start the server once (CPU; it loads whichever model a request names)
|
|
94
|
+
# uv run --no-project --python 3.12 eval/embed_server.py
|
|
95
|
+
# duplicatecode scan . --embed minilm # or: qwen3, potion (default endpoint 127.0.0.1:8099)
|
|
96
|
+
# Hosted presets: --embed openai (OPENAI_API_KEY), --embed cohere (COHERE_API_KEY)
|
|
97
|
+
|
|
98
|
+
duplicatecode embed-test fetchUser getUser # check credentials
|
|
99
|
+
duplicatecode scan . --embed-code # whole-unit embeddings
|
|
100
|
+
duplicatecode scan . --embeddings # names only
|
|
101
|
+
```
|
|
102
|
+
|
|
103
|
+
## API
|
|
104
|
+
|
|
105
|
+
### Command line
|
|
106
|
+
|
|
107
|
+
| command | what it does |
|
|
108
|
+
| --- | --- |
|
|
109
|
+
| `scan <paths>` | similar units inside one or more folders; `--pairs`, `--explain`, `--json`, `--fail-on-found` |
|
|
110
|
+
| `diff` | check code added in a git diff (stdin or `--diff`) against the repo (`--repo`) |
|
|
111
|
+
| `review` | ranked candidates for a diff, for a human or an agent to judge |
|
|
112
|
+
| `fragments <paths>` | copied blocks of identical statements inside different functions |
|
|
113
|
+
| `find "<description>" <paths>` | existing units closest to a plain-language description (needs embeddings) |
|
|
114
|
+
| `units <path>` / `show file:10-40` | list extracted units / print a unit's source |
|
|
115
|
+
| `embed-test <names...>` | check the embedding provider and calibrate `--embed-floor` |
|
|
116
|
+
| `eval-groups`, `bench`, `make-mutation-groups` | evaluation (see `eval/README.md`) |
|
|
117
|
+
|
|
118
|
+
Shared options: `--profile copies|reimpl`, `--threshold`, `--min-tokens`, `--min-lines`, `--exclude`, `--skip-tests`,
|
|
119
|
+
`--cross-file`, `--json`. Embedding options (all off by default): `--embed <preset>`, `--embed-code`,
|
|
120
|
+
`--embed-weight`, `--embed-max-chars`, `--embeddings`, `--embed-cache`, `--embed-dims`.
|
|
121
|
+
|
|
122
|
+
```sh
|
|
123
|
+
duplicatecode scan . --profile reimpl --min-name 0 --embed minilm --pairs --explain
|
|
124
|
+
duplicatecode fragments src/ --min-stmts 4 --min-tokens 30 --cross-file --json
|
|
125
|
+
duplicatecode find "retry with exponential backoff" src/ --top 5 --json
|
|
126
|
+
duplicatecode diff --repo . < change.diff --embed openai
|
|
127
|
+
```
|
|
128
|
+
|
|
129
|
+
JSON output:
|
|
130
|
+
|
|
131
|
+
- `find --json`: `[{score, file, name, kind, start_line, end_line}]`
|
|
132
|
+
- `fragments --json`: `[{a, b, statements, tokens}]` with `a`/`b` = `{file, unit, start_line, end_line, coverage}`
|
|
133
|
+
- `scan --pairs --json`: `[{query, candidate, scores}]`; with `--explain`: `[{match, explanation}]`
|
|
134
|
+
- `scan --json` (groups): `[{score, units: [{file, name, kind, start_line, end_line}]}]`
|
|
135
|
+
|
|
136
|
+
`scan --fail-on-found` exits with status 1 when anything is reported, for CI.
|
|
137
|
+
|
|
138
|
+
### Rust library
|
|
139
|
+
|
|
140
|
+
The engine crate `duplicatecode-engine` is what the CLI uses:
|
|
141
|
+
|
|
142
|
+
```rust
|
|
143
|
+
use duplicatecode_engine::embed::{embed_unit_code, EmbedConfig, EmbeddingCache};
|
|
144
|
+
use duplicatecode_engine::explain::explain;
|
|
145
|
+
use duplicatecode_engine::fragments::{find_fragments, FragmentOptions};
|
|
146
|
+
use duplicatecode_engine::index::Weights;
|
|
147
|
+
use duplicatecode_engine::{find_matches, load_units_with, Corpus, MatchOptions};
|
|
148
|
+
|
|
149
|
+
let mut units = load_units_with(Path::new("src/"), &[]);
|
|
150
|
+
|
|
151
|
+
// optional: embed every unit; units without a vector simply skip the embedding term
|
|
152
|
+
let cfg = EmbedConfig::from_env(None).ok_or("no embedding provider configured")?;
|
|
153
|
+
let mut cache = EmbeddingCache::load(&EmbeddingCache::default_path());
|
|
154
|
+
embed_unit_code(&mut units, &cfg, &mut cache, 3000)?;
|
|
155
|
+
|
|
156
|
+
// pairs: queries against a corpus
|
|
157
|
+
let weights = Weights::default().with_embed(0.35); // 0.0 = static only
|
|
158
|
+
let corpus = Corpus::new(units.clone());
|
|
159
|
+
let pairs = find_matches(&units, &corpus, MatchOptions { weights, threshold: 0.42, ..Default::default() });
|
|
160
|
+
|
|
161
|
+
// copied blocks, and what differs inside a pair
|
|
162
|
+
let blocks = find_fragments(&units, FragmentOptions::default());
|
|
163
|
+
println!("{}", explain(&units[0], &units[1]).summary());
|
|
164
|
+
```
|
|
165
|
+
|
|
166
|
+
### Not built yet
|
|
167
|
+
|
|
168
|
+
Planned, not available today: a `.duplicatecode.toml` with an `[embed]` section; `--embed-optional` (warn and
|
|
169
|
+
fall back to the static score when the provider is unreachable; today an unreachable provider is an error);
|
|
170
|
+
`--allow-upload` (required for the hosted presets, which send source text out); a `duplicatecode[embed]` Python
|
|
171
|
+
extra that starts the local server on demand; an `Embedder` trait so providers can plug in without HTTP.
|
|
172
|
+
|
|
45
173
|
## How it works
|
|
46
174
|
|
|
47
175
|
Units (functions, methods, classes, arrow-function components) are extracted with tree-sitter and
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
# duplicatecode
|
|
2
2
|
|
|
3
|
-
Static (LLM-free) detection of duplicate / similar code in Python, TypeScript (TSX) and C#,
|
|
3
|
+
Static (LLM-free) detection of duplicate / similar code in Python, TypeScript/JavaScript (TSX/JSX), SQL and C#,
|
|
4
4
|
aimed at catching an LLM re-implementing something that already exists. Input can be a git diff
|
|
5
5
|
checked against existing source.
|
|
6
6
|
|
|
@@ -32,6 +32,134 @@ duplicatecode scan . --pairs --json # machine-readable pairs in
|
|
|
32
32
|
duplicatecode bench --dataset dataset [--file-level] [--mutations] [--negatives <other repo>]
|
|
33
33
|
```
|
|
34
34
|
|
|
35
|
+
## More ways to look
|
|
36
|
+
|
|
37
|
+
```sh
|
|
38
|
+
duplicatecode fragments packages/ # copied blocks inside different functions (4+ identical statements)
|
|
39
|
+
duplicatecode scan . --pairs --explain # say what differs: literals, calls, lines (what to parameterize)
|
|
40
|
+
duplicatecode find "retry with exponential backoff" src/ # does something like this already exist? (needs embeddings)
|
|
41
|
+
```
|
|
42
|
+
|
|
43
|
+
- `fragments` indexes windows of consecutive normalized statements, so a block pasted into another
|
|
44
|
+
function is found even when the surrounding functions differ. On an injection benchmark (renamed
|
|
45
|
+
blocks of real functions pasted into others) it finds 93-97% of blocks of 6+ statements; see
|
|
46
|
+
`eval/REPORT.md`. Tune with `--min-stmts` / `--min-tokens`; constructors and dunders are ignored.
|
|
47
|
+
- `--explain` adds, per pair, the literals and calls only one side has and the source lines with no
|
|
48
|
+
counterpart ("16 of 18/17 statements shared").
|
|
49
|
+
- `find` embeds the query and every unit and ranks by similarity. With MiniLM, 50 task descriptions
|
|
50
|
+
against 168 implementation units rank the right one first 94% of the time; with 1,379 unrelated real
|
|
51
|
+
units mixed in, 82% first and 96% in the top 3 (every query within the top 10).
|
|
52
|
+
- Default thresholds depend on the profile: `copies` keeps its tuned 0.6 (`scan`), 0.4 (`diff`), 0.45
|
|
53
|
+
(`review`); `reimpl` scores live lower, so it defaults to 0.35 / 0.28 / 0.30, which correspond to roughly
|
|
54
|
+
0.1% / 1% false-positive rates on unrelated code in the benchmarks. With `--embed` the `reimpl`
|
|
55
|
+
defaults rise by 0.07 because blended scores of unrelated code rise too.
|
|
56
|
+
|
|
57
|
+
## Embeddings (bring your own key)
|
|
58
|
+
|
|
59
|
+
Two optional signals, both off by default (the detector stays LLM-free unless you ask):
|
|
60
|
+
|
|
61
|
+
- `--embeddings` embeds identifier *names* (cheap) as an extra name-similarity signal.
|
|
62
|
+
- `--embed <preset>` is the short form (`minilm`, `qwen3`, `potion`, `openai`, `cohere`; `qwen3` is about 5x slower than `minilm` on CPU, so use it with a GPU or prefer `minilm`) and implies
|
|
63
|
+
`--embed-code`. Model ids go into the cache key, so vectors of different models never mix.
|
|
64
|
+
- `--embed-code` embeds the *whole text of every unit* (function, class, file, SQL statement) and blends
|
|
65
|
+
the cosine into the score (`--embed-weight`, default 0.35, others scaled by 1 - weight). It finds
|
|
66
|
+
re-implementations that share no tokens. On function-level LLM re-implementations it lifted Python from
|
|
67
|
+
0.70 to 0.89 on a held-out half; see `eval/REPORT.md` for the numbers and caveats. Vectors are cached per
|
|
68
|
+
unit text in `~/.cache/duplicatecode/embeddings.bin`, so a rescan only embeds what changed.
|
|
69
|
+
|
|
70
|
+
Credentials come from the environment:
|
|
71
|
+
|
|
72
|
+
```sh
|
|
73
|
+
# OpenAI
|
|
74
|
+
export OPENAI_API_KEY=sk-... # optional: OPENAI_EMBEDDING_MODEL (default text-embedding-3-small)
|
|
75
|
+
# Cohere (native /v2/embed)
|
|
76
|
+
export COHERE_API_KEY=... # optional: COHERE_EMBEDDING_MODEL (default embed-v4.0)
|
|
77
|
+
# OpenAI-compatible server (vLLM, Ollama, LiteLLM, ...)
|
|
78
|
+
export OPENAI_BASE_URL=http://localhost:11434/v1 OPENAI_API_KEY=anything OPENAI_EMBEDDING_MODEL=nomic-embed-text
|
|
79
|
+
# Azure AI Foundry (key, or `az login` if no key is set)
|
|
80
|
+
export AZURE_AI_FOUNDRY_ENDPOINT=https://<res>.services.ai.azure.com
|
|
81
|
+
export AZURE_AI_FOUNDRY_API_KEY=... AZURE_AI_FOUNDRY_EMBEDDING_DEPLOYMENT=text-embedding-3-small
|
|
82
|
+
# Azure OpenAI: AZURE_OPENAI_ENDPOINT / AZURE_OPENAI_API_KEY / AZURE_OPENAI_EMBEDDING_DEPLOYMENT
|
|
83
|
+
# Fully local, no key: start the server once (CPU; it loads whichever model a request names)
|
|
84
|
+
# uv run --no-project --python 3.12 eval/embed_server.py
|
|
85
|
+
# duplicatecode scan . --embed minilm # or: qwen3, potion (default endpoint 127.0.0.1:8099)
|
|
86
|
+
# Hosted presets: --embed openai (OPENAI_API_KEY), --embed cohere (COHERE_API_KEY)
|
|
87
|
+
|
|
88
|
+
duplicatecode embed-test fetchUser getUser # check credentials
|
|
89
|
+
duplicatecode scan . --embed-code # whole-unit embeddings
|
|
90
|
+
duplicatecode scan . --embeddings # names only
|
|
91
|
+
```
|
|
92
|
+
|
|
93
|
+
## API
|
|
94
|
+
|
|
95
|
+
### Command line
|
|
96
|
+
|
|
97
|
+
| command | what it does |
|
|
98
|
+
| --- | --- |
|
|
99
|
+
| `scan <paths>` | similar units inside one or more folders; `--pairs`, `--explain`, `--json`, `--fail-on-found` |
|
|
100
|
+
| `diff` | check code added in a git diff (stdin or `--diff`) against the repo (`--repo`) |
|
|
101
|
+
| `review` | ranked candidates for a diff, for a human or an agent to judge |
|
|
102
|
+
| `fragments <paths>` | copied blocks of identical statements inside different functions |
|
|
103
|
+
| `find "<description>" <paths>` | existing units closest to a plain-language description (needs embeddings) |
|
|
104
|
+
| `units <path>` / `show file:10-40` | list extracted units / print a unit's source |
|
|
105
|
+
| `embed-test <names...>` | check the embedding provider and calibrate `--embed-floor` |
|
|
106
|
+
| `eval-groups`, `bench`, `make-mutation-groups` | evaluation (see `eval/README.md`) |
|
|
107
|
+
|
|
108
|
+
Shared options: `--profile copies|reimpl`, `--threshold`, `--min-tokens`, `--min-lines`, `--exclude`, `--skip-tests`,
|
|
109
|
+
`--cross-file`, `--json`. Embedding options (all off by default): `--embed <preset>`, `--embed-code`,
|
|
110
|
+
`--embed-weight`, `--embed-max-chars`, `--embeddings`, `--embed-cache`, `--embed-dims`.
|
|
111
|
+
|
|
112
|
+
```sh
|
|
113
|
+
duplicatecode scan . --profile reimpl --min-name 0 --embed minilm --pairs --explain
|
|
114
|
+
duplicatecode fragments src/ --min-stmts 4 --min-tokens 30 --cross-file --json
|
|
115
|
+
duplicatecode find "retry with exponential backoff" src/ --top 5 --json
|
|
116
|
+
duplicatecode diff --repo . < change.diff --embed openai
|
|
117
|
+
```
|
|
118
|
+
|
|
119
|
+
JSON output:
|
|
120
|
+
|
|
121
|
+
- `find --json`: `[{score, file, name, kind, start_line, end_line}]`
|
|
122
|
+
- `fragments --json`: `[{a, b, statements, tokens}]` with `a`/`b` = `{file, unit, start_line, end_line, coverage}`
|
|
123
|
+
- `scan --pairs --json`: `[{query, candidate, scores}]`; with `--explain`: `[{match, explanation}]`
|
|
124
|
+
- `scan --json` (groups): `[{score, units: [{file, name, kind, start_line, end_line}]}]`
|
|
125
|
+
|
|
126
|
+
`scan --fail-on-found` exits with status 1 when anything is reported, for CI.
|
|
127
|
+
|
|
128
|
+
### Rust library
|
|
129
|
+
|
|
130
|
+
The engine crate `duplicatecode-engine` is what the CLI uses:
|
|
131
|
+
|
|
132
|
+
```rust
|
|
133
|
+
use duplicatecode_engine::embed::{embed_unit_code, EmbedConfig, EmbeddingCache};
|
|
134
|
+
use duplicatecode_engine::explain::explain;
|
|
135
|
+
use duplicatecode_engine::fragments::{find_fragments, FragmentOptions};
|
|
136
|
+
use duplicatecode_engine::index::Weights;
|
|
137
|
+
use duplicatecode_engine::{find_matches, load_units_with, Corpus, MatchOptions};
|
|
138
|
+
|
|
139
|
+
let mut units = load_units_with(Path::new("src/"), &[]);
|
|
140
|
+
|
|
141
|
+
// optional: embed every unit; units without a vector simply skip the embedding term
|
|
142
|
+
let cfg = EmbedConfig::from_env(None).ok_or("no embedding provider configured")?;
|
|
143
|
+
let mut cache = EmbeddingCache::load(&EmbeddingCache::default_path());
|
|
144
|
+
embed_unit_code(&mut units, &cfg, &mut cache, 3000)?;
|
|
145
|
+
|
|
146
|
+
// pairs: queries against a corpus
|
|
147
|
+
let weights = Weights::default().with_embed(0.35); // 0.0 = static only
|
|
148
|
+
let corpus = Corpus::new(units.clone());
|
|
149
|
+
let pairs = find_matches(&units, &corpus, MatchOptions { weights, threshold: 0.42, ..Default::default() });
|
|
150
|
+
|
|
151
|
+
// copied blocks, and what differs inside a pair
|
|
152
|
+
let blocks = find_fragments(&units, FragmentOptions::default());
|
|
153
|
+
println!("{}", explain(&units[0], &units[1]).summary());
|
|
154
|
+
```
|
|
155
|
+
|
|
156
|
+
### Not built yet
|
|
157
|
+
|
|
158
|
+
Planned, not available today: a `.duplicatecode.toml` with an `[embed]` section; `--embed-optional` (warn and
|
|
159
|
+
fall back to the static score when the provider is unreachable; today an unreachable provider is an error);
|
|
160
|
+
`--allow-upload` (required for the hosted presets, which send source text out); a `duplicatecode[embed]` Python
|
|
161
|
+
extra that starts the local server on demand; an `Embedder` trait so providers can plug in without HTTP.
|
|
162
|
+
|
|
35
163
|
## How it works
|
|
36
164
|
|
|
37
165
|
Units (functions, methods, classes, arrow-function components) are extracted with tree-sitter and
|
|
@@ -389,6 +389,9 @@ fn fit(samples: &[&Sample]) -> Weights {
|
|
|
389
389
|
let mut w = Weights {
|
|
390
390
|
bias: 0.0,
|
|
391
391
|
w: [0.0; N_FEATURES],
|
|
392
|
+
sql: None,
|
|
393
|
+
file: None,
|
|
394
|
+
renormalize: false,
|
|
392
395
|
name_floor: 0.5,
|
|
393
396
|
};
|
|
394
397
|
for _ in 0..2500 {
|
|
@@ -415,7 +418,7 @@ fn apply(w: &Weights, x: &[f64; N_FEATURES]) -> f64 {
|
|
|
415
418
|
}
|
|
416
419
|
|
|
417
420
|
/// Fraction of positives scoring above the (1 - fpr) quantile of the negatives.
|
|
418
|
-
fn tpr_at_fpr(scores: &[(f64, bool)], fpr: f64) -> f64 {
|
|
421
|
+
pub(crate) fn tpr_at_fpr(scores: &[(f64, bool)], fpr: f64) -> f64 {
|
|
419
422
|
let mut neg: Vec<f64> = scores.iter().filter(|s| !s.1).map(|s| s.0).collect();
|
|
420
423
|
let pos: Vec<f64> = scores.iter().filter(|s| s.1).map(|s| s.0).collect();
|
|
421
424
|
if neg.is_empty() || pos.is_empty() {
|
|
@@ -0,0 +1,267 @@
|
|
|
1
|
+
//! Labeled-group benchmark: files inside one group are clones, files of different groups are not.
|
|
2
|
+
//!
|
|
3
|
+
//! Layout `<root>/<dataset>/<group>/<file>`. Every file is one whole-file unit; all pairs inside a
|
|
4
|
+
//! dataset are scored. Reports AUC, recall at fixed false-positive rates and top-1 retrieval.
|
|
5
|
+
|
|
6
|
+
use crate::bench::tpr_at_fpr;
|
|
7
|
+
use anyhow::{bail, Result};
|
|
8
|
+
use duplicatecode_engine::index::{score_with_idf, Idf, Scores, Weights};
|
|
9
|
+
use duplicatecode_engine::{load_file_units, load_units, Unit};
|
|
10
|
+
use rayon::prelude::*;
|
|
11
|
+
use std::path::Path;
|
|
12
|
+
|
|
13
|
+
type Pick = fn(&Scores) -> f64;
|
|
14
|
+
const SIGNALS: [(&str, Pick); 7] = [
|
|
15
|
+
("structure", |s| s.structural),
|
|
16
|
+
("loose", |s| s.loose),
|
|
17
|
+
("kinds", |s| s.kinds),
|
|
18
|
+
("stmt_exact", |s| s.stmt_exact),
|
|
19
|
+
("stmt_shape", |s| s.stmt_shape),
|
|
20
|
+
("stmt_lcs", |s| s.stmt_lcs),
|
|
21
|
+
("combined", |s| s.combined),
|
|
22
|
+
];
|
|
23
|
+
|
|
24
|
+
/// Area under the ROC curve with tie handling (Mann-Whitney U).
|
|
25
|
+
fn auc(scores: &[(f64, bool)]) -> f64 {
|
|
26
|
+
let mut v: Vec<(f64, bool)> = scores.to_vec();
|
|
27
|
+
v.sort_by(|a, b| a.0.total_cmp(&b.0));
|
|
28
|
+
let (mut rank_sum, mut n_pos, mut i) = (0.0, 0usize, 0usize);
|
|
29
|
+
while i < v.len() {
|
|
30
|
+
let mut j = i;
|
|
31
|
+
while j < v.len() && v[j].0 == v[i].0 {
|
|
32
|
+
j += 1;
|
|
33
|
+
}
|
|
34
|
+
let avg_rank = (i + j + 1) as f64 / 2.0; // 1-based average rank of the tie block
|
|
35
|
+
let pos = v[i..j].iter().filter(|x| x.1).count();
|
|
36
|
+
rank_sum += avg_rank * pos as f64;
|
|
37
|
+
n_pos += pos;
|
|
38
|
+
i = j;
|
|
39
|
+
}
|
|
40
|
+
let n_neg = v.len() - n_pos;
|
|
41
|
+
if n_pos == 0 || n_neg == 0 {
|
|
42
|
+
return f64::NAN;
|
|
43
|
+
}
|
|
44
|
+
(rank_sum - (n_pos * (n_pos + 1)) as f64 / 2.0) / (n_pos * n_neg) as f64
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
fn group_of(u: &Unit) -> &str {
|
|
48
|
+
u.file.split('/').next().unwrap_or("")
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
/// Score above which only `fpr` of the negative pairs lie.
|
|
52
|
+
fn score_at_fpr(scores: &[(f64, bool)], fpr: f64) -> f64 {
|
|
53
|
+
let mut neg: Vec<f64> = scores.iter().filter(|s| !s.1).map(|s| s.0).collect();
|
|
54
|
+
neg.sort_by(|a, b| a.total_cmp(b));
|
|
55
|
+
neg.get(((1.0 - fpr) * neg.len() as f64) as usize)
|
|
56
|
+
.or(neg.last())
|
|
57
|
+
.copied()
|
|
58
|
+
.unwrap_or(f64::NAN)
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
struct Report {
|
|
62
|
+
thr1: f64,
|
|
63
|
+
thr01: f64,
|
|
64
|
+
auc: f64,
|
|
65
|
+
tpr1: f64,
|
|
66
|
+
tpr5: f64,
|
|
67
|
+
top1: f64,
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
fn evaluate(units: &[Unit], pick: Pick, all: &[Vec<Scores>]) -> Report {
|
|
71
|
+
let n = units.len();
|
|
72
|
+
let mut pairs = Vec::with_capacity(n * n / 2);
|
|
73
|
+
let mut hit = 0usize;
|
|
74
|
+
for i in 0..n {
|
|
75
|
+
let mut best: Option<(f64, bool)> = None;
|
|
76
|
+
for j in 0..n {
|
|
77
|
+
if i == j || units[i].file == units[j].file {
|
|
78
|
+
continue;
|
|
79
|
+
}
|
|
80
|
+
let v = pick(&all[i][j]);
|
|
81
|
+
let same = group_of(&units[i]) == group_of(&units[j]);
|
|
82
|
+
if j > i {
|
|
83
|
+
pairs.push((v, same));
|
|
84
|
+
}
|
|
85
|
+
if best.is_none_or(|b| v > b.0) {
|
|
86
|
+
best = Some((v, same));
|
|
87
|
+
}
|
|
88
|
+
}
|
|
89
|
+
hit += best.is_some_and(|b| b.1) as usize;
|
|
90
|
+
}
|
|
91
|
+
Report {
|
|
92
|
+
thr1: score_at_fpr(&pairs, 0.01),
|
|
93
|
+
thr01: score_at_fpr(&pairs, 0.001),
|
|
94
|
+
auc: auc(&pairs),
|
|
95
|
+
tpr1: tpr_at_fpr(&pairs, 0.01),
|
|
96
|
+
tpr5: tpr_at_fpr(&pairs, 0.05),
|
|
97
|
+
top1: hit as f64 / n as f64,
|
|
98
|
+
}
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
pub fn run(
|
|
102
|
+
root: &Path,
|
|
103
|
+
quiet: bool,
|
|
104
|
+
dump: Option<&Path>,
|
|
105
|
+
prepare: &dyn Fn(&mut [Unit]) -> Result<()>,
|
|
106
|
+
weights: Weights,
|
|
107
|
+
) -> Result<()> {
|
|
108
|
+
let mut dump_out = dump.map(std::fs::File::create).transpose()?;
|
|
109
|
+
let mut datasets: Vec<_> = std::fs::read_dir(root)?
|
|
110
|
+
.filter_map(Result::ok)
|
|
111
|
+
.map(|e| e.path())
|
|
112
|
+
.filter(|p| p.is_dir() && !p.file_name().unwrap().to_string_lossy().starts_with('_'))
|
|
113
|
+
.collect();
|
|
114
|
+
datasets.sort();
|
|
115
|
+
if datasets.is_empty() {
|
|
116
|
+
bail!(
|
|
117
|
+
"no datasets below {} (run eval/fetch_codenet.py)",
|
|
118
|
+
root.display()
|
|
119
|
+
);
|
|
120
|
+
}
|
|
121
|
+
let mut headline = Vec::new();
|
|
122
|
+
for dir in datasets {
|
|
123
|
+
let name = dir.file_name().unwrap().to_string_lossy().to_string();
|
|
124
|
+
// `fn-*` datasets compare functions/classes (product path); the others whole files
|
|
125
|
+
let functions = name.starts_with("fn-");
|
|
126
|
+
let mut units = if functions {
|
|
127
|
+
load_units(&dir)
|
|
128
|
+
.into_iter()
|
|
129
|
+
.filter(|u| !u.boilerplate && u.token_count() >= 8)
|
|
130
|
+
.collect()
|
|
131
|
+
} else {
|
|
132
|
+
load_file_units(&dir)
|
|
133
|
+
};
|
|
134
|
+
units.sort_by(|a, b| (&a.file, a.start_line).cmp(&(&b.file, b.start_line)));
|
|
135
|
+
prepare(&mut units)?;
|
|
136
|
+
let n = units.len();
|
|
137
|
+
let idf = Idf::from_units(&units);
|
|
138
|
+
// full score matrix once; each signal is a projection of it
|
|
139
|
+
let all: Vec<Vec<Scores>> = (0..n)
|
|
140
|
+
.into_par_iter()
|
|
141
|
+
.map(|i| {
|
|
142
|
+
(0..n)
|
|
143
|
+
.map(|j| score_with_idf(&units[i], &units[j], &weights, Some(&idf)))
|
|
144
|
+
.collect()
|
|
145
|
+
})
|
|
146
|
+
.collect();
|
|
147
|
+
if let Some(f) = dump_out.as_mut() {
|
|
148
|
+
use std::io::Write;
|
|
149
|
+
for i in 0..n {
|
|
150
|
+
for j in i + 1..n {
|
|
151
|
+
if units[i].file == units[j].file {
|
|
152
|
+
continue;
|
|
153
|
+
}
|
|
154
|
+
let feats: Vec<String> = all[i][j]
|
|
155
|
+
.features()
|
|
156
|
+
.iter()
|
|
157
|
+
.map(|x| format!("{x:.5}"))
|
|
158
|
+
.collect();
|
|
159
|
+
writeln!(
|
|
160
|
+
f,
|
|
161
|
+
"{name}\t{}:{}\t{}:{}\t{}\t{}\t{}\t{}\t{}",
|
|
162
|
+
units[i].file,
|
|
163
|
+
units[i].start_line,
|
|
164
|
+
units[j].file,
|
|
165
|
+
units[j].start_line,
|
|
166
|
+
group_of(&units[i]),
|
|
167
|
+
group_of(&units[j]),
|
|
168
|
+
units[i].tokens.len().min(units[j].tokens.len()),
|
|
169
|
+
units[i].tokens.len().max(units[j].tokens.len()),
|
|
170
|
+
feats.join("\t")
|
|
171
|
+
)?;
|
|
172
|
+
}
|
|
173
|
+
}
|
|
174
|
+
}
|
|
175
|
+
let groups: std::collections::BTreeSet<&str> = units.iter().map(group_of).collect();
|
|
176
|
+
if !quiet {
|
|
177
|
+
println!("== {name}: {n} files, {} groups", groups.len());
|
|
178
|
+
println!(
|
|
179
|
+
"{:<12} {:>6} {:>8} {:>8} {:>7}",
|
|
180
|
+
"signal", "AUC", "TPR@1%", "TPR@5%", "top1"
|
|
181
|
+
);
|
|
182
|
+
}
|
|
183
|
+
for (label, pick) in SIGNALS {
|
|
184
|
+
let r = evaluate(&units, pick, &all);
|
|
185
|
+
if !quiet {
|
|
186
|
+
println!(
|
|
187
|
+
"{label:<12} {:>6.3} {:>8.3} {:>8.3} {:>7.3}",
|
|
188
|
+
r.auc, r.tpr1, r.tpr5, r.top1
|
|
189
|
+
);
|
|
190
|
+
}
|
|
191
|
+
if label == "combined" {
|
|
192
|
+
headline.push((name.clone(), (r.auc + r.tpr1) / 2.0));
|
|
193
|
+
println!(
|
|
194
|
+
"dataset_threshold {name} score@FPR1%={:.3} score@FPR0.1%={:.3}",
|
|
195
|
+
r.thr1, r.thr01
|
|
196
|
+
);
|
|
197
|
+
}
|
|
198
|
+
}
|
|
199
|
+
}
|
|
200
|
+
for (name, s) in &headline {
|
|
201
|
+
println!("dataset_score {name}={s:.4}");
|
|
202
|
+
}
|
|
203
|
+
let mean = headline.iter().map(|h| h.1).sum::<f64>() / headline.len() as f64;
|
|
204
|
+
println!("score={mean:.4}");
|
|
205
|
+
Ok(())
|
|
206
|
+
}
|
|
207
|
+
|
|
208
|
+
/// Build `<out>/<id>/<variant>.tsx` groups from real components: the original plus mechanically
|
|
209
|
+
/// rewritten variants (renames, statement swaps, temp variables, logging, dead code, all combined).
|
|
210
|
+
/// Other components act as negatives. `skip` lets a holdout set use disjoint files.
|
|
211
|
+
pub fn make_mutation_groups(
|
|
212
|
+
src: &Path,
|
|
213
|
+
out: &Path,
|
|
214
|
+
n: usize,
|
|
215
|
+
seed: u64,
|
|
216
|
+
skip: usize,
|
|
217
|
+
) -> Result<()> {
|
|
218
|
+
use duplicatecode_engine::mutate::{apply, Mutation};
|
|
219
|
+
use duplicatecode_engine::{walk_files, Lang};
|
|
220
|
+
let mut files: Vec<_> = walk_files(src, &[])
|
|
221
|
+
.into_iter()
|
|
222
|
+
.filter(|p| {
|
|
223
|
+
let s = p.to_string_lossy();
|
|
224
|
+
s.ends_with(".tsx")
|
|
225
|
+
&& !s.contains(".test.")
|
|
226
|
+
&& !s.contains(".stories.")
|
|
227
|
+
&& !s.contains("/generated/")
|
|
228
|
+
})
|
|
229
|
+
.filter(|p| {
|
|
230
|
+
std::fs::read_to_string(p)
|
|
231
|
+
.map(|t| (30..=300).contains(&t.lines().count()) && t.contains("</"))
|
|
232
|
+
.unwrap_or(false)
|
|
233
|
+
})
|
|
234
|
+
.collect();
|
|
235
|
+
files.sort();
|
|
236
|
+
let mut state = seed.wrapping_mul(0x9E37_79B9_7F4A_7C15) | 1;
|
|
237
|
+
let mut next = || {
|
|
238
|
+
state ^= state << 13;
|
|
239
|
+
state ^= state >> 7;
|
|
240
|
+
state ^= state << 17;
|
|
241
|
+
state
|
|
242
|
+
};
|
|
243
|
+
for i in (1..files.len()).rev() {
|
|
244
|
+
files.swap(i, (next() % (i as u64 + 1)) as usize);
|
|
245
|
+
}
|
|
246
|
+
for (k, p) in files.iter().skip(skip).take(n).enumerate() {
|
|
247
|
+
let text = std::fs::read_to_string(p)?;
|
|
248
|
+
let dir = out.join(format!("c{:03}", skip + k));
|
|
249
|
+
std::fs::create_dir_all(&dir)?;
|
|
250
|
+
std::fs::write(dir.join("orig.tsx"), &text)?;
|
|
251
|
+
for m in Mutation::ALL {
|
|
252
|
+
if m == Mutation::LoopToComprehension {
|
|
253
|
+
continue; // Python only
|
|
254
|
+
}
|
|
255
|
+
let v = apply(m, Lang::Tsx, &text);
|
|
256
|
+
if v != text {
|
|
257
|
+
std::fs::write(dir.join(format!("{}.tsx", m.name().replace(' ', "_"))), v)?;
|
|
258
|
+
}
|
|
259
|
+
}
|
|
260
|
+
}
|
|
261
|
+
println!(
|
|
262
|
+
"wrote {} component groups to {}",
|
|
263
|
+
files.len().saturating_sub(skip).min(n),
|
|
264
|
+
out.display()
|
|
265
|
+
);
|
|
266
|
+
Ok(())
|
|
267
|
+
}
|