duplicatecode 0.2.0__tar.gz → 0.3.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {duplicatecode-0.2.0 → duplicatecode-0.3.1}/Cargo.lock +15 -4
- {duplicatecode-0.2.0 → duplicatecode-0.3.1}/Cargo.toml +1 -1
- {duplicatecode-0.2.0 → duplicatecode-0.3.1}/PKG-INFO +142 -7
- {duplicatecode-0.2.0 → duplicatecode-0.3.1}/README.md +140 -5
- {duplicatecode-0.2.0 → duplicatecode-0.3.1}/crates/duplicatecode-cli/src/bench.rs +4 -1
- duplicatecode-0.3.1/crates/duplicatecode-cli/src/groups.rs +269 -0
- {duplicatecode-0.2.0 → duplicatecode-0.3.1}/crates/duplicatecode-cli/src/main.rs +446 -34
- {duplicatecode-0.2.0 → duplicatecode-0.3.1}/crates/duplicatecode-engine/Cargo.toml +2 -1
- duplicatecode-0.3.1/crates/duplicatecode-engine/src/embed.rs +1105 -0
- duplicatecode-0.3.1/crates/duplicatecode-engine/src/explain.rs +130 -0
- duplicatecode-0.3.1/crates/duplicatecode-engine/src/fragments.rs +201 -0
- {duplicatecode-0.2.0 → duplicatecode-0.3.1}/crates/duplicatecode-engine/src/index.rs +340 -36
- {duplicatecode-0.2.0 → duplicatecode-0.3.1}/crates/duplicatecode-engine/src/lang.rs +6 -1
- {duplicatecode-0.2.0 → duplicatecode-0.3.1}/crates/duplicatecode-engine/src/lib.rs +2 -0
- {duplicatecode-0.2.0 → duplicatecode-0.3.1}/crates/duplicatecode-engine/src/similarity.rs +32 -0
- {duplicatecode-0.2.0 → duplicatecode-0.3.1}/crates/duplicatecode-engine/src/units.rs +699 -9
- {duplicatecode-0.2.0 → duplicatecode-0.3.1}/pyproject.toml +1 -1
- duplicatecode-0.2.0/crates/duplicatecode-engine/src/embed.rs +0 -552
- {duplicatecode-0.2.0 → duplicatecode-0.3.1}/crates/duplicatecode-cli/Cargo.toml +0 -0
- {duplicatecode-0.2.0 → duplicatecode-0.3.1}/crates/duplicatecode-cli/src/review.rs +0 -0
- {duplicatecode-0.2.0 → duplicatecode-0.3.1}/crates/duplicatecode-engine/src/diff.rs +0 -0
- {duplicatecode-0.2.0 → duplicatecode-0.3.1}/crates/duplicatecode-engine/src/fingerprint.rs +0 -0
- {duplicatecode-0.2.0 → duplicatecode-0.3.1}/crates/duplicatecode-engine/src/mutate.rs +0 -0
- {duplicatecode-0.2.0 → duplicatecode-0.3.1}/crates/duplicatecode-engine/src/naming.rs +0 -0
|
@@ -91,9 +91,9 @@ dependencies = [
|
|
|
91
91
|
|
|
92
92
|
[[package]]
|
|
93
93
|
name = "cc"
|
|
94
|
-
version = "1.
|
|
94
|
+
version = "1.2.67"
|
|
95
95
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
|
96
|
-
checksum = "
|
|
96
|
+
checksum = "e17dd265a7d0f31ef544e1b20e03add05d3b45b491b633b10d67145d2acc1a38"
|
|
97
97
|
dependencies = [
|
|
98
98
|
"find-msvc-tools",
|
|
99
99
|
"shlex",
|
|
@@ -198,7 +198,7 @@ dependencies = [
|
|
|
198
198
|
|
|
199
199
|
[[package]]
|
|
200
200
|
name = "duplicatecode"
|
|
201
|
-
version = "0.
|
|
201
|
+
version = "0.3.1"
|
|
202
202
|
dependencies = [
|
|
203
203
|
"anyhow",
|
|
204
204
|
"clap",
|
|
@@ -212,7 +212,7 @@ dependencies = [
|
|
|
212
212
|
|
|
213
213
|
[[package]]
|
|
214
214
|
name = "duplicatecode-engine"
|
|
215
|
-
version = "0.
|
|
215
|
+
version = "0.3.1"
|
|
216
216
|
dependencies = [
|
|
217
217
|
"ignore",
|
|
218
218
|
"serde",
|
|
@@ -220,6 +220,7 @@ dependencies = [
|
|
|
220
220
|
"tree-sitter",
|
|
221
221
|
"tree-sitter-c-sharp",
|
|
222
222
|
"tree-sitter-python",
|
|
223
|
+
"tree-sitter-sequel",
|
|
223
224
|
"tree-sitter-typescript",
|
|
224
225
|
"ureq",
|
|
225
226
|
"walkdir",
|
|
@@ -790,6 +791,16 @@ dependencies = [
|
|
|
790
791
|
"tree-sitter-language",
|
|
791
792
|
]
|
|
792
793
|
|
|
794
|
+
[[package]]
|
|
795
|
+
name = "tree-sitter-sequel"
|
|
796
|
+
version = "0.3.11"
|
|
797
|
+
source = "registry+https://github.com/rust-lang/crates.io-index"
|
|
798
|
+
checksum = "9d198ad3c319c02e43c21efa1ec796b837afcb96ffaef1a40c1978fbdcec7d17"
|
|
799
|
+
dependencies = [
|
|
800
|
+
"cc",
|
|
801
|
+
"tree-sitter-language",
|
|
802
|
+
]
|
|
803
|
+
|
|
793
804
|
[[package]]
|
|
794
805
|
name = "tree-sitter-typescript"
|
|
795
806
|
version = "0.23.2"
|
|
@@ -1,16 +1,16 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: duplicatecode
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.3.1
|
|
4
4
|
Classifier: Programming Language :: Rust
|
|
5
5
|
Classifier: Environment :: Console
|
|
6
|
-
Summary: Static (LLM-free) detection of duplicate/similar code in Python, TypeScript and C#
|
|
6
|
+
Summary: Static (LLM-free) detection of duplicate/similar code in Python, TypeScript/JavaScript, SQL and C#
|
|
7
7
|
License: MIT
|
|
8
8
|
Requires-Python: >=3.9
|
|
9
9
|
Description-Content-Type: text/markdown; charset=UTF-8; variant=GFM
|
|
10
10
|
|
|
11
11
|
# duplicatecode
|
|
12
12
|
|
|
13
|
-
Static (LLM-free) detection of duplicate / similar code in Python, TypeScript (TSX) and C#,
|
|
13
|
+
Static by default (LLM-free, no network unless `--embed*` is used) detection of duplicate / similar code in Python, TypeScript/JavaScript (TSX/JSX), SQL and C#,
|
|
14
14
|
aimed at catching an LLM re-implementing something that already exists. Input can be a git diff
|
|
15
15
|
checked against existing source.
|
|
16
16
|
|
|
@@ -42,6 +42,141 @@ duplicatecode scan . --pairs --json # machine-readable pairs in
|
|
|
42
42
|
duplicatecode bench --dataset dataset [--file-level] [--mutations] [--negatives <other repo>]
|
|
43
43
|
```
|
|
44
44
|
|
|
45
|
+
## More ways to look
|
|
46
|
+
|
|
47
|
+
```sh
|
|
48
|
+
duplicatecode fragments packages/ # copied blocks inside different functions (4+ identical statements)
|
|
49
|
+
duplicatecode scan . --pairs --explain # say what differs: literals, calls, lines (what to parameterize)
|
|
50
|
+
duplicatecode find "retry with exponential backoff" src/ # does something like this already exist? (needs embeddings)
|
|
51
|
+
```
|
|
52
|
+
|
|
53
|
+
- `fragments` indexes windows of consecutive normalized statements, so a block pasted into another
|
|
54
|
+
function is found even when the surrounding functions differ. On an injection benchmark (renamed
|
|
55
|
+
blocks of real functions pasted into others) it finds 93-97% of blocks of 6+ statements; see
|
|
56
|
+
`eval/REPORT.md`. Tune with `--min-stmts` / `--min-tokens`; constructors and dunders are ignored.
|
|
57
|
+
- `--explain` adds, per pair, the literals and calls only one side has and the source lines with no
|
|
58
|
+
counterpart ("16 of 18/17 statements shared").
|
|
59
|
+
- `find` embeds the query and every unit and ranks by similarity. With MiniLM, 50 task descriptions
|
|
60
|
+
against 168 implementation units rank the right one first 94% of the time; with 1,379 unrelated real
|
|
61
|
+
units mixed in, 82% first and 96% in the top 3 (every query within the top 10).
|
|
62
|
+
- Default thresholds depend on the profile: `copies` keeps its tuned 0.6 (`scan`), 0.4 (`diff`), 0.45
|
|
63
|
+
(`review`); `reimpl` scores live lower, so it defaults to 0.35 / 0.28 / 0.30, which correspond to roughly
|
|
64
|
+
0.1% / 1% false-positive rates on unrelated code in the benchmarks. With `--embed` the `reimpl`
|
|
65
|
+
defaults rise by 0.07 because blended scores of unrelated code rise too.
|
|
66
|
+
|
|
67
|
+
## Embeddings (bring your own key)
|
|
68
|
+
|
|
69
|
+
Two optional signals, both off by default (the detector stays static and offline unless you ask).
|
|
70
|
+
|
|
71
|
+
> **Privacy warning.** `--embed openai`, `--embed cohere`, and `--embed-code` / `--embeddings` pointed at any
|
|
72
|
+
> hosted endpoint (OpenAI, Cohere, Azure, ...) **upload the source text of your units** to that provider
|
|
73
|
+
> (`--embeddings` sends identifier names only). `find` embeds **every unit under the given paths**, so
|
|
74
|
+
> `find ... --embed openai` uploads all of it. The local presets (`minilm`, `qwen3`, `potion`) are meant for a
|
|
75
|
+
> server on loopback (default `127.0.0.1:8099`; `DUPLICATECODE_EMBED_ENDPOINT` overrides it): pointing that
|
|
76
|
+
> variable at a remote host sends text there too. Do not use hosted embeddings on code you may not share.
|
|
77
|
+
|
|
78
|
+
- `--embeddings` embeds identifier *names* (cheap) as an extra name-similarity signal.
|
|
79
|
+
- `--embed <preset>` is the short form (`minilm`, `qwen3`, `potion`, `openai`, `cohere`; `qwen3` is about 5x slower than `minilm` on CPU, so use it with a GPU or prefer `minilm`) and implies
|
|
80
|
+
`--embed-code`. Model ids go into the cache key, so vectors of different models never mix.
|
|
81
|
+
- `--embed-code` embeds the *whole text of every unit* (function, class, file, SQL statement) and blends
|
|
82
|
+
the cosine into the score (`--embed-weight`, default 0.35, others scaled by 1 - weight). It finds
|
|
83
|
+
re-implementations that share no tokens. On function-level LLM re-implementations it lifted Python from
|
|
84
|
+
0.70 to 0.89 on a held-out half; see `eval/REPORT.md` for the numbers and caveats. Vectors are cached per
|
|
85
|
+
unit text in `~/.cache/duplicatecode/embeddings.bin`, so a rescan only embeds what changed.
|
|
86
|
+
|
|
87
|
+
Credentials come from the environment:
|
|
88
|
+
|
|
89
|
+
```sh
|
|
90
|
+
# OpenAI
|
|
91
|
+
export OPENAI_API_KEY=sk-... # optional: OPENAI_EMBEDDING_MODEL (default text-embedding-3-small)
|
|
92
|
+
# Cohere (native /v2/embed)
|
|
93
|
+
export COHERE_API_KEY=... # optional: COHERE_EMBEDDING_MODEL (default embed-v4.0)
|
|
94
|
+
# OpenAI-compatible server (vLLM, Ollama, LiteLLM, ...)
|
|
95
|
+
export OPENAI_BASE_URL=http://localhost:11434/v1 OPENAI_API_KEY=anything OPENAI_EMBEDDING_MODEL=nomic-embed-text
|
|
96
|
+
# Azure AI Foundry (key, or `az login` if no key is set)
|
|
97
|
+
export AZURE_AI_FOUNDRY_ENDPOINT=https://<res>.services.ai.azure.com
|
|
98
|
+
export AZURE_AI_FOUNDRY_API_KEY=... AZURE_AI_FOUNDRY_EMBEDDING_DEPLOYMENT=text-embedding-3-small
|
|
99
|
+
# Azure OpenAI: AZURE_OPENAI_ENDPOINT / AZURE_OPENAI_API_KEY / AZURE_OPENAI_EMBEDDING_DEPLOYMENT
|
|
100
|
+
# Fully local, no key: start the server once (CPU; it loads whichever model a request names)
|
|
101
|
+
# uv run --no-project --python 3.12 eval/embed_server.py
|
|
102
|
+
# duplicatecode scan . --embed minilm # or: qwen3, potion (default endpoint 127.0.0.1:8099)
|
|
103
|
+
# Hosted presets: --embed openai (OPENAI_API_KEY), --embed cohere (COHERE_API_KEY)
|
|
104
|
+
|
|
105
|
+
duplicatecode embed-test fetchUser getUser # check credentials
|
|
106
|
+
duplicatecode scan . --embed-code # whole-unit embeddings
|
|
107
|
+
duplicatecode scan . --embeddings # names only
|
|
108
|
+
```
|
|
109
|
+
|
|
110
|
+
## API
|
|
111
|
+
|
|
112
|
+
### Command line
|
|
113
|
+
|
|
114
|
+
| command | what it does |
|
|
115
|
+
| --- | --- |
|
|
116
|
+
| `scan <paths>` | similar units inside one or more folders; `--pairs`, `--explain`, `--json`, `--fail-on-found` |
|
|
117
|
+
| `diff` | check code added in a git diff (stdin or `--diff`) against the repo (`--repo`) |
|
|
118
|
+
| `review` | ranked candidates for a diff, for a human or an agent to judge |
|
|
119
|
+
| `fragments <paths>` | copied blocks of identical statements inside different functions |
|
|
120
|
+
| `find "<description>" <paths>` | existing units closest to a plain-language description (needs embeddings) |
|
|
121
|
+
| `units <path>` / `show file:10-40` | list extracted units / print a unit's source |
|
|
122
|
+
| `embed-test <names...>` | check the embedding provider and calibrate `--embed-floor` |
|
|
123
|
+
| `eval-groups`, `bench`, `make-mutation-groups` | evaluation (see `eval/README.md`) |
|
|
124
|
+
|
|
125
|
+
Shared options: `--profile copies|reimpl`, `--threshold`, `--min-tokens`, `--min-lines`, `--exclude`, `--skip-tests`,
|
|
126
|
+
`--cross-file`, `--json`. Embedding options (all off by default): `--embed <preset>`, `--embed-code`,
|
|
127
|
+
`--embed-weight`, `--embed-max-chars`, `--embeddings`, `--embed-cache`, `--embed-dims`.
|
|
128
|
+
|
|
129
|
+
```sh
|
|
130
|
+
duplicatecode scan . --profile reimpl --min-name 0 --embed minilm --pairs --explain
|
|
131
|
+
duplicatecode fragments src/ --min-stmts 4 --min-tokens 30 --cross-file --json
|
|
132
|
+
duplicatecode find "retry with exponential backoff" src/ --top 5 --json
|
|
133
|
+
duplicatecode diff --repo . < change.diff --embed openai
|
|
134
|
+
```
|
|
135
|
+
|
|
136
|
+
JSON output:
|
|
137
|
+
|
|
138
|
+
- `find --json`: `[{score, file, name, kind, start_line, end_line}]`
|
|
139
|
+
- `fragments --json`: `[{a, b, statements, tokens}]` with `a`/`b` = `{file, unit, start_line, end_line, coverage}`
|
|
140
|
+
- `scan --pairs --json`: `[{query, candidate, scores}]`; with `--explain`: `[{match, explanation}]`
|
|
141
|
+
- `scan --json` (groups): `[{score, units: [{file, name, kind, start_line, end_line}]}]`
|
|
142
|
+
|
|
143
|
+
`scan --fail-on-found` exits with status 1 when anything is reported, for CI.
|
|
144
|
+
|
|
145
|
+
### Rust library
|
|
146
|
+
|
|
147
|
+
The engine crate `duplicatecode-engine` is what the CLI uses:
|
|
148
|
+
|
|
149
|
+
```rust
|
|
150
|
+
use duplicatecode_engine::embed::{embed_unit_code, EmbedConfig, EmbeddingCache};
|
|
151
|
+
use duplicatecode_engine::explain::explain;
|
|
152
|
+
use duplicatecode_engine::fragments::{find_fragments, FragmentOptions};
|
|
153
|
+
use duplicatecode_engine::index::Weights;
|
|
154
|
+
use duplicatecode_engine::{find_matches, load_units_with, Corpus, MatchOptions};
|
|
155
|
+
|
|
156
|
+
let mut units = load_units_with(Path::new("src/"), &[]);
|
|
157
|
+
|
|
158
|
+
// optional: embed every unit; units without a vector simply skip the embedding term
|
|
159
|
+
let cfg = EmbedConfig::from_env(None).ok_or("no embedding provider configured")?;
|
|
160
|
+
let mut cache = EmbeddingCache::load(&EmbeddingCache::default_path());
|
|
161
|
+
embed_unit_code(&mut units, &cfg, &mut cache, 3000)?;
|
|
162
|
+
|
|
163
|
+
// pairs: queries against a corpus
|
|
164
|
+
let weights = Weights::default().with_embed(0.35); // 0.0 = static only
|
|
165
|
+
let corpus = Corpus::new(units.clone());
|
|
166
|
+
let pairs = find_matches(&units, &corpus, MatchOptions { weights, threshold: 0.42, ..Default::default() });
|
|
167
|
+
|
|
168
|
+
// copied blocks, and what differs inside a pair
|
|
169
|
+
let blocks = find_fragments(&units, FragmentOptions::default());
|
|
170
|
+
println!("{}", explain(&units[0], &units[1]).summary());
|
|
171
|
+
```
|
|
172
|
+
|
|
173
|
+
### Not built yet
|
|
174
|
+
|
|
175
|
+
Planned, not available today: a `.duplicatecode.toml` with an `[embed]` section; `--embed-optional` (warn and
|
|
176
|
+
fall back to the static score when the provider is unreachable; today an unreachable provider is an error);
|
|
177
|
+
`--allow-upload` (required for the hosted presets, which send source text out); a `duplicatecode[embed]` Python
|
|
178
|
+
extra that starts the local server on demand; an `Embedder` trait so providers can plug in without HTTP.
|
|
179
|
+
|
|
45
180
|
## How it works
|
|
46
181
|
|
|
47
182
|
Units (functions, methods, classes, arrow-function components) are extracted with tree-sitter and
|
|
@@ -87,9 +222,9 @@ Self-scans of three internal repositories (Python + TypeScript), 189 pairs judge
|
|
|
87
222
|
before trusting them. A structure-heavy `--profile reimpl` exists for renamed re-implementations
|
|
88
223
|
(use with `--min-name 0`) but has no real-repo precision data yet.
|
|
89
224
|
|
|
90
|
-
### Fresh held-out check (copies profile, threshold 0.6)
|
|
225
|
+
### Fresh held-out check (copies profile, threshold 0.6 = the `copies` default for `scan`)
|
|
91
226
|
|
|
92
|
-
A third, untouched sample of 57 pairs (nothing was tuned on it): 30% true duplicates, 47% incl. partial
|
|
227
|
+
A third, untouched sample of 57 pairs (nothing was tuned on it; the `reimpl` profile has different defaults, 0.35 for `scan`, and is not covered by this sample): 30% true duplicates, 47% incl. partial
|
|
93
228
|
(OneSales 0/24, MDMApp 11/24, CCMT2 6/9 true). The 50%/76% above was optimistic because it was measured on the
|
|
94
229
|
pairs used to choose the weights. Test code is about half of the noise but also holds real copies
|
|
95
230
|
(25% true either way), so it is kept by default; `--skip-tests` drops it.
|
|
@@ -107,7 +242,7 @@ by Haiku and Sonnet without seeing the repo. Fraction where the original is foun
|
|
|
107
242
|
| 0.5 | 45% | 14% | 30% | 1% |
|
|
108
243
|
| 0.6 | 24% | 5% | 14% | 0% |
|
|
109
244
|
|
|
110
|
-
So `diff` (checking new code) defaults to threshold 0.4 while `scan` (existing copies) defaults to 0.6. About half of
|
|
245
|
+
So with the `copies` profile `diff` (checking new code) defaults to threshold 0.4 while `scan` (existing copies) defaults to 0.6; the `reimpl` profile has its own lower defaults (see above). About half of
|
|
111
246
|
independent re-implementations are caught; the rest are genuinely different code. The structure-heavy `reimpl`
|
|
112
247
|
profile is not better on this test.
|
|
113
248
|
|
|
@@ -124,7 +259,7 @@ read/grep, two with this CLI. All distinct groups (82) were then judged blind by
|
|
|
124
259
|
| OneSales, with CLI | 21 | 10 | 48% / 86% | 67% | 9 |
|
|
125
260
|
|
|
126
261
|
Only 18 of 82 groups were found by both, so the approaches are complementary. The agents' reports are capped at 30
|
|
127
|
-
groups, so the raw tool is a better measure of recall: `scan --threshold 0.6` (defaults) finds 38 of the 40 judged
|
|
262
|
+
groups, so the raw tool is a better measure of recall: `scan --threshold 0.6` (the `copies` defaults) finds 38 of the 40 judged
|
|
128
263
|
true duplicates (25/25 MDMApp, 13/15 OneSales) and 23 of the 25 that the LLM-only agents found independently.
|
|
129
264
|
What it still misses: a differently-written picker function (same purpose, different code) and a formatFileSize
|
|
130
265
|
variant with different constants. Fixed after this test: tiny same-name exact copies (`min_tokens` 20 -> 8, near-exact
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
# duplicatecode
|
|
2
2
|
|
|
3
|
-
Static (LLM-free) detection of duplicate / similar code in Python, TypeScript (TSX) and C#,
|
|
3
|
+
Static by default (LLM-free, no network unless `--embed*` is used) detection of duplicate / similar code in Python, TypeScript/JavaScript (TSX/JSX), SQL and C#,
|
|
4
4
|
aimed at catching an LLM re-implementing something that already exists. Input can be a git diff
|
|
5
5
|
checked against existing source.
|
|
6
6
|
|
|
@@ -32,6 +32,141 @@ duplicatecode scan . --pairs --json # machine-readable pairs in
|
|
|
32
32
|
duplicatecode bench --dataset dataset [--file-level] [--mutations] [--negatives <other repo>]
|
|
33
33
|
```
|
|
34
34
|
|
|
35
|
+
## More ways to look
|
|
36
|
+
|
|
37
|
+
```sh
|
|
38
|
+
duplicatecode fragments packages/ # copied blocks inside different functions (4+ identical statements)
|
|
39
|
+
duplicatecode scan . --pairs --explain # say what differs: literals, calls, lines (what to parameterize)
|
|
40
|
+
duplicatecode find "retry with exponential backoff" src/ # does something like this already exist? (needs embeddings)
|
|
41
|
+
```
|
|
42
|
+
|
|
43
|
+
- `fragments` indexes windows of consecutive normalized statements, so a block pasted into another
|
|
44
|
+
function is found even when the surrounding functions differ. On an injection benchmark (renamed
|
|
45
|
+
blocks of real functions pasted into others) it finds 93-97% of blocks of 6+ statements; see
|
|
46
|
+
`eval/REPORT.md`. Tune with `--min-stmts` / `--min-tokens`; constructors and dunders are ignored.
|
|
47
|
+
- `--explain` adds, per pair, the literals and calls only one side has and the source lines with no
|
|
48
|
+
counterpart ("16 of 18/17 statements shared").
|
|
49
|
+
- `find` embeds the query and every unit and ranks by similarity. With MiniLM, 50 task descriptions
|
|
50
|
+
against 168 implementation units rank the right one first 94% of the time; with 1,379 unrelated real
|
|
51
|
+
units mixed in, 82% first and 96% in the top 3 (every query within the top 10).
|
|
52
|
+
- Default thresholds depend on the profile: `copies` keeps its tuned 0.6 (`scan`), 0.4 (`diff`), 0.45
|
|
53
|
+
(`review`); `reimpl` scores live lower, so it defaults to 0.35 / 0.28 / 0.30, which correspond to roughly
|
|
54
|
+
0.1% / 1% false-positive rates on unrelated code in the benchmarks. With `--embed` the `reimpl`
|
|
55
|
+
defaults rise by 0.07 because blended scores of unrelated code rise too.
|
|
56
|
+
|
|
57
|
+
## Embeddings (bring your own key)
|
|
58
|
+
|
|
59
|
+
Two optional signals, both off by default (the detector stays static and offline unless you ask).
|
|
60
|
+
|
|
61
|
+
> **Privacy warning.** `--embed openai`, `--embed cohere`, and `--embed-code` / `--embeddings` pointed at any
|
|
62
|
+
> hosted endpoint (OpenAI, Cohere, Azure, ...) **upload the source text of your units** to that provider
|
|
63
|
+
> (`--embeddings` sends identifier names only). `find` embeds **every unit under the given paths**, so
|
|
64
|
+
> `find ... --embed openai` uploads all of it. The local presets (`minilm`, `qwen3`, `potion`) are meant for a
|
|
65
|
+
> server on loopback (default `127.0.0.1:8099`; `DUPLICATECODE_EMBED_ENDPOINT` overrides it): pointing that
|
|
66
|
+
> variable at a remote host sends text there too. Do not use hosted embeddings on code you may not share.
|
|
67
|
+
|
|
68
|
+
- `--embeddings` embeds identifier *names* (cheap) as an extra name-similarity signal.
|
|
69
|
+
- `--embed <preset>` is the short form (`minilm`, `qwen3`, `potion`, `openai`, `cohere`; `qwen3` is about 5x slower than `minilm` on CPU, so use it with a GPU or prefer `minilm`) and implies
|
|
70
|
+
`--embed-code`. Model ids go into the cache key, so vectors of different models never mix.
|
|
71
|
+
- `--embed-code` embeds the *whole text of every unit* (function, class, file, SQL statement) and blends
|
|
72
|
+
the cosine into the score (`--embed-weight`, default 0.35, others scaled by 1 - weight). It finds
|
|
73
|
+
re-implementations that share no tokens. On function-level LLM re-implementations it lifted Python from
|
|
74
|
+
0.70 to 0.89 on a held-out half; see `eval/REPORT.md` for the numbers and caveats. Vectors are cached per
|
|
75
|
+
unit text in `~/.cache/duplicatecode/embeddings.bin`, so a rescan only embeds what changed.
|
|
76
|
+
|
|
77
|
+
Credentials come from the environment:
|
|
78
|
+
|
|
79
|
+
```sh
|
|
80
|
+
# OpenAI
|
|
81
|
+
export OPENAI_API_KEY=sk-... # optional: OPENAI_EMBEDDING_MODEL (default text-embedding-3-small)
|
|
82
|
+
# Cohere (native /v2/embed)
|
|
83
|
+
export COHERE_API_KEY=... # optional: COHERE_EMBEDDING_MODEL (default embed-v4.0)
|
|
84
|
+
# OpenAI-compatible server (vLLM, Ollama, LiteLLM, ...)
|
|
85
|
+
export OPENAI_BASE_URL=http://localhost:11434/v1 OPENAI_API_KEY=anything OPENAI_EMBEDDING_MODEL=nomic-embed-text
|
|
86
|
+
# Azure AI Foundry (key, or `az login` if no key is set)
|
|
87
|
+
export AZURE_AI_FOUNDRY_ENDPOINT=https://<res>.services.ai.azure.com
|
|
88
|
+
export AZURE_AI_FOUNDRY_API_KEY=... AZURE_AI_FOUNDRY_EMBEDDING_DEPLOYMENT=text-embedding-3-small
|
|
89
|
+
# Azure OpenAI: AZURE_OPENAI_ENDPOINT / AZURE_OPENAI_API_KEY / AZURE_OPENAI_EMBEDDING_DEPLOYMENT
|
|
90
|
+
# Fully local, no key: start the server once (CPU; it loads whichever model a request names)
|
|
91
|
+
# uv run --no-project --python 3.12 eval/embed_server.py
|
|
92
|
+
# duplicatecode scan . --embed minilm # or: qwen3, potion (default endpoint 127.0.0.1:8099)
|
|
93
|
+
# Hosted presets: --embed openai (OPENAI_API_KEY), --embed cohere (COHERE_API_KEY)
|
|
94
|
+
|
|
95
|
+
duplicatecode embed-test fetchUser getUser # check credentials
|
|
96
|
+
duplicatecode scan . --embed-code # whole-unit embeddings
|
|
97
|
+
duplicatecode scan . --embeddings # names only
|
|
98
|
+
```
|
|
99
|
+
|
|
100
|
+
## API
|
|
101
|
+
|
|
102
|
+
### Command line
|
|
103
|
+
|
|
104
|
+
| command | what it does |
|
|
105
|
+
| --- | --- |
|
|
106
|
+
| `scan <paths>` | similar units inside one or more folders; `--pairs`, `--explain`, `--json`, `--fail-on-found` |
|
|
107
|
+
| `diff` | check code added in a git diff (stdin or `--diff`) against the repo (`--repo`) |
|
|
108
|
+
| `review` | ranked candidates for a diff, for a human or an agent to judge |
|
|
109
|
+
| `fragments <paths>` | copied blocks of identical statements inside different functions |
|
|
110
|
+
| `find "<description>" <paths>` | existing units closest to a plain-language description (needs embeddings) |
|
|
111
|
+
| `units <path>` / `show file:10-40` | list extracted units / print a unit's source |
|
|
112
|
+
| `embed-test <names...>` | check the embedding provider and calibrate `--embed-floor` |
|
|
113
|
+
| `eval-groups`, `bench`, `make-mutation-groups` | evaluation (see `eval/README.md`) |
|
|
114
|
+
|
|
115
|
+
Shared options: `--profile copies|reimpl`, `--threshold`, `--min-tokens`, `--min-lines`, `--exclude`, `--skip-tests`,
|
|
116
|
+
`--cross-file`, `--json`. Embedding options (all off by default): `--embed <preset>`, `--embed-code`,
|
|
117
|
+
`--embed-weight`, `--embed-max-chars`, `--embeddings`, `--embed-cache`, `--embed-dims`.
|
|
118
|
+
|
|
119
|
+
```sh
|
|
120
|
+
duplicatecode scan . --profile reimpl --min-name 0 --embed minilm --pairs --explain
|
|
121
|
+
duplicatecode fragments src/ --min-stmts 4 --min-tokens 30 --cross-file --json
|
|
122
|
+
duplicatecode find "retry with exponential backoff" src/ --top 5 --json
|
|
123
|
+
duplicatecode diff --repo . < change.diff --embed openai
|
|
124
|
+
```
|
|
125
|
+
|
|
126
|
+
JSON output:
|
|
127
|
+
|
|
128
|
+
- `find --json`: `[{score, file, name, kind, start_line, end_line}]`
|
|
129
|
+
- `fragments --json`: `[{a, b, statements, tokens}]` with `a`/`b` = `{file, unit, start_line, end_line, coverage}`
|
|
130
|
+
- `scan --pairs --json`: `[{query, candidate, scores}]`; with `--explain`: `[{match, explanation}]`
|
|
131
|
+
- `scan --json` (groups): `[{score, units: [{file, name, kind, start_line, end_line}]}]`
|
|
132
|
+
|
|
133
|
+
`scan --fail-on-found` exits with status 1 when anything is reported, for CI.
|
|
134
|
+
|
|
135
|
+
### Rust library
|
|
136
|
+
|
|
137
|
+
The engine crate `duplicatecode-engine` is what the CLI uses:
|
|
138
|
+
|
|
139
|
+
```rust
|
|
140
|
+
use duplicatecode_engine::embed::{embed_unit_code, EmbedConfig, EmbeddingCache};
|
|
141
|
+
use duplicatecode_engine::explain::explain;
|
|
142
|
+
use duplicatecode_engine::fragments::{find_fragments, FragmentOptions};
|
|
143
|
+
use duplicatecode_engine::index::Weights;
|
|
144
|
+
use duplicatecode_engine::{find_matches, load_units_with, Corpus, MatchOptions};
|
|
145
|
+
|
|
146
|
+
let mut units = load_units_with(Path::new("src/"), &[]);
|
|
147
|
+
|
|
148
|
+
// optional: embed every unit; units without a vector simply skip the embedding term
|
|
149
|
+
let cfg = EmbedConfig::from_env(None).ok_or("no embedding provider configured")?;
|
|
150
|
+
let mut cache = EmbeddingCache::load(&EmbeddingCache::default_path());
|
|
151
|
+
embed_unit_code(&mut units, &cfg, &mut cache, 3000)?;
|
|
152
|
+
|
|
153
|
+
// pairs: queries against a corpus
|
|
154
|
+
let weights = Weights::default().with_embed(0.35); // 0.0 = static only
|
|
155
|
+
let corpus = Corpus::new(units.clone());
|
|
156
|
+
let pairs = find_matches(&units, &corpus, MatchOptions { weights, threshold: 0.42, ..Default::default() });
|
|
157
|
+
|
|
158
|
+
// copied blocks, and what differs inside a pair
|
|
159
|
+
let blocks = find_fragments(&units, FragmentOptions::default());
|
|
160
|
+
println!("{}", explain(&units[0], &units[1]).summary());
|
|
161
|
+
```
|
|
162
|
+
|
|
163
|
+
### Not built yet
|
|
164
|
+
|
|
165
|
+
Planned, not available today: a `.duplicatecode.toml` with an `[embed]` section; `--embed-optional` (warn and
|
|
166
|
+
fall back to the static score when the provider is unreachable; today an unreachable provider is an error);
|
|
167
|
+
`--allow-upload` (required for the hosted presets, which send source text out); a `duplicatecode[embed]` Python
|
|
168
|
+
extra that starts the local server on demand; an `Embedder` trait so providers can plug in without HTTP.
|
|
169
|
+
|
|
35
170
|
## How it works
|
|
36
171
|
|
|
37
172
|
Units (functions, methods, classes, arrow-function components) are extracted with tree-sitter and
|
|
@@ -77,9 +212,9 @@ Self-scans of three internal repositories (Python + TypeScript), 189 pairs judge
|
|
|
77
212
|
before trusting them. A structure-heavy `--profile reimpl` exists for renamed re-implementations
|
|
78
213
|
(use with `--min-name 0`) but has no real-repo precision data yet.
|
|
79
214
|
|
|
80
|
-
### Fresh held-out check (copies profile, threshold 0.6)
|
|
215
|
+
### Fresh held-out check (copies profile, threshold 0.6 = the `copies` default for `scan`)
|
|
81
216
|
|
|
82
|
-
A third, untouched sample of 57 pairs (nothing was tuned on it): 30% true duplicates, 47% incl. partial
|
|
217
|
+
A third, untouched sample of 57 pairs (nothing was tuned on it; the `reimpl` profile has different defaults, 0.35 for `scan`, and is not covered by this sample): 30% true duplicates, 47% incl. partial
|
|
83
218
|
(OneSales 0/24, MDMApp 11/24, CCMT2 6/9 true). The 50%/76% above was optimistic because it was measured on the
|
|
84
219
|
pairs used to choose the weights. Test code is about half of the noise but also holds real copies
|
|
85
220
|
(25% true either way), so it is kept by default; `--skip-tests` drops it.
|
|
@@ -97,7 +232,7 @@ by Haiku and Sonnet without seeing the repo. Fraction where the original is foun
|
|
|
97
232
|
| 0.5 | 45% | 14% | 30% | 1% |
|
|
98
233
|
| 0.6 | 24% | 5% | 14% | 0% |
|
|
99
234
|
|
|
100
|
-
So `diff` (checking new code) defaults to threshold 0.4 while `scan` (existing copies) defaults to 0.6. About half of
|
|
235
|
+
So with the `copies` profile `diff` (checking new code) defaults to threshold 0.4 while `scan` (existing copies) defaults to 0.6; the `reimpl` profile has its own lower defaults (see above). About half of
|
|
101
236
|
independent re-implementations are caught; the rest are genuinely different code. The structure-heavy `reimpl`
|
|
102
237
|
profile is not better on this test.
|
|
103
238
|
|
|
@@ -114,7 +249,7 @@ read/grep, two with this CLI. All distinct groups (82) were then judged blind by
|
|
|
114
249
|
| OneSales, with CLI | 21 | 10 | 48% / 86% | 67% | 9 |
|
|
115
250
|
|
|
116
251
|
Only 18 of 82 groups were found by both, so the approaches are complementary. The agents' reports are capped at 30
|
|
117
|
-
groups, so the raw tool is a better measure of recall: `scan --threshold 0.6` (defaults) finds 38 of the 40 judged
|
|
252
|
+
groups, so the raw tool is a better measure of recall: `scan --threshold 0.6` (the `copies` defaults) finds 38 of the 40 judged
|
|
118
253
|
true duplicates (25/25 MDMApp, 13/15 OneSales) and 23 of the 25 that the LLM-only agents found independently.
|
|
119
254
|
What it still misses: a differently-written picker function (same purpose, different code) and a formatFileSize
|
|
120
255
|
variant with different constants. Fixed after this test: tiny same-name exact copies (`min_tokens` 20 -> 8, near-exact
|
|
@@ -389,6 +389,9 @@ fn fit(samples: &[&Sample]) -> Weights {
|
|
|
389
389
|
let mut w = Weights {
|
|
390
390
|
bias: 0.0,
|
|
391
391
|
w: [0.0; N_FEATURES],
|
|
392
|
+
sql: None,
|
|
393
|
+
file: None,
|
|
394
|
+
renormalize: false,
|
|
392
395
|
name_floor: 0.5,
|
|
393
396
|
};
|
|
394
397
|
for _ in 0..2500 {
|
|
@@ -415,7 +418,7 @@ fn apply(w: &Weights, x: &[f64; N_FEATURES]) -> f64 {
|
|
|
415
418
|
}
|
|
416
419
|
|
|
417
420
|
/// Fraction of positives scoring above the (1 - fpr) quantile of the negatives.
|
|
418
|
-
fn tpr_at_fpr(scores: &[(f64, bool)], fpr: f64) -> f64 {
|
|
421
|
+
pub(crate) fn tpr_at_fpr(scores: &[(f64, bool)], fpr: f64) -> f64 {
|
|
419
422
|
let mut neg: Vec<f64> = scores.iter().filter(|s| !s.1).map(|s| s.0).collect();
|
|
420
423
|
let pos: Vec<f64> = scores.iter().filter(|s| s.1).map(|s| s.0).collect();
|
|
421
424
|
if neg.is_empty() || pos.is_empty() {
|