duplicatecode 0.3.0__tar.gz → 0.3.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {duplicatecode-0.3.0 → duplicatecode-0.3.1}/Cargo.lock +2 -2
- {duplicatecode-0.3.0 → duplicatecode-0.3.1}/Cargo.toml +1 -1
- {duplicatecode-0.3.0 → duplicatecode-0.3.1}/PKG-INFO +14 -7
- {duplicatecode-0.3.0 → duplicatecode-0.3.1}/README.md +13 -6
- {duplicatecode-0.3.0 → duplicatecode-0.3.1}/crates/duplicatecode-cli/src/groups.rs +2 -0
- {duplicatecode-0.3.0 → duplicatecode-0.3.1}/crates/duplicatecode-cli/src/main.rs +89 -27
- {duplicatecode-0.3.0 → duplicatecode-0.3.1}/crates/duplicatecode-engine/src/embed.rs +412 -83
- {duplicatecode-0.3.0 → duplicatecode-0.3.1}/crates/duplicatecode-engine/src/index.rs +190 -26
- {duplicatecode-0.3.0 → duplicatecode-0.3.1}/crates/duplicatecode-engine/src/lang.rs +3 -2
- {duplicatecode-0.3.0 → duplicatecode-0.3.1}/crates/duplicatecode-engine/src/units.rs +142 -12
- {duplicatecode-0.3.0 → duplicatecode-0.3.1}/crates/duplicatecode-cli/Cargo.toml +0 -0
- {duplicatecode-0.3.0 → duplicatecode-0.3.1}/crates/duplicatecode-cli/src/bench.rs +0 -0
- {duplicatecode-0.3.0 → duplicatecode-0.3.1}/crates/duplicatecode-cli/src/review.rs +0 -0
- {duplicatecode-0.3.0 → duplicatecode-0.3.1}/crates/duplicatecode-engine/Cargo.toml +0 -0
- {duplicatecode-0.3.0 → duplicatecode-0.3.1}/crates/duplicatecode-engine/src/diff.rs +0 -0
- {duplicatecode-0.3.0 → duplicatecode-0.3.1}/crates/duplicatecode-engine/src/explain.rs +0 -0
- {duplicatecode-0.3.0 → duplicatecode-0.3.1}/crates/duplicatecode-engine/src/fingerprint.rs +0 -0
- {duplicatecode-0.3.0 → duplicatecode-0.3.1}/crates/duplicatecode-engine/src/fragments.rs +0 -0
- {duplicatecode-0.3.0 → duplicatecode-0.3.1}/crates/duplicatecode-engine/src/lib.rs +0 -0
- {duplicatecode-0.3.0 → duplicatecode-0.3.1}/crates/duplicatecode-engine/src/mutate.rs +0 -0
- {duplicatecode-0.3.0 → duplicatecode-0.3.1}/crates/duplicatecode-engine/src/naming.rs +0 -0
- {duplicatecode-0.3.0 → duplicatecode-0.3.1}/crates/duplicatecode-engine/src/similarity.rs +0 -0
- {duplicatecode-0.3.0 → duplicatecode-0.3.1}/pyproject.toml +0 -0
|
@@ -198,7 +198,7 @@ dependencies = [
|
|
|
198
198
|
|
|
199
199
|
[[package]]
|
|
200
200
|
name = "duplicatecode"
|
|
201
|
-
version = "0.3.
|
|
201
|
+
version = "0.3.1"
|
|
202
202
|
dependencies = [
|
|
203
203
|
"anyhow",
|
|
204
204
|
"clap",
|
|
@@ -212,7 +212,7 @@ dependencies = [
|
|
|
212
212
|
|
|
213
213
|
[[package]]
|
|
214
214
|
name = "duplicatecode-engine"
|
|
215
|
-
version = "0.3.
|
|
215
|
+
version = "0.3.1"
|
|
216
216
|
dependencies = [
|
|
217
217
|
"ignore",
|
|
218
218
|
"serde",
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: duplicatecode
|
|
3
|
-
Version: 0.3.
|
|
3
|
+
Version: 0.3.1
|
|
4
4
|
Classifier: Programming Language :: Rust
|
|
5
5
|
Classifier: Environment :: Console
|
|
6
6
|
Summary: Static (LLM-free) detection of duplicate/similar code in Python, TypeScript/JavaScript, SQL and C#
|
|
@@ -10,7 +10,7 @@ Description-Content-Type: text/markdown; charset=UTF-8; variant=GFM
|
|
|
10
10
|
|
|
11
11
|
# duplicatecode
|
|
12
12
|
|
|
13
|
-
Static (LLM-free) detection of duplicate / similar code in Python, TypeScript/JavaScript (TSX/JSX), SQL and C#,
|
|
13
|
+
Static by default (LLM-free, no network unless `--embed*` is used) detection of duplicate / similar code in Python, TypeScript/JavaScript (TSX/JSX), SQL and C#,
|
|
14
14
|
aimed at catching an LLM re-implementing something that already exists. Input can be a git diff
|
|
15
15
|
checked against existing source.
|
|
16
16
|
|
|
@@ -66,7 +66,14 @@ duplicatecode find "retry with exponential backoff" src/ # does something like
|
|
|
66
66
|
|
|
67
67
|
## Embeddings (bring your own key)
|
|
68
68
|
|
|
69
|
-
Two optional signals, both off by default (the detector stays
|
|
69
|
+
Two optional signals, both off by default (the detector stays static and offline unless you ask).
|
|
70
|
+
|
|
71
|
+
> **Privacy warning.** `--embed openai`, `--embed cohere`, and `--embed-code` / `--embeddings` pointed at any
|
|
72
|
+
> hosted endpoint (OpenAI, Cohere, Azure, ...) **upload the source text of your units** to that provider
|
|
73
|
+
> (`--embeddings` sends identifier names only). `find` embeds **every unit under the given paths**, so
|
|
74
|
+
> `find ... --embed openai` uploads all of it. The local presets (`minilm`, `qwen3`, `potion`) are meant for a
|
|
75
|
+
> server on loopback (default `127.0.0.1:8099`; `DUPLICATECODE_EMBED_ENDPOINT` overrides it): pointing that
|
|
76
|
+
> variable at a remote host sends text there too. Do not use hosted embeddings on code you may not share.
|
|
70
77
|
|
|
71
78
|
- `--embeddings` embeds identifier *names* (cheap) as an extra name-similarity signal.
|
|
72
79
|
- `--embed <preset>` is the short form (`minilm`, `qwen3`, `potion`, `openai`, `cohere`; `qwen3` is about 5x slower than `minilm` on CPU, so use it with a GPU or prefer `minilm`) and implies
|
|
@@ -215,9 +222,9 @@ Self-scans of three internal repositories (Python + TypeScript), 189 pairs judge
|
|
|
215
222
|
before trusting them. A structure-heavy `--profile reimpl` exists for renamed re-implementations
|
|
216
223
|
(use with `--min-name 0`) but has no real-repo precision data yet.
|
|
217
224
|
|
|
218
|
-
### Fresh held-out check (copies profile, threshold 0.6)
|
|
225
|
+
### Fresh held-out check (copies profile, threshold 0.6 = the `copies` default for `scan`)
|
|
219
226
|
|
|
220
|
-
A third, untouched sample of 57 pairs (nothing was tuned on it): 30% true duplicates, 47% incl. partial
|
|
227
|
+
A third, untouched sample of 57 pairs (nothing was tuned on it; the `reimpl` profile has different defaults, 0.35 for `scan`, and is not covered by this sample): 30% true duplicates, 47% incl. partial
|
|
221
228
|
(OneSales 0/24, MDMApp 11/24, CCMT2 6/9 true). The 50%/76% above was optimistic because it was measured on the
|
|
222
229
|
pairs used to choose the weights. Test code is about half of the noise but also holds real copies
|
|
223
230
|
(25% true either way), so it is kept by default; `--skip-tests` drops it.
|
|
@@ -235,7 +242,7 @@ by Haiku and Sonnet without seeing the repo. Fraction where the original is foun
|
|
|
235
242
|
| 0.5 | 45% | 14% | 30% | 1% |
|
|
236
243
|
| 0.6 | 24% | 5% | 14% | 0% |
|
|
237
244
|
|
|
238
|
-
So `diff` (checking new code) defaults to threshold 0.4 while `scan` (existing copies) defaults to 0.6. About half of
|
|
245
|
+
So with the `copies` profile `diff` (checking new code) defaults to threshold 0.4 while `scan` (existing copies) defaults to 0.6; the `reimpl` profile has its own lower defaults (see above). About half of
|
|
239
246
|
independent re-implementations are caught; the rest are genuinely different code. The structure-heavy `reimpl`
|
|
240
247
|
profile is not better on this test.
|
|
241
248
|
|
|
@@ -252,7 +259,7 @@ read/grep, two with this CLI. All distinct groups (82) were then judged blind by
|
|
|
252
259
|
| OneSales, with CLI | 21 | 10 | 48% / 86% | 67% | 9 |
|
|
253
260
|
|
|
254
261
|
Only 18 of 82 groups were found by both, so the approaches are complementary. The agents' reports are capped at 30
|
|
255
|
-
groups, so the raw tool is a better measure of recall: `scan --threshold 0.6` (defaults) finds 38 of the 40 judged
|
|
262
|
+
groups, so the raw tool is a better measure of recall: `scan --threshold 0.6` (the `copies` defaults) finds 38 of the 40 judged
|
|
256
263
|
true duplicates (25/25 MDMApp, 13/15 OneSales) and 23 of the 25 that the LLM-only agents found independently.
|
|
257
264
|
What it still misses: a differently-written picker function (same purpose, different code) and a formatFileSize
|
|
258
265
|
variant with different constants. Fixed after this test: tiny same-name exact copies (`min_tokens` 20 -> 8, near-exact
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
# duplicatecode
|
|
2
2
|
|
|
3
|
-
Static (LLM-free) detection of duplicate / similar code in Python, TypeScript/JavaScript (TSX/JSX), SQL and C#,
|
|
3
|
+
Static by default (LLM-free, no network unless `--embed*` is used) detection of duplicate / similar code in Python, TypeScript/JavaScript (TSX/JSX), SQL and C#,
|
|
4
4
|
aimed at catching an LLM re-implementing something that already exists. Input can be a git diff
|
|
5
5
|
checked against existing source.
|
|
6
6
|
|
|
@@ -56,7 +56,14 @@ duplicatecode find "retry with exponential backoff" src/ # does something like
|
|
|
56
56
|
|
|
57
57
|
## Embeddings (bring your own key)
|
|
58
58
|
|
|
59
|
-
Two optional signals, both off by default (the detector stays
|
|
59
|
+
Two optional signals, both off by default (the detector stays static and offline unless you ask).
|
|
60
|
+
|
|
61
|
+
> **Privacy warning.** `--embed openai`, `--embed cohere`, and `--embed-code` / `--embeddings` pointed at any
|
|
62
|
+
> hosted endpoint (OpenAI, Cohere, Azure, ...) **upload the source text of your units** to that provider
|
|
63
|
+
> (`--embeddings` sends identifier names only). `find` embeds **every unit under the given paths**, so
|
|
64
|
+
> `find ... --embed openai` uploads all of it. The local presets (`minilm`, `qwen3`, `potion`) are meant for a
|
|
65
|
+
> server on loopback (default `127.0.0.1:8099`; `DUPLICATECODE_EMBED_ENDPOINT` overrides it): pointing that
|
|
66
|
+
> variable at a remote host sends text there too. Do not use hosted embeddings on code you may not share.
|
|
60
67
|
|
|
61
68
|
- `--embeddings` embeds identifier *names* (cheap) as an extra name-similarity signal.
|
|
62
69
|
- `--embed <preset>` is the short form (`minilm`, `qwen3`, `potion`, `openai`, `cohere`; `qwen3` is about 5x slower than `minilm` on CPU, so use it with a GPU or prefer `minilm`) and implies
|
|
@@ -205,9 +212,9 @@ Self-scans of three internal repositories (Python + TypeScript), 189 pairs judge
|
|
|
205
212
|
before trusting them. A structure-heavy `--profile reimpl` exists for renamed re-implementations
|
|
206
213
|
(use with `--min-name 0`) but has no real-repo precision data yet.
|
|
207
214
|
|
|
208
|
-
### Fresh held-out check (copies profile, threshold 0.6)
|
|
215
|
+
### Fresh held-out check (copies profile, threshold 0.6 = the `copies` default for `scan`)
|
|
209
216
|
|
|
210
|
-
A third, untouched sample of 57 pairs (nothing was tuned on it): 30% true duplicates, 47% incl. partial
|
|
217
|
+
A third, untouched sample of 57 pairs (nothing was tuned on it; the `reimpl` profile has different defaults, 0.35 for `scan`, and is not covered by this sample): 30% true duplicates, 47% incl. partial
|
|
211
218
|
(OneSales 0/24, MDMApp 11/24, CCMT2 6/9 true). The 50%/76% above was optimistic because it was measured on the
|
|
212
219
|
pairs used to choose the weights. Test code is about half of the noise but also holds real copies
|
|
213
220
|
(25% true either way), so it is kept by default; `--skip-tests` drops it.
|
|
@@ -225,7 +232,7 @@ by Haiku and Sonnet without seeing the repo. Fraction where the original is foun
|
|
|
225
232
|
| 0.5 | 45% | 14% | 30% | 1% |
|
|
226
233
|
| 0.6 | 24% | 5% | 14% | 0% |
|
|
227
234
|
|
|
228
|
-
So `diff` (checking new code) defaults to threshold 0.4 while `scan` (existing copies) defaults to 0.6. About half of
|
|
235
|
+
So with the `copies` profile `diff` (checking new code) defaults to threshold 0.4 while `scan` (existing copies) defaults to 0.6; the `reimpl` profile has its own lower defaults (see above). About half of
|
|
229
236
|
independent re-implementations are caught; the rest are genuinely different code. The structure-heavy `reimpl`
|
|
230
237
|
profile is not better on this test.
|
|
231
238
|
|
|
@@ -242,7 +249,7 @@ read/grep, two with this CLI. All distinct groups (82) were then judged blind by
|
|
|
242
249
|
| OneSales, with CLI | 21 | 10 | 48% / 86% | 67% | 9 |
|
|
243
250
|
|
|
244
251
|
Only 18 of 82 groups were found by both, so the approaches are complementary. The agents' reports are capped at 30
|
|
245
|
-
groups, so the raw tool is a better measure of recall: `scan --threshold 0.6` (defaults) finds 38 of the 40 judged
|
|
252
|
+
groups, so the raw tool is a better measure of recall: `scan --threshold 0.6` (the `copies` defaults) finds 38 of the 40 judged
|
|
246
253
|
true duplicates (25/25 MDMApp, 13/15 OneSales) and 23 of the 25 that the LLM-only agents found independently.
|
|
247
254
|
What it still misses: a differently-written picker function (same purpose, different code) and a formatFileSize
|
|
248
255
|
variant with different constants. Fixed after this test: tiny same-name exact copies (`min_tokens` 20 -> 8, near-exact
|
|
@@ -246,6 +246,8 @@ pub fn make_mutation_groups(
|
|
|
246
246
|
for (k, p) in files.iter().skip(skip).take(n).enumerate() {
|
|
247
247
|
let text = std::fs::read_to_string(p)?;
|
|
248
248
|
let dir = out.join(format!("c{:03}", skip + k));
|
|
249
|
+
// a leftover variant of a previous component would be mislabeled as a clone of this one
|
|
250
|
+
let _ = std::fs::remove_dir_all(&dir);
|
|
249
251
|
std::fs::create_dir_all(&dir)?;
|
|
250
252
|
std::fs::write(dir.join("orig.tsx"), &text)?;
|
|
251
253
|
for m in Mutation::ALL {
|
|
@@ -30,19 +30,26 @@ enum Preset {
|
|
|
30
30
|
impl Preset {
|
|
31
31
|
fn config(self, dims: Option<u32>) -> Result<duplicatecode_engine::embed::EmbedConfig> {
|
|
32
32
|
use duplicatecode_engine::embed::EmbedConfig;
|
|
33
|
-
let local = |model: &str| {
|
|
33
|
+
let local = |model: &str| -> Result<EmbedConfig> {
|
|
34
34
|
let endpoint = std::env::var("DUPLICATECODE_EMBED_ENDPOINT")
|
|
35
35
|
.unwrap_or_else(|_| "http://127.0.0.1:8099/v1".into());
|
|
36
36
|
// the local server returns full vectors whatever `dimensions` asks for
|
|
37
|
-
EmbedConfig::new(endpoint, model.into(), Some("local".into()), None, None)
|
|
37
|
+
let cfg = EmbedConfig::new(endpoint, model.into(), Some("local".into()), None, None);
|
|
38
|
+
anyhow::ensure!(
|
|
39
|
+
cfg.is_loopback(),
|
|
40
|
+
"the local presets only talk to a server on this machine, but DUPLICATECODE_EMBED_ENDPOINT points at {}; \
|
|
41
|
+
unset it, or use --embed openai|cohere (or --embed-code with OPENAI_API_KEY etc.) to send code to a hosted provider on purpose",
|
|
42
|
+
cfg.host()
|
|
43
|
+
);
|
|
44
|
+
Ok(cfg)
|
|
38
45
|
};
|
|
39
46
|
let key = |name: &str| {
|
|
40
47
|
std::env::var(name).with_context(|| format!("this preset needs {name} to be set"))
|
|
41
48
|
};
|
|
42
49
|
Ok(match self {
|
|
43
|
-
Preset::Minilm => local("sentence-transformers/all-MiniLM-L6-v2")
|
|
44
|
-
Preset::Qwen3 => local("Qwen/Qwen3-Embedding-0.6B")
|
|
45
|
-
Preset::Potion => local("minishlab/potion-base-8M")
|
|
50
|
+
Preset::Minilm => local("sentence-transformers/all-MiniLM-L6-v2")?,
|
|
51
|
+
Preset::Qwen3 => local("Qwen/Qwen3-Embedding-0.6B")?,
|
|
52
|
+
Preset::Potion => local("minishlab/potion-base-8M")?,
|
|
46
53
|
Preset::Openai => EmbedConfig::new(
|
|
47
54
|
"https://api.openai.com/v1".into(),
|
|
48
55
|
"text-embedding-3-small".into(),
|
|
@@ -94,6 +101,11 @@ struct EmbedArgs {
|
|
|
94
101
|
}
|
|
95
102
|
|
|
96
103
|
impl EmbedArgs {
|
|
104
|
+
/// Any embedding feature is on.
|
|
105
|
+
fn enabled(&self) -> bool {
|
|
106
|
+
self.embeddings || self.code_enabled()
|
|
107
|
+
}
|
|
108
|
+
|
|
97
109
|
/// Whole-unit embeddings are on for `--embed-code` and for any `--embed <preset>`.
|
|
98
110
|
fn code_enabled(&self) -> bool {
|
|
99
111
|
self.embed_code || self.embed.is_some()
|
|
@@ -113,6 +125,27 @@ impl EmbedArgs {
|
|
|
113
125
|
return Ok(());
|
|
114
126
|
}
|
|
115
127
|
let cfg = self.config()?;
|
|
128
|
+
// say where the text goes BEFORE anything is sent
|
|
129
|
+
if cfg.is_loopback() {
|
|
130
|
+
eprintln!(
|
|
131
|
+
"embedding {} units via {} (this machine)",
|
|
132
|
+
units.len(),
|
|
133
|
+
cfg.host()
|
|
134
|
+
);
|
|
135
|
+
} else {
|
|
136
|
+
eprintln!(
|
|
137
|
+
"embedding {} units: unit text is sent to {} (model {})",
|
|
138
|
+
units.len(),
|
|
139
|
+
cfg.host(),
|
|
140
|
+
cfg.deployment
|
|
141
|
+
);
|
|
142
|
+
if cfg.is_cleartext_remote() {
|
|
143
|
+
eprintln!(
|
|
144
|
+
"warning: {} is plain http, so your API key and source text travel unencrypted",
|
|
145
|
+
cfg.endpoint
|
|
146
|
+
);
|
|
147
|
+
}
|
|
148
|
+
}
|
|
116
149
|
let path = self
|
|
117
150
|
.embed_cache
|
|
118
151
|
.clone()
|
|
@@ -163,6 +196,30 @@ impl EmbedArgs {
|
|
|
163
196
|
}
|
|
164
197
|
}
|
|
165
198
|
|
|
199
|
+
/// Units under the options' size/kind filters can never be reported, so there is no reason to embed
|
|
200
|
+
/// (upload) them.
|
|
201
|
+
fn retain_matchable(
|
|
202
|
+
units: &mut Vec<duplicatecode_engine::Unit>,
|
|
203
|
+
min_tokens: usize,
|
|
204
|
+
skip_tests: bool,
|
|
205
|
+
) {
|
|
206
|
+
units.retain(|u| u.token_count() >= min_tokens && !u.boilerplate && !(skip_tests && u.is_test));
|
|
207
|
+
}
|
|
208
|
+
|
|
209
|
+
/// Units of every root; with several roots the file names carry the root so equal names stay apart.
|
|
210
|
+
fn load_roots(paths: &[PathBuf], exclude: &[String]) -> Vec<duplicatecode_engine::Unit> {
|
|
211
|
+
let mut units = Vec::new();
|
|
212
|
+
for p in paths {
|
|
213
|
+
for mut u in load_units_with(p, exclude) {
|
|
214
|
+
if paths.len() > 1 {
|
|
215
|
+
u.file = format!("{}/{}", p.display(), u.file);
|
|
216
|
+
}
|
|
217
|
+
units.push(u);
|
|
218
|
+
}
|
|
219
|
+
}
|
|
220
|
+
units
|
|
221
|
+
}
|
|
222
|
+
|
|
166
223
|
#[derive(Clone, Copy, clap::ValueEnum)]
|
|
167
224
|
enum Profile {
|
|
168
225
|
Copies,
|
|
@@ -397,7 +454,7 @@ enum Cmd {
|
|
|
397
454
|
EvalGroups {
|
|
398
455
|
#[arg(long, default_value = "eval/data/codenet")]
|
|
399
456
|
root: PathBuf,
|
|
400
|
-
/// Print only the
|
|
457
|
+
/// Print only the summary lines (`dataset_score`, `dataset_threshold`, `score=`).
|
|
401
458
|
#[arg(long)]
|
|
402
459
|
quiet: bool,
|
|
403
460
|
/// Write every pair's feature vector (TSV) for offline analysis.
|
|
@@ -548,6 +605,10 @@ fn main() -> Result<()> {
|
|
|
548
605
|
}));
|
|
549
606
|
}
|
|
550
607
|
let mut corpus_units = load_units(&repo);
|
|
608
|
+
if embed.enabled() {
|
|
609
|
+
retain_matchable(&mut corpus_units, min_tokens, false);
|
|
610
|
+
retain_matchable(&mut queries, min_tokens, false);
|
|
611
|
+
}
|
|
551
612
|
embed.apply(&mut corpus_units)?;
|
|
552
613
|
embed.apply(&mut queries)?;
|
|
553
614
|
let corpus = Corpus::new(corpus_units);
|
|
@@ -616,6 +677,9 @@ fn main() -> Result<()> {
|
|
|
616
677
|
units.push(u);
|
|
617
678
|
}
|
|
618
679
|
}
|
|
680
|
+
if embed.enabled() {
|
|
681
|
+
units.retain(|u| !u.boilerplate && !(skip_tests && u.is_test));
|
|
682
|
+
}
|
|
619
683
|
embed.apply(&mut units)?;
|
|
620
684
|
let o = review::ReviewOptions {
|
|
621
685
|
threshold,
|
|
@@ -711,10 +775,7 @@ fn main() -> Result<()> {
|
|
|
711
775
|
cross_file,
|
|
712
776
|
json,
|
|
713
777
|
} => {
|
|
714
|
-
let mut units =
|
|
715
|
-
for p in &paths {
|
|
716
|
-
units.extend(load_units_with(p, &exclude));
|
|
717
|
-
}
|
|
778
|
+
let mut units = load_roots(&paths, &exclude);
|
|
718
779
|
units.retain(|u| !u.boilerplate && !(skip_tests && u.is_test));
|
|
719
780
|
let mut found = duplicatecode_engine::fragments::find_fragments(
|
|
720
781
|
&units,
|
|
@@ -760,10 +821,7 @@ fn main() -> Result<()> {
|
|
|
760
821
|
json,
|
|
761
822
|
} => {
|
|
762
823
|
embed.embed_code = true; // searching by description needs unit embeddings
|
|
763
|
-
let mut units =
|
|
764
|
-
for p in &paths {
|
|
765
|
-
units.extend(load_units_with(p, &exclude));
|
|
766
|
-
}
|
|
824
|
+
let mut units = load_roots(&paths, &exclude);
|
|
767
825
|
units.retain(|u| u.token_count() >= min_tokens && !u.boilerplate);
|
|
768
826
|
embed.apply(&mut units)?;
|
|
769
827
|
let cfg = embed.config()?;
|
|
@@ -814,14 +872,9 @@ fn main() -> Result<()> {
|
|
|
814
872
|
} => {
|
|
815
873
|
let threshold =
|
|
816
874
|
threshold.unwrap_or(profile.default_threshold(Command::Scan, embed.code_enabled()));
|
|
817
|
-
let mut units =
|
|
818
|
-
|
|
819
|
-
|
|
820
|
-
if paths.len() > 1 {
|
|
821
|
-
u.file = format!("{}/{}", p.display(), u.file);
|
|
822
|
-
}
|
|
823
|
-
units.push(u);
|
|
824
|
-
}
|
|
875
|
+
let mut units = load_roots(&paths, &exclude);
|
|
876
|
+
if embed.enabled() {
|
|
877
|
+
retain_matchable(&mut units, min_tokens, skip_tests);
|
|
825
878
|
}
|
|
826
879
|
embed.apply(&mut units)?;
|
|
827
880
|
let corpus = Corpus::new(units.clone());
|
|
@@ -851,14 +904,23 @@ fn main() -> Result<()> {
|
|
|
851
904
|
.collect();
|
|
852
905
|
pairs.sort_by(|a, b| b.scores.combined.total_cmp(&a.scores.combined));
|
|
853
906
|
let groups = group_pairs(&pairs);
|
|
854
|
-
|
|
907
|
+
// keyed by the full place: units that start on the same line (minified code, one-line
|
|
908
|
+
// classes) must not be mixed up
|
|
909
|
+
let place = |r: &duplicatecode_engine::index::UnitRef| {
|
|
910
|
+
(r.file.clone(), r.start_line, r.end_line, r.name.clone())
|
|
911
|
+
};
|
|
912
|
+
let by_place: std::collections::HashMap<_, &duplicatecode_engine::Unit> = units
|
|
855
913
|
.iter()
|
|
856
|
-
.map(|u|
|
|
914
|
+
.map(|u| {
|
|
915
|
+
(
|
|
916
|
+
(u.file.clone(), u.start_line, u.end_line, u.name.clone()),
|
|
917
|
+
u,
|
|
918
|
+
)
|
|
919
|
+
})
|
|
857
920
|
.collect();
|
|
858
921
|
let explanation = |m: &duplicatecode_engine::Match| {
|
|
859
|
-
let a = by_place.get(&
|
|
860
|
-
let b =
|
|
861
|
-
by_place.get(&format!("{}:{}", m.candidate.file, m.candidate.start_line))?;
|
|
922
|
+
let a = by_place.get(&place(&m.query))?;
|
|
923
|
+
let b = by_place.get(&place(&m.candidate))?;
|
|
862
924
|
Some(duplicatecode_engine::explain::explain(a, b))
|
|
863
925
|
};
|
|
864
926
|
if pairs_out {
|