duplicatecode 0.3.0__tar.gz → 0.3.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (23) hide show
  1. {duplicatecode-0.3.0 → duplicatecode-0.3.2}/Cargo.lock +2 -2
  2. {duplicatecode-0.3.0 → duplicatecode-0.3.2}/Cargo.toml +1 -1
  3. {duplicatecode-0.3.0 → duplicatecode-0.3.2}/PKG-INFO +20 -10
  4. {duplicatecode-0.3.0 → duplicatecode-0.3.2}/README.md +19 -9
  5. {duplicatecode-0.3.0 → duplicatecode-0.3.2}/crates/duplicatecode-cli/src/groups.rs +2 -0
  6. {duplicatecode-0.3.0 → duplicatecode-0.3.2}/crates/duplicatecode-cli/src/main.rs +100 -32
  7. {duplicatecode-0.3.0 → duplicatecode-0.3.2}/crates/duplicatecode-engine/src/embed.rs +412 -83
  8. {duplicatecode-0.3.0 → duplicatecode-0.3.2}/crates/duplicatecode-engine/src/index.rs +190 -26
  9. {duplicatecode-0.3.0 → duplicatecode-0.3.2}/crates/duplicatecode-engine/src/lang.rs +3 -2
  10. {duplicatecode-0.3.0 → duplicatecode-0.3.2}/crates/duplicatecode-engine/src/units.rs +142 -12
  11. {duplicatecode-0.3.0 → duplicatecode-0.3.2}/crates/duplicatecode-cli/Cargo.toml +0 -0
  12. {duplicatecode-0.3.0 → duplicatecode-0.3.2}/crates/duplicatecode-cli/src/bench.rs +0 -0
  13. {duplicatecode-0.3.0 → duplicatecode-0.3.2}/crates/duplicatecode-cli/src/review.rs +0 -0
  14. {duplicatecode-0.3.0 → duplicatecode-0.3.2}/crates/duplicatecode-engine/Cargo.toml +0 -0
  15. {duplicatecode-0.3.0 → duplicatecode-0.3.2}/crates/duplicatecode-engine/src/diff.rs +0 -0
  16. {duplicatecode-0.3.0 → duplicatecode-0.3.2}/crates/duplicatecode-engine/src/explain.rs +0 -0
  17. {duplicatecode-0.3.0 → duplicatecode-0.3.2}/crates/duplicatecode-engine/src/fingerprint.rs +0 -0
  18. {duplicatecode-0.3.0 → duplicatecode-0.3.2}/crates/duplicatecode-engine/src/fragments.rs +0 -0
  19. {duplicatecode-0.3.0 → duplicatecode-0.3.2}/crates/duplicatecode-engine/src/lib.rs +0 -0
  20. {duplicatecode-0.3.0 → duplicatecode-0.3.2}/crates/duplicatecode-engine/src/mutate.rs +0 -0
  21. {duplicatecode-0.3.0 → duplicatecode-0.3.2}/crates/duplicatecode-engine/src/naming.rs +0 -0
  22. {duplicatecode-0.3.0 → duplicatecode-0.3.2}/crates/duplicatecode-engine/src/similarity.rs +0 -0
  23. {duplicatecode-0.3.0 → duplicatecode-0.3.2}/pyproject.toml +0 -0
@@ -198,7 +198,7 @@ dependencies = [
198
198
 
199
199
  [[package]]
200
200
  name = "duplicatecode"
201
- version = "0.3.0"
201
+ version = "0.3.2"
202
202
  dependencies = [
203
203
  "anyhow",
204
204
  "clap",
@@ -212,7 +212,7 @@ dependencies = [
212
212
 
213
213
  [[package]]
214
214
  name = "duplicatecode-engine"
215
- version = "0.3.0"
215
+ version = "0.3.2"
216
216
  dependencies = [
217
217
  "ignore",
218
218
  "serde",
@@ -3,6 +3,6 @@ resolver = "2"
3
3
  members = ["crates/duplicatecode-engine", "crates/duplicatecode-cli"]
4
4
 
5
5
  [workspace.package]
6
- version = "0.3.0"
6
+ version = "0.3.2"
7
7
  edition = "2021"
8
8
  license = "MIT"
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: duplicatecode
3
- Version: 0.3.0
3
+ Version: 0.3.2
4
4
  Classifier: Programming Language :: Rust
5
5
  Classifier: Environment :: Console
6
6
  Summary: Static (LLM-free) detection of duplicate/similar code in Python, TypeScript/JavaScript, SQL and C#
@@ -10,7 +10,7 @@ Description-Content-Type: text/markdown; charset=UTF-8; variant=GFM
10
10
 
11
11
  # duplicatecode
12
12
 
13
- Static (LLM-free) detection of duplicate / similar code in Python, TypeScript/JavaScript (TSX/JSX), SQL and C#,
13
+ Static by default (LLM-free, no network unless `--embed*` is used) detection of duplicate / similar code in Python, TypeScript/JavaScript (TSX/JSX), SQL and C#,
14
14
  aimed at catching an LLM re-implementing something that already exists. Input can be a git diff
15
15
  checked against existing source.
16
16
 
@@ -66,11 +66,20 @@ duplicatecode find "retry with exponential backoff" src/ # does something like
66
66
 
67
67
  ## Embeddings (bring your own key)
68
68
 
69
- Two optional signals, both off by default (the detector stays LLM-free unless you ask):
69
+ Two optional signals, both off by default (the detector stays static and offline unless you ask).
70
+
71
+ > **Privacy warning.** `--embed openai`, `--embed cohere`, and `--embed-code` / `--embeddings` pointed at any
72
+ > hosted endpoint (OpenAI, Cohere, Azure, ...) **upload the source text of your units** to that provider
73
+ > (`--embeddings` sends identifier names only). `find` embeds **every unit under the given paths**, so
74
+ > `find ... --embed openai` uploads all of it. The local presets (`gemma`, `minilm`, `qwen3`, `potion`) are meant for a
75
+ > server on loopback (`gemma`: ollama on `127.0.0.1:11434`, the others `127.0.0.1:8099`; `DUPLICATECODE_EMBED_ENDPOINT` overrides it): pointing that
76
+ > variable at a remote host sends text there too. Do not use hosted embeddings on code you may not share.
70
77
 
71
78
  - `--embeddings` embeds identifier *names* (cheap) as an extra name-similarity signal.
72
- - `--embed <preset>` is the short form (`minilm`, `qwen3`, `potion`, `openai`, `cohere`; `qwen3` is about 5x slower than `minilm` on CPU, so use it with a GPU or prefer `minilm`) and implies
73
- `--embed-code`. Model ids go into the cache key, so vectors of different models never mix.
79
+ - `--embed` is the short form and implies `--embed-code`. A bare `--embed` uses **`gemma`** (embeddinggemma via a local
80
+ [ollama](https://ollama.com): `ollama pull embeddinggemma`), the best model in the codebase benchmark (issue #12: re-implementation
81
+ retrieval R@1 0.94 vs 0.89 for the static detector alone, unchanged when function names are masked). Other presets:
82
+ `--embed=minilm|qwen3|potion|openai|cohere`; MiniLM is smaller but clearly weaker (0.85) and leans on names. Model ids go into the cache key, so vectors of different models never mix.
74
83
  - `--embed-code` embeds the *whole text of every unit* (function, class, file, SQL statement) and blends
75
84
  the cosine into the score (`--embed-weight`, default 0.35, others scaled by 1 - weight). It finds
76
85
  re-implementations that share no tokens. On function-level LLM re-implementations it lifted Python from
@@ -90,7 +99,8 @@ export OPENAI_BASE_URL=http://localhost:11434/v1 OPENAI_API_KEY=anything OPENAI_
90
99
  export AZURE_AI_FOUNDRY_ENDPOINT=https://<res>.services.ai.azure.com
91
100
  export AZURE_AI_FOUNDRY_API_KEY=... AZURE_AI_FOUNDRY_EMBEDDING_DEPLOYMENT=text-embedding-3-small
92
101
  # Azure OpenAI: AZURE_OPENAI_ENDPOINT / AZURE_OPENAI_API_KEY / AZURE_OPENAI_EMBEDDING_DEPLOYMENT
93
- # Fully local, no key: start the server once (CPU; it loads whichever model a request names)
102
+ # Fully local, no key, recommended: ollama pull embeddinggemma && duplicatecode scan . --embed (GPU: ~5 ms per unit)
103
+ # Other local models: start the server once (CPU; it loads whichever model a request names)
94
104
  # uv run --no-project --python 3.12 eval/embed_server.py
95
105
  # duplicatecode scan . --embed minilm # or: qwen3, potion (default endpoint 127.0.0.1:8099)
96
106
  # Hosted presets: --embed openai (OPENAI_API_KEY), --embed cohere (COHERE_API_KEY)
@@ -215,9 +225,9 @@ Self-scans of three internal repositories (Python + TypeScript), 189 pairs judge
215
225
  before trusting them. A structure-heavy `--profile reimpl` exists for renamed re-implementations
216
226
  (use with `--min-name 0`) but has no real-repo precision data yet.
217
227
 
218
- ### Fresh held-out check (copies profile, threshold 0.6)
228
+ ### Fresh held-out check (copies profile, threshold 0.6 = the `copies` default for `scan`)
219
229
 
220
- A third, untouched sample of 57 pairs (nothing was tuned on it): 30% true duplicates, 47% incl. partial
230
+ A third, untouched sample of 57 pairs (nothing was tuned on it; the `reimpl` profile has different defaults, 0.35 for `scan`, and is not covered by this sample): 30% true duplicates, 47% incl. partial
221
231
  (OneSales 0/24, MDMApp 11/24, CCMT2 6/9 true). The 50%/76% above was optimistic because it was measured on the
222
232
  pairs used to choose the weights. Test code is about half of the noise but also holds real copies
223
233
  (25% true either way), so it is kept by default; `--skip-tests` drops it.
@@ -235,7 +245,7 @@ by Haiku and Sonnet without seeing the repo. Fraction where the original is foun
235
245
  | 0.5 | 45% | 14% | 30% | 1% |
236
246
  | 0.6 | 24% | 5% | 14% | 0% |
237
247
 
238
- So `diff` (checking new code) defaults to threshold 0.4 while `scan` (existing copies) defaults to 0.6. About half of
248
+ So with the `copies` profile `diff` (checking new code) defaults to threshold 0.4 while `scan` (existing copies) defaults to 0.6; the `reimpl` profile has its own lower defaults (see above). About half of
239
249
  independent re-implementations are caught; the rest are genuinely different code. The structure-heavy `reimpl`
240
250
  profile is not better on this test.
241
251
 
@@ -252,7 +262,7 @@ read/grep, two with this CLI. All distinct groups (82) were then judged blind by
252
262
  | OneSales, with CLI | 21 | 10 | 48% / 86% | 67% | 9 |
253
263
 
254
264
  Only 18 of 82 groups were found by both, so the approaches are complementary. The agents' reports are capped at 30
255
- groups, so the raw tool is a better measure of recall: `scan --threshold 0.6` (defaults) finds 38 of the 40 judged
265
+ groups, so the raw tool is a better measure of recall: `scan --threshold 0.6` (the `copies` defaults) finds 38 of the 40 judged
256
266
  true duplicates (25/25 MDMApp, 13/15 OneSales) and 23 of the 25 that the LLM-only agents found independently.
257
267
  What it still misses: a differently-written picker function (same purpose, different code) and a formatFileSize
258
268
  variant with different constants. Fixed after this test: tiny same-name exact copies (`min_tokens` 20 -> 8, near-exact
@@ -1,6 +1,6 @@
1
1
  # duplicatecode
2
2
 
3
- Static (LLM-free) detection of duplicate / similar code in Python, TypeScript/JavaScript (TSX/JSX), SQL and C#,
3
+ Static by default (LLM-free, no network unless `--embed*` is used) detection of duplicate / similar code in Python, TypeScript/JavaScript (TSX/JSX), SQL and C#,
4
4
  aimed at catching an LLM re-implementing something that already exists. Input can be a git diff
5
5
  checked against existing source.
6
6
 
@@ -56,11 +56,20 @@ duplicatecode find "retry with exponential backoff" src/ # does something like
56
56
 
57
57
  ## Embeddings (bring your own key)
58
58
 
59
- Two optional signals, both off by default (the detector stays LLM-free unless you ask):
59
+ Two optional signals, both off by default (the detector stays static and offline unless you ask).
60
+
61
+ > **Privacy warning.** `--embed openai`, `--embed cohere`, and `--embed-code` / `--embeddings` pointed at any
62
+ > hosted endpoint (OpenAI, Cohere, Azure, ...) **upload the source text of your units** to that provider
63
+ > (`--embeddings` sends identifier names only). `find` embeds **every unit under the given paths**, so
64
+ > `find ... --embed openai` uploads all of it. The local presets (`gemma`, `minilm`, `qwen3`, `potion`) are meant for a
65
+ > server on loopback (`gemma`: ollama on `127.0.0.1:11434`, the others `127.0.0.1:8099`; `DUPLICATECODE_EMBED_ENDPOINT` overrides it): pointing that
66
+ > variable at a remote host sends text there too. Do not use hosted embeddings on code you may not share.
60
67
 
61
68
  - `--embeddings` embeds identifier *names* (cheap) as an extra name-similarity signal.
62
- - `--embed <preset>` is the short form (`minilm`, `qwen3`, `potion`, `openai`, `cohere`; `qwen3` is about 5x slower than `minilm` on CPU, so use it with a GPU or prefer `minilm`) and implies
63
- `--embed-code`. Model ids go into the cache key, so vectors of different models never mix.
69
+ - `--embed` is the short form and implies `--embed-code`. A bare `--embed` uses **`gemma`** (embeddinggemma via a local
70
+ [ollama](https://ollama.com): `ollama pull embeddinggemma`), the best model in the codebase benchmark (issue #12: re-implementation
71
+ retrieval R@1 0.94 vs 0.89 for the static detector alone, unchanged when function names are masked). Other presets:
72
+ `--embed=minilm|qwen3|potion|openai|cohere`; MiniLM is smaller but clearly weaker (0.85) and leans on names. Model ids go into the cache key, so vectors of different models never mix.
64
73
  - `--embed-code` embeds the *whole text of every unit* (function, class, file, SQL statement) and blends
65
74
  the cosine into the score (`--embed-weight`, default 0.35, others scaled by 1 - weight). It finds
66
75
  re-implementations that share no tokens. On function-level LLM re-implementations it lifted Python from
@@ -80,7 +89,8 @@ export OPENAI_BASE_URL=http://localhost:11434/v1 OPENAI_API_KEY=anything OPENAI_
80
89
  export AZURE_AI_FOUNDRY_ENDPOINT=https://<res>.services.ai.azure.com
81
90
  export AZURE_AI_FOUNDRY_API_KEY=... AZURE_AI_FOUNDRY_EMBEDDING_DEPLOYMENT=text-embedding-3-small
82
91
  # Azure OpenAI: AZURE_OPENAI_ENDPOINT / AZURE_OPENAI_API_KEY / AZURE_OPENAI_EMBEDDING_DEPLOYMENT
83
- # Fully local, no key: start the server once (CPU; it loads whichever model a request names)
92
+ # Fully local, no key, recommended: ollama pull embeddinggemma && duplicatecode scan . --embed (GPU: ~5 ms per unit)
93
+ # Other local models: start the server once (CPU; it loads whichever model a request names)
84
94
  # uv run --no-project --python 3.12 eval/embed_server.py
85
95
  # duplicatecode scan . --embed minilm # or: qwen3, potion (default endpoint 127.0.0.1:8099)
86
96
  # Hosted presets: --embed openai (OPENAI_API_KEY), --embed cohere (COHERE_API_KEY)
@@ -205,9 +215,9 @@ Self-scans of three internal repositories (Python + TypeScript), 189 pairs judge
205
215
  before trusting them. A structure-heavy `--profile reimpl` exists for renamed re-implementations
206
216
  (use with `--min-name 0`) but has no real-repo precision data yet.
207
217
 
208
- ### Fresh held-out check (copies profile, threshold 0.6)
218
+ ### Fresh held-out check (copies profile, threshold 0.6 = the `copies` default for `scan`)
209
219
 
210
- A third, untouched sample of 57 pairs (nothing was tuned on it): 30% true duplicates, 47% incl. partial
220
+ A third, untouched sample of 57 pairs (nothing was tuned on it; the `reimpl` profile has different defaults, 0.35 for `scan`, and is not covered by this sample): 30% true duplicates, 47% incl. partial
211
221
  (OneSales 0/24, MDMApp 11/24, CCMT2 6/9 true). The 50%/76% above was optimistic because it was measured on the
212
222
  pairs used to choose the weights. Test code is about half of the noise but also holds real copies
213
223
  (25% true either way), so it is kept by default; `--skip-tests` drops it.
@@ -225,7 +235,7 @@ by Haiku and Sonnet without seeing the repo. Fraction where the original is foun
225
235
  | 0.5 | 45% | 14% | 30% | 1% |
226
236
  | 0.6 | 24% | 5% | 14% | 0% |
227
237
 
228
- So `diff` (checking new code) defaults to threshold 0.4 while `scan` (existing copies) defaults to 0.6. About half of
238
+ So with the `copies` profile `diff` (checking new code) defaults to threshold 0.4 while `scan` (existing copies) defaults to 0.6; the `reimpl` profile has its own lower defaults (see above). About half of
229
239
  independent re-implementations are caught; the rest are genuinely different code. The structure-heavy `reimpl`
230
240
  profile is not better on this test.
231
241
 
@@ -242,7 +252,7 @@ read/grep, two with this CLI. All distinct groups (82) were then judged blind by
242
252
  | OneSales, with CLI | 21 | 10 | 48% / 86% | 67% | 9 |
243
253
 
244
254
  Only 18 of 82 groups were found by both, so the approaches are complementary. The agents' reports are capped at 30
245
- groups, so the raw tool is a better measure of recall: `scan --threshold 0.6` (defaults) finds 38 of the 40 judged
255
+ groups, so the raw tool is a better measure of recall: `scan --threshold 0.6` (the `copies` defaults) finds 38 of the 40 judged
246
256
  true duplicates (25/25 MDMApp, 13/15 OneSales) and 23 of the 25 that the LLM-only agents found independently.
247
257
  What it still misses: a differently-written picker function (same purpose, different code) and a formatFileSize
248
258
  variant with different constants. Fixed after this test: tiny same-name exact copies (`min_tokens` 20 -> 8, near-exact
@@ -246,6 +246,8 @@ pub fn make_mutation_groups(
246
246
  for (k, p) in files.iter().skip(skip).take(n).enumerate() {
247
247
  let text = std::fs::read_to_string(p)?;
248
248
  let dir = out.join(format!("c{:03}", skip + k));
249
+ // a leftover variant of a previous component would be mislabeled as a clone of this one
250
+ let _ = std::fs::remove_dir_all(&dir);
249
251
  std::fs::create_dir_all(&dir)?;
250
252
  std::fs::write(dir.join("orig.tsx"), &text)?;
251
253
  for m in Mutation::ALL {
@@ -11,11 +11,14 @@ use duplicatecode_engine::{
11
11
  use std::io::Read;
12
12
  use std::path::PathBuf;
13
13
 
14
- /// Named embedding setups for `--embed`. Local presets talk to `eval/embed_server.py` (default
15
- /// http://127.0.0.1:8099/v1, override with DUPLICATECODE_EMBED_ENDPOINT), which loads the named model.
14
+ /// Named embedding setups for `--embed`. `gemma` talks to a local ollama (default
15
+ /// http://127.0.0.1:11434/v1); the other local presets talk to `eval/embed_server.py` (default
16
+ /// http://127.0.0.1:8099/v1). DUPLICATECODE_EMBED_ENDPOINT overrides either, loopback hosts only.
16
17
  #[derive(Clone, Copy, clap::ValueEnum)]
17
18
  enum Preset {
18
- /// all-MiniLM-L6-v2, local: fastest, tied for best quality in the bake-off
19
+ /// embeddinggemma via local ollama (`ollama pull embeddinggemma`): best in the codebase bake-off, the default of a bare `--embed`
20
+ Gemma,
21
+ /// all-MiniLM-L6-v2, local server: small and fast, but below `gemma` on re-implementation retrieval
19
22
  Minilm,
20
23
  /// Qwen3-Embedding-0.6B, local: 8k-token context, slower
21
24
  Qwen3,
@@ -30,19 +33,28 @@ enum Preset {
30
33
  impl Preset {
31
34
  fn config(self, dims: Option<u32>) -> Result<duplicatecode_engine::embed::EmbedConfig> {
32
35
  use duplicatecode_engine::embed::EmbedConfig;
33
- let local = |model: &str| {
36
+ let local = |model: &str, default_endpoint: &str| -> Result<EmbedConfig> {
34
37
  let endpoint = std::env::var("DUPLICATECODE_EMBED_ENDPOINT")
35
- .unwrap_or_else(|_| "http://127.0.0.1:8099/v1".into());
38
+ .unwrap_or_else(|_| default_endpoint.into());
36
39
  // the local server returns full vectors whatever `dimensions` asks for
37
- EmbedConfig::new(endpoint, model.into(), Some("local".into()), None, None)
40
+ let cfg = EmbedConfig::new(endpoint, model.into(), Some("local".into()), None, None);
41
+ anyhow::ensure!(
42
+ cfg.is_loopback(),
43
+ "the local presets only talk to a server on this machine, but DUPLICATECODE_EMBED_ENDPOINT points at {}; \
44
+ unset it, or use --embed openai|cohere (or --embed-code with OPENAI_API_KEY etc.) to send code to a hosted provider on purpose",
45
+ cfg.host()
46
+ );
47
+ Ok(cfg)
38
48
  };
49
+ const SERVER: &str = "http://127.0.0.1:8099/v1";
39
50
  let key = |name: &str| {
40
51
  std::env::var(name).with_context(|| format!("this preset needs {name} to be set"))
41
52
  };
42
53
  Ok(match self {
43
- Preset::Minilm => local("sentence-transformers/all-MiniLM-L6-v2"),
44
- Preset::Qwen3 => local("Qwen/Qwen3-Embedding-0.6B"),
45
- Preset::Potion => local("minishlab/potion-base-8M"),
54
+ Preset::Gemma => local("embeddinggemma", "http://127.0.0.1:11434/v1")?,
55
+ Preset::Minilm => local("sentence-transformers/all-MiniLM-L6-v2", SERVER)?,
56
+ Preset::Qwen3 => local("Qwen/Qwen3-Embedding-0.6B", SERVER)?,
57
+ Preset::Potion => local("minishlab/potion-base-8M", SERVER)?,
46
58
  Preset::Openai => EmbedConfig::new(
47
59
  "https://api.openai.com/v1".into(),
48
60
  "text-embedding-3-small".into(),
@@ -70,7 +82,8 @@ struct EmbedArgs {
70
82
  #[arg(long)]
71
83
  embeddings: bool,
72
84
  /// Embed every unit with a named setup (implies --embed-code); see `--help` for the presets.
73
- #[arg(long, value_enum)]
85
+ /// A bare `--embed` means `gemma`; write `--embed=minilm` etc. for the others.
86
+ #[arg(long, value_enum, num_args = 0..=1, require_equals = true, default_missing_value = "gemma")]
74
87
  embed: Option<Preset>,
75
88
  /// Embed the full text of every unit (function, class, file, statement) and blend the cosine of
76
89
  /// the two vectors into the score. Finds re-implementations that share no tokens.
@@ -94,6 +107,11 @@ struct EmbedArgs {
94
107
  }
95
108
 
96
109
  impl EmbedArgs {
110
+ /// Any embedding feature is on.
111
+ fn enabled(&self) -> bool {
112
+ self.embeddings || self.code_enabled()
113
+ }
114
+
97
115
  /// Whole-unit embeddings are on for `--embed-code` and for any `--embed <preset>`.
98
116
  fn code_enabled(&self) -> bool {
99
117
  self.embed_code || self.embed.is_some()
@@ -113,6 +131,27 @@ impl EmbedArgs {
113
131
  return Ok(());
114
132
  }
115
133
  let cfg = self.config()?;
134
+ // say where the text goes BEFORE anything is sent
135
+ if cfg.is_loopback() {
136
+ eprintln!(
137
+ "embedding {} units via {} (this machine)",
138
+ units.len(),
139
+ cfg.host()
140
+ );
141
+ } else {
142
+ eprintln!(
143
+ "embedding {} units: unit text is sent to {} (model {})",
144
+ units.len(),
145
+ cfg.host(),
146
+ cfg.deployment
147
+ );
148
+ if cfg.is_cleartext_remote() {
149
+ eprintln!(
150
+ "warning: {} is plain http, so your API key and source text travel unencrypted",
151
+ cfg.endpoint
152
+ );
153
+ }
154
+ }
116
155
  let path = self
117
156
  .embed_cache
118
157
  .clone()
@@ -163,6 +202,30 @@ impl EmbedArgs {
163
202
  }
164
203
  }
165
204
 
205
+ /// Units under the options' size/kind filters can never be reported, so there is no reason to embed
206
+ /// (upload) them.
207
+ fn retain_matchable(
208
+ units: &mut Vec<duplicatecode_engine::Unit>,
209
+ min_tokens: usize,
210
+ skip_tests: bool,
211
+ ) {
212
+ units.retain(|u| u.token_count() >= min_tokens && !u.boilerplate && !(skip_tests && u.is_test));
213
+ }
214
+
215
+ /// Units of every root; with several roots the file names carry the root so equal names stay apart.
216
+ fn load_roots(paths: &[PathBuf], exclude: &[String]) -> Vec<duplicatecode_engine::Unit> {
217
+ let mut units = Vec::new();
218
+ for p in paths {
219
+ for mut u in load_units_with(p, exclude) {
220
+ if paths.len() > 1 {
221
+ u.file = format!("{}/{}", p.display(), u.file);
222
+ }
223
+ units.push(u);
224
+ }
225
+ }
226
+ units
227
+ }
228
+
166
229
  #[derive(Clone, Copy, clap::ValueEnum)]
167
230
  enum Profile {
168
231
  Copies,
@@ -397,7 +460,7 @@ enum Cmd {
397
460
  EvalGroups {
398
461
  #[arg(long, default_value = "eval/data/codenet")]
399
462
  root: PathBuf,
400
- /// Print only the final `score=` line.
463
+ /// Print only the summary lines (`dataset_score`, `dataset_threshold`, `score=`).
401
464
  #[arg(long)]
402
465
  quiet: bool,
403
466
  /// Write every pair's feature vector (TSV) for offline analysis.
@@ -548,6 +611,10 @@ fn main() -> Result<()> {
548
611
  }));
549
612
  }
550
613
  let mut corpus_units = load_units(&repo);
614
+ if embed.enabled() {
615
+ retain_matchable(&mut corpus_units, min_tokens, false);
616
+ retain_matchable(&mut queries, min_tokens, false);
617
+ }
551
618
  embed.apply(&mut corpus_units)?;
552
619
  embed.apply(&mut queries)?;
553
620
  let corpus = Corpus::new(corpus_units);
@@ -616,6 +683,9 @@ fn main() -> Result<()> {
616
683
  units.push(u);
617
684
  }
618
685
  }
686
+ if embed.enabled() {
687
+ units.retain(|u| !u.boilerplate && !(skip_tests && u.is_test));
688
+ }
619
689
  embed.apply(&mut units)?;
620
690
  let o = review::ReviewOptions {
621
691
  threshold,
@@ -711,10 +781,7 @@ fn main() -> Result<()> {
711
781
  cross_file,
712
782
  json,
713
783
  } => {
714
- let mut units = Vec::new();
715
- for p in &paths {
716
- units.extend(load_units_with(p, &exclude));
717
- }
784
+ let mut units = load_roots(&paths, &exclude);
718
785
  units.retain(|u| !u.boilerplate && !(skip_tests && u.is_test));
719
786
  let mut found = duplicatecode_engine::fragments::find_fragments(
720
787
  &units,
@@ -760,10 +827,7 @@ fn main() -> Result<()> {
760
827
  json,
761
828
  } => {
762
829
  embed.embed_code = true; // searching by description needs unit embeddings
763
- let mut units = Vec::new();
764
- for p in &paths {
765
- units.extend(load_units_with(p, &exclude));
766
- }
830
+ let mut units = load_roots(&paths, &exclude);
767
831
  units.retain(|u| u.token_count() >= min_tokens && !u.boilerplate);
768
832
  embed.apply(&mut units)?;
769
833
  let cfg = embed.config()?;
@@ -814,14 +878,9 @@ fn main() -> Result<()> {
814
878
  } => {
815
879
  let threshold =
816
880
  threshold.unwrap_or(profile.default_threshold(Command::Scan, embed.code_enabled()));
817
- let mut units = Vec::new();
818
- for p in &paths {
819
- for mut u in load_units_with(p, &exclude) {
820
- if paths.len() > 1 {
821
- u.file = format!("{}/{}", p.display(), u.file);
822
- }
823
- units.push(u);
824
- }
881
+ let mut units = load_roots(&paths, &exclude);
882
+ if embed.enabled() {
883
+ retain_matchable(&mut units, min_tokens, skip_tests);
825
884
  }
826
885
  embed.apply(&mut units)?;
827
886
  let corpus = Corpus::new(units.clone());
@@ -851,14 +910,23 @@ fn main() -> Result<()> {
851
910
  .collect();
852
911
  pairs.sort_by(|a, b| b.scores.combined.total_cmp(&a.scores.combined));
853
912
  let groups = group_pairs(&pairs);
854
- let by_place: std::collections::HashMap<String, &duplicatecode_engine::Unit> = units
913
+ // keyed by the full place: units that start on the same line (minified code, one-line
914
+ // classes) must not be mixed up
915
+ let place = |r: &duplicatecode_engine::index::UnitRef| {
916
+ (r.file.clone(), r.start_line, r.end_line, r.name.clone())
917
+ };
918
+ let by_place: std::collections::HashMap<_, &duplicatecode_engine::Unit> = units
855
919
  .iter()
856
- .map(|u| (format!("{}:{}", u.file, u.start_line), u))
920
+ .map(|u| {
921
+ (
922
+ (u.file.clone(), u.start_line, u.end_line, u.name.clone()),
923
+ u,
924
+ )
925
+ })
857
926
  .collect();
858
927
  let explanation = |m: &duplicatecode_engine::Match| {
859
- let a = by_place.get(&format!("{}:{}", m.query.file, m.query.start_line))?;
860
- let b =
861
- by_place.get(&format!("{}:{}", m.candidate.file, m.candidate.start_line))?;
928
+ let a = by_place.get(&place(&m.query))?;
929
+ let b = by_place.get(&place(&m.candidate))?;
862
930
  Some(duplicatecode_engine::explain::explain(a, b))
863
931
  };
864
932
  if pairs_out {