duplicatecode 0.3.0__tar.gz → 0.3.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (23) hide show
  1. {duplicatecode-0.3.0 → duplicatecode-0.3.1}/Cargo.lock +2 -2
  2. {duplicatecode-0.3.0 → duplicatecode-0.3.1}/Cargo.toml +1 -1
  3. {duplicatecode-0.3.0 → duplicatecode-0.3.1}/PKG-INFO +14 -7
  4. {duplicatecode-0.3.0 → duplicatecode-0.3.1}/README.md +13 -6
  5. {duplicatecode-0.3.0 → duplicatecode-0.3.1}/crates/duplicatecode-cli/src/groups.rs +2 -0
  6. {duplicatecode-0.3.0 → duplicatecode-0.3.1}/crates/duplicatecode-cli/src/main.rs +89 -27
  7. {duplicatecode-0.3.0 → duplicatecode-0.3.1}/crates/duplicatecode-engine/src/embed.rs +412 -83
  8. {duplicatecode-0.3.0 → duplicatecode-0.3.1}/crates/duplicatecode-engine/src/index.rs +190 -26
  9. {duplicatecode-0.3.0 → duplicatecode-0.3.1}/crates/duplicatecode-engine/src/lang.rs +3 -2
  10. {duplicatecode-0.3.0 → duplicatecode-0.3.1}/crates/duplicatecode-engine/src/units.rs +142 -12
  11. {duplicatecode-0.3.0 → duplicatecode-0.3.1}/crates/duplicatecode-cli/Cargo.toml +0 -0
  12. {duplicatecode-0.3.0 → duplicatecode-0.3.1}/crates/duplicatecode-cli/src/bench.rs +0 -0
  13. {duplicatecode-0.3.0 → duplicatecode-0.3.1}/crates/duplicatecode-cli/src/review.rs +0 -0
  14. {duplicatecode-0.3.0 → duplicatecode-0.3.1}/crates/duplicatecode-engine/Cargo.toml +0 -0
  15. {duplicatecode-0.3.0 → duplicatecode-0.3.1}/crates/duplicatecode-engine/src/diff.rs +0 -0
  16. {duplicatecode-0.3.0 → duplicatecode-0.3.1}/crates/duplicatecode-engine/src/explain.rs +0 -0
  17. {duplicatecode-0.3.0 → duplicatecode-0.3.1}/crates/duplicatecode-engine/src/fingerprint.rs +0 -0
  18. {duplicatecode-0.3.0 → duplicatecode-0.3.1}/crates/duplicatecode-engine/src/fragments.rs +0 -0
  19. {duplicatecode-0.3.0 → duplicatecode-0.3.1}/crates/duplicatecode-engine/src/lib.rs +0 -0
  20. {duplicatecode-0.3.0 → duplicatecode-0.3.1}/crates/duplicatecode-engine/src/mutate.rs +0 -0
  21. {duplicatecode-0.3.0 → duplicatecode-0.3.1}/crates/duplicatecode-engine/src/naming.rs +0 -0
  22. {duplicatecode-0.3.0 → duplicatecode-0.3.1}/crates/duplicatecode-engine/src/similarity.rs +0 -0
  23. {duplicatecode-0.3.0 → duplicatecode-0.3.1}/pyproject.toml +0 -0
@@ -198,7 +198,7 @@ dependencies = [
198
198
 
199
199
  [[package]]
200
200
  name = "duplicatecode"
201
- version = "0.3.0"
201
+ version = "0.3.1"
202
202
  dependencies = [
203
203
  "anyhow",
204
204
  "clap",
@@ -212,7 +212,7 @@ dependencies = [
212
212
 
213
213
  [[package]]
214
214
  name = "duplicatecode-engine"
215
- version = "0.3.0"
215
+ version = "0.3.1"
216
216
  dependencies = [
217
217
  "ignore",
218
218
  "serde",
@@ -3,6 +3,6 @@ resolver = "2"
3
3
  members = ["crates/duplicatecode-engine", "crates/duplicatecode-cli"]
4
4
 
5
5
  [workspace.package]
6
- version = "0.3.0"
6
+ version = "0.3.1"
7
7
  edition = "2021"
8
8
  license = "MIT"
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: duplicatecode
3
- Version: 0.3.0
3
+ Version: 0.3.1
4
4
  Classifier: Programming Language :: Rust
5
5
  Classifier: Environment :: Console
6
6
  Summary: Static (LLM-free) detection of duplicate/similar code in Python, TypeScript/JavaScript, SQL and C#
@@ -10,7 +10,7 @@ Description-Content-Type: text/markdown; charset=UTF-8; variant=GFM
10
10
 
11
11
  # duplicatecode
12
12
 
13
- Static (LLM-free) detection of duplicate / similar code in Python, TypeScript/JavaScript (TSX/JSX), SQL and C#,
13
+ Static by default (LLM-free, no network unless `--embed*` is used) detection of duplicate / similar code in Python, TypeScript/JavaScript (TSX/JSX), SQL and C#,
14
14
  aimed at catching an LLM re-implementing something that already exists. Input can be a git diff
15
15
  checked against existing source.
16
16
 
@@ -66,7 +66,14 @@ duplicatecode find "retry with exponential backoff" src/ # does something like
66
66
 
67
67
  ## Embeddings (bring your own key)
68
68
 
69
- Two optional signals, both off by default (the detector stays LLM-free unless you ask):
69
+ Two optional signals, both off by default (the detector stays static and offline unless you ask).
70
+
71
+ > **Privacy warning.** `--embed openai`, `--embed cohere`, and `--embed-code` / `--embeddings` pointed at any
72
+ > hosted endpoint (OpenAI, Cohere, Azure, ...) **upload the source text of your units** to that provider
73
+ > (`--embeddings` sends identifier names only). `find` embeds **every unit under the given paths**, so
74
+ > `find ... --embed openai` uploads all of it. The local presets (`minilm`, `qwen3`, `potion`) are meant for a
75
+ > server on loopback (default `127.0.0.1:8099`; `DUPLICATECODE_EMBED_ENDPOINT` overrides it): pointing that
76
+ > variable at a remote host sends text there too. Do not use hosted embeddings on code you may not share.
70
77
 
71
78
  - `--embeddings` embeds identifier *names* (cheap) as an extra name-similarity signal.
72
79
  - `--embed <preset>` is the short form (`minilm`, `qwen3`, `potion`, `openai`, `cohere`; `qwen3` is about 5x slower than `minilm` on CPU, so use it with a GPU or prefer `minilm`) and implies
@@ -215,9 +222,9 @@ Self-scans of three internal repositories (Python + TypeScript), 189 pairs judge
215
222
  before trusting them. A structure-heavy `--profile reimpl` exists for renamed re-implementations
216
223
  (use with `--min-name 0`) but has no real-repo precision data yet.
217
224
 
218
- ### Fresh held-out check (copies profile, threshold 0.6)
225
+ ### Fresh held-out check (copies profile, threshold 0.6 = the `copies` default for `scan`)
219
226
 
220
- A third, untouched sample of 57 pairs (nothing was tuned on it): 30% true duplicates, 47% incl. partial
227
+ A third, untouched sample of 57 pairs (nothing was tuned on it; the `reimpl` profile has different defaults, 0.35 for `scan`, and is not covered by this sample): 30% true duplicates, 47% incl. partial
221
228
  (OneSales 0/24, MDMApp 11/24, CCMT2 6/9 true). The 50%/76% above was optimistic because it was measured on the
222
229
  pairs used to choose the weights. Test code is about half of the noise but also holds real copies
223
230
  (25% true either way), so it is kept by default; `--skip-tests` drops it.
@@ -235,7 +242,7 @@ by Haiku and Sonnet without seeing the repo. Fraction where the original is foun
235
242
  | 0.5 | 45% | 14% | 30% | 1% |
236
243
  | 0.6 | 24% | 5% | 14% | 0% |
237
244
 
238
- So `diff` (checking new code) defaults to threshold 0.4 while `scan` (existing copies) defaults to 0.6. About half of
245
+ So with the `copies` profile `diff` (checking new code) defaults to threshold 0.4 while `scan` (existing copies) defaults to 0.6; the `reimpl` profile has its own lower defaults (see above). About half of
239
246
  independent re-implementations are caught; the rest are genuinely different code. The structure-heavy `reimpl`
240
247
  profile is not better on this test.
241
248
 
@@ -252,7 +259,7 @@ read/grep, two with this CLI. All distinct groups (82) were then judged blind by
252
259
  | OneSales, with CLI | 21 | 10 | 48% / 86% | 67% | 9 |
253
260
 
254
261
  Only 18 of 82 groups were found by both, so the approaches are complementary. The agents' reports are capped at 30
255
- groups, so the raw tool is a better measure of recall: `scan --threshold 0.6` (defaults) finds 38 of the 40 judged
262
+ groups, so the raw tool is a better measure of recall: `scan --threshold 0.6` (the `copies` defaults) finds 38 of the 40 judged
256
263
  true duplicates (25/25 MDMApp, 13/15 OneSales) and 23 of the 25 that the LLM-only agents found independently.
257
264
  What it still misses: a differently-written picker function (same purpose, different code) and a formatFileSize
258
265
  variant with different constants. Fixed after this test: tiny same-name exact copies (`min_tokens` 20 -> 8, near-exact
@@ -1,6 +1,6 @@
1
1
  # duplicatecode
2
2
 
3
- Static (LLM-free) detection of duplicate / similar code in Python, TypeScript/JavaScript (TSX/JSX), SQL and C#,
3
+ Static by default (LLM-free, no network unless `--embed*` is used) detection of duplicate / similar code in Python, TypeScript/JavaScript (TSX/JSX), SQL and C#,
4
4
  aimed at catching an LLM re-implementing something that already exists. Input can be a git diff
5
5
  checked against existing source.
6
6
 
@@ -56,7 +56,14 @@ duplicatecode find "retry with exponential backoff" src/ # does something like
56
56
 
57
57
  ## Embeddings (bring your own key)
58
58
 
59
- Two optional signals, both off by default (the detector stays LLM-free unless you ask):
59
+ Two optional signals, both off by default (the detector stays static and offline unless you ask).
60
+
61
+ > **Privacy warning.** `--embed openai`, `--embed cohere`, and `--embed-code` / `--embeddings` pointed at any
62
+ > hosted endpoint (OpenAI, Cohere, Azure, ...) **upload the source text of your units** to that provider
63
+ > (`--embeddings` sends identifier names only). `find` embeds **every unit under the given paths**, so
64
+ > `find ... --embed openai` uploads all of it. The local presets (`minilm`, `qwen3`, `potion`) are meant for a
65
+ > server on loopback (default `127.0.0.1:8099`; `DUPLICATECODE_EMBED_ENDPOINT` overrides it): pointing that
66
+ > variable at a remote host sends text there too. Do not use hosted embeddings on code you may not share.
60
67
 
61
68
  - `--embeddings` embeds identifier *names* (cheap) as an extra name-similarity signal.
62
69
  - `--embed <preset>` is the short form (`minilm`, `qwen3`, `potion`, `openai`, `cohere`; `qwen3` is about 5x slower than `minilm` on CPU, so use it with a GPU or prefer `minilm`) and implies
@@ -205,9 +212,9 @@ Self-scans of three internal repositories (Python + TypeScript), 189 pairs judge
205
212
  before trusting them. A structure-heavy `--profile reimpl` exists for renamed re-implementations
206
213
  (use with `--min-name 0`) but has no real-repo precision data yet.
207
214
 
208
- ### Fresh held-out check (copies profile, threshold 0.6)
215
+ ### Fresh held-out check (copies profile, threshold 0.6 = the `copies` default for `scan`)
209
216
 
210
- A third, untouched sample of 57 pairs (nothing was tuned on it): 30% true duplicates, 47% incl. partial
217
+ A third, untouched sample of 57 pairs (nothing was tuned on it; the `reimpl` profile has different defaults, 0.35 for `scan`, and is not covered by this sample): 30% true duplicates, 47% incl. partial
211
218
  (OneSales 0/24, MDMApp 11/24, CCMT2 6/9 true). The 50%/76% above was optimistic because it was measured on the
212
219
  pairs used to choose the weights. Test code is about half of the noise but also holds real copies
213
220
  (25% true either way), so it is kept by default; `--skip-tests` drops it.
@@ -225,7 +232,7 @@ by Haiku and Sonnet without seeing the repo. Fraction where the original is foun
225
232
  | 0.5 | 45% | 14% | 30% | 1% |
226
233
  | 0.6 | 24% | 5% | 14% | 0% |
227
234
 
228
- So `diff` (checking new code) defaults to threshold 0.4 while `scan` (existing copies) defaults to 0.6. About half of
235
+ So with the `copies` profile `diff` (checking new code) defaults to threshold 0.4 while `scan` (existing copies) defaults to 0.6; the `reimpl` profile has its own lower defaults (see above). About half of
229
236
  independent re-implementations are caught; the rest are genuinely different code. The structure-heavy `reimpl`
230
237
  profile is not better on this test.
231
238
 
@@ -242,7 +249,7 @@ read/grep, two with this CLI. All distinct groups (82) were then judged blind by
242
249
  | OneSales, with CLI | 21 | 10 | 48% / 86% | 67% | 9 |
243
250
 
244
251
  Only 18 of 82 groups were found by both, so the approaches are complementary. The agents' reports are capped at 30
245
- groups, so the raw tool is a better measure of recall: `scan --threshold 0.6` (defaults) finds 38 of the 40 judged
252
+ groups, so the raw tool is a better measure of recall: `scan --threshold 0.6` (the `copies` defaults) finds 38 of the 40 judged
246
253
  true duplicates (25/25 MDMApp, 13/15 OneSales) and 23 of the 25 that the LLM-only agents found independently.
247
254
  What it still misses: a differently-written picker function (same purpose, different code) and a formatFileSize
248
255
  variant with different constants. Fixed after this test: tiny same-name exact copies (`min_tokens` 20 -> 8, near-exact
@@ -246,6 +246,8 @@ pub fn make_mutation_groups(
246
246
  for (k, p) in files.iter().skip(skip).take(n).enumerate() {
247
247
  let text = std::fs::read_to_string(p)?;
248
248
  let dir = out.join(format!("c{:03}", skip + k));
249
+ // a leftover variant of a previous component would be mislabeled as a clone of this one
250
+ let _ = std::fs::remove_dir_all(&dir);
249
251
  std::fs::create_dir_all(&dir)?;
250
252
  std::fs::write(dir.join("orig.tsx"), &text)?;
251
253
  for m in Mutation::ALL {
@@ -30,19 +30,26 @@ enum Preset {
30
30
  impl Preset {
31
31
  fn config(self, dims: Option<u32>) -> Result<duplicatecode_engine::embed::EmbedConfig> {
32
32
  use duplicatecode_engine::embed::EmbedConfig;
33
- let local = |model: &str| {
33
+ let local = |model: &str| -> Result<EmbedConfig> {
34
34
  let endpoint = std::env::var("DUPLICATECODE_EMBED_ENDPOINT")
35
35
  .unwrap_or_else(|_| "http://127.0.0.1:8099/v1".into());
36
36
  // the local server returns full vectors whatever `dimensions` asks for
37
- EmbedConfig::new(endpoint, model.into(), Some("local".into()), None, None)
37
+ let cfg = EmbedConfig::new(endpoint, model.into(), Some("local".into()), None, None);
38
+ anyhow::ensure!(
39
+ cfg.is_loopback(),
40
+ "the local presets only talk to a server on this machine, but DUPLICATECODE_EMBED_ENDPOINT points at {}; \
41
+ unset it, or use --embed openai|cohere (or --embed-code with OPENAI_API_KEY etc.) to send code to a hosted provider on purpose",
42
+ cfg.host()
43
+ );
44
+ Ok(cfg)
38
45
  };
39
46
  let key = |name: &str| {
40
47
  std::env::var(name).with_context(|| format!("this preset needs {name} to be set"))
41
48
  };
42
49
  Ok(match self {
43
- Preset::Minilm => local("sentence-transformers/all-MiniLM-L6-v2"),
44
- Preset::Qwen3 => local("Qwen/Qwen3-Embedding-0.6B"),
45
- Preset::Potion => local("minishlab/potion-base-8M"),
50
+ Preset::Minilm => local("sentence-transformers/all-MiniLM-L6-v2")?,
51
+ Preset::Qwen3 => local("Qwen/Qwen3-Embedding-0.6B")?,
52
+ Preset::Potion => local("minishlab/potion-base-8M")?,
46
53
  Preset::Openai => EmbedConfig::new(
47
54
  "https://api.openai.com/v1".into(),
48
55
  "text-embedding-3-small".into(),
@@ -94,6 +101,11 @@ struct EmbedArgs {
94
101
  }
95
102
 
96
103
  impl EmbedArgs {
104
+ /// Any embedding feature is on.
105
+ fn enabled(&self) -> bool {
106
+ self.embeddings || self.code_enabled()
107
+ }
108
+
97
109
  /// Whole-unit embeddings are on for `--embed-code` and for any `--embed <preset>`.
98
110
  fn code_enabled(&self) -> bool {
99
111
  self.embed_code || self.embed.is_some()
@@ -113,6 +125,27 @@ impl EmbedArgs {
113
125
  return Ok(());
114
126
  }
115
127
  let cfg = self.config()?;
128
+ // say where the text goes BEFORE anything is sent
129
+ if cfg.is_loopback() {
130
+ eprintln!(
131
+ "embedding {} units via {} (this machine)",
132
+ units.len(),
133
+ cfg.host()
134
+ );
135
+ } else {
136
+ eprintln!(
137
+ "embedding {} units: unit text is sent to {} (model {})",
138
+ units.len(),
139
+ cfg.host(),
140
+ cfg.deployment
141
+ );
142
+ if cfg.is_cleartext_remote() {
143
+ eprintln!(
144
+ "warning: {} is plain http, so your API key and source text travel unencrypted",
145
+ cfg.endpoint
146
+ );
147
+ }
148
+ }
116
149
  let path = self
117
150
  .embed_cache
118
151
  .clone()
@@ -163,6 +196,30 @@ impl EmbedArgs {
163
196
  }
164
197
  }
165
198
 
199
+ /// Units under the options' size/kind filters can never be reported, so there is no reason to embed
200
+ /// (upload) them.
201
+ fn retain_matchable(
202
+ units: &mut Vec<duplicatecode_engine::Unit>,
203
+ min_tokens: usize,
204
+ skip_tests: bool,
205
+ ) {
206
+ units.retain(|u| u.token_count() >= min_tokens && !u.boilerplate && !(skip_tests && u.is_test));
207
+ }
208
+
209
+ /// Units of every root; with several roots the file names carry the root so equal names stay apart.
210
+ fn load_roots(paths: &[PathBuf], exclude: &[String]) -> Vec<duplicatecode_engine::Unit> {
211
+ let mut units = Vec::new();
212
+ for p in paths {
213
+ for mut u in load_units_with(p, exclude) {
214
+ if paths.len() > 1 {
215
+ u.file = format!("{}/{}", p.display(), u.file);
216
+ }
217
+ units.push(u);
218
+ }
219
+ }
220
+ units
221
+ }
222
+
166
223
  #[derive(Clone, Copy, clap::ValueEnum)]
167
224
  enum Profile {
168
225
  Copies,
@@ -397,7 +454,7 @@ enum Cmd {
397
454
  EvalGroups {
398
455
  #[arg(long, default_value = "eval/data/codenet")]
399
456
  root: PathBuf,
400
- /// Print only the final `score=` line.
457
+ /// Print only the summary lines (`dataset_score`, `dataset_threshold`, `score=`).
401
458
  #[arg(long)]
402
459
  quiet: bool,
403
460
  /// Write every pair's feature vector (TSV) for offline analysis.
@@ -548,6 +605,10 @@ fn main() -> Result<()> {
548
605
  }));
549
606
  }
550
607
  let mut corpus_units = load_units(&repo);
608
+ if embed.enabled() {
609
+ retain_matchable(&mut corpus_units, min_tokens, false);
610
+ retain_matchable(&mut queries, min_tokens, false);
611
+ }
551
612
  embed.apply(&mut corpus_units)?;
552
613
  embed.apply(&mut queries)?;
553
614
  let corpus = Corpus::new(corpus_units);
@@ -616,6 +677,9 @@ fn main() -> Result<()> {
616
677
  units.push(u);
617
678
  }
618
679
  }
680
+ if embed.enabled() {
681
+ units.retain(|u| !u.boilerplate && !(skip_tests && u.is_test));
682
+ }
619
683
  embed.apply(&mut units)?;
620
684
  let o = review::ReviewOptions {
621
685
  threshold,
@@ -711,10 +775,7 @@ fn main() -> Result<()> {
711
775
  cross_file,
712
776
  json,
713
777
  } => {
714
- let mut units = Vec::new();
715
- for p in &paths {
716
- units.extend(load_units_with(p, &exclude));
717
- }
778
+ let mut units = load_roots(&paths, &exclude);
718
779
  units.retain(|u| !u.boilerplate && !(skip_tests && u.is_test));
719
780
  let mut found = duplicatecode_engine::fragments::find_fragments(
720
781
  &units,
@@ -760,10 +821,7 @@ fn main() -> Result<()> {
760
821
  json,
761
822
  } => {
762
823
  embed.embed_code = true; // searching by description needs unit embeddings
763
- let mut units = Vec::new();
764
- for p in &paths {
765
- units.extend(load_units_with(p, &exclude));
766
- }
824
+ let mut units = load_roots(&paths, &exclude);
767
825
  units.retain(|u| u.token_count() >= min_tokens && !u.boilerplate);
768
826
  embed.apply(&mut units)?;
769
827
  let cfg = embed.config()?;
@@ -814,14 +872,9 @@ fn main() -> Result<()> {
814
872
  } => {
815
873
  let threshold =
816
874
  threshold.unwrap_or(profile.default_threshold(Command::Scan, embed.code_enabled()));
817
- let mut units = Vec::new();
818
- for p in &paths {
819
- for mut u in load_units_with(p, &exclude) {
820
- if paths.len() > 1 {
821
- u.file = format!("{}/{}", p.display(), u.file);
822
- }
823
- units.push(u);
824
- }
875
+ let mut units = load_roots(&paths, &exclude);
876
+ if embed.enabled() {
877
+ retain_matchable(&mut units, min_tokens, skip_tests);
825
878
  }
826
879
  embed.apply(&mut units)?;
827
880
  let corpus = Corpus::new(units.clone());
@@ -851,14 +904,23 @@ fn main() -> Result<()> {
851
904
  .collect();
852
905
  pairs.sort_by(|a, b| b.scores.combined.total_cmp(&a.scores.combined));
853
906
  let groups = group_pairs(&pairs);
854
- let by_place: std::collections::HashMap<String, &duplicatecode_engine::Unit> = units
907
+ // keyed by the full place: units that start on the same line (minified code, one-line
908
+ // classes) must not be mixed up
909
+ let place = |r: &duplicatecode_engine::index::UnitRef| {
910
+ (r.file.clone(), r.start_line, r.end_line, r.name.clone())
911
+ };
912
+ let by_place: std::collections::HashMap<_, &duplicatecode_engine::Unit> = units
855
913
  .iter()
856
- .map(|u| (format!("{}:{}", u.file, u.start_line), u))
914
+ .map(|u| {
915
+ (
916
+ (u.file.clone(), u.start_line, u.end_line, u.name.clone()),
917
+ u,
918
+ )
919
+ })
857
920
  .collect();
858
921
  let explanation = |m: &duplicatecode_engine::Match| {
859
- let a = by_place.get(&format!("{}:{}", m.query.file, m.query.start_line))?;
860
- let b =
861
- by_place.get(&format!("{}:{}", m.candidate.file, m.candidate.start_line))?;
922
+ let a = by_place.get(&place(&m.query))?;
923
+ let b = by_place.get(&place(&m.candidate))?;
862
924
  Some(duplicatecode_engine::explain::explain(a, b))
863
925
  };
864
926
  if pairs_out {