duplicatecode 0.3.1__tar.gz → 0.3.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (23) hide show
  1. {duplicatecode-0.3.1 → duplicatecode-0.3.2}/Cargo.lock +2 -2
  2. {duplicatecode-0.3.1 → duplicatecode-0.3.2}/Cargo.toml +1 -1
  3. {duplicatecode-0.3.1 → duplicatecode-0.3.2}/PKG-INFO +9 -6
  4. {duplicatecode-0.3.1 → duplicatecode-0.3.2}/README.md +8 -5
  5. {duplicatecode-0.3.1 → duplicatecode-0.3.2}/crates/duplicatecode-cli/src/main.rs +15 -9
  6. {duplicatecode-0.3.1 → duplicatecode-0.3.2}/crates/duplicatecode-cli/Cargo.toml +0 -0
  7. {duplicatecode-0.3.1 → duplicatecode-0.3.2}/crates/duplicatecode-cli/src/bench.rs +0 -0
  8. {duplicatecode-0.3.1 → duplicatecode-0.3.2}/crates/duplicatecode-cli/src/groups.rs +0 -0
  9. {duplicatecode-0.3.1 → duplicatecode-0.3.2}/crates/duplicatecode-cli/src/review.rs +0 -0
  10. {duplicatecode-0.3.1 → duplicatecode-0.3.2}/crates/duplicatecode-engine/Cargo.toml +0 -0
  11. {duplicatecode-0.3.1 → duplicatecode-0.3.2}/crates/duplicatecode-engine/src/diff.rs +0 -0
  12. {duplicatecode-0.3.1 → duplicatecode-0.3.2}/crates/duplicatecode-engine/src/embed.rs +0 -0
  13. {duplicatecode-0.3.1 → duplicatecode-0.3.2}/crates/duplicatecode-engine/src/explain.rs +0 -0
  14. {duplicatecode-0.3.1 → duplicatecode-0.3.2}/crates/duplicatecode-engine/src/fingerprint.rs +0 -0
  15. {duplicatecode-0.3.1 → duplicatecode-0.3.2}/crates/duplicatecode-engine/src/fragments.rs +0 -0
  16. {duplicatecode-0.3.1 → duplicatecode-0.3.2}/crates/duplicatecode-engine/src/index.rs +0 -0
  17. {duplicatecode-0.3.1 → duplicatecode-0.3.2}/crates/duplicatecode-engine/src/lang.rs +0 -0
  18. {duplicatecode-0.3.1 → duplicatecode-0.3.2}/crates/duplicatecode-engine/src/lib.rs +0 -0
  19. {duplicatecode-0.3.1 → duplicatecode-0.3.2}/crates/duplicatecode-engine/src/mutate.rs +0 -0
  20. {duplicatecode-0.3.1 → duplicatecode-0.3.2}/crates/duplicatecode-engine/src/naming.rs +0 -0
  21. {duplicatecode-0.3.1 → duplicatecode-0.3.2}/crates/duplicatecode-engine/src/similarity.rs +0 -0
  22. {duplicatecode-0.3.1 → duplicatecode-0.3.2}/crates/duplicatecode-engine/src/units.rs +0 -0
  23. {duplicatecode-0.3.1 → duplicatecode-0.3.2}/pyproject.toml +0 -0
@@ -198,7 +198,7 @@ dependencies = [
198
198
 
199
199
  [[package]]
200
200
  name = "duplicatecode"
201
- version = "0.3.1"
201
+ version = "0.3.2"
202
202
  dependencies = [
203
203
  "anyhow",
204
204
  "clap",
@@ -212,7 +212,7 @@ dependencies = [
212
212
 
213
213
  [[package]]
214
214
  name = "duplicatecode-engine"
215
- version = "0.3.1"
215
+ version = "0.3.2"
216
216
  dependencies = [
217
217
  "ignore",
218
218
  "serde",
@@ -3,6 +3,6 @@ resolver = "2"
3
3
  members = ["crates/duplicatecode-engine", "crates/duplicatecode-cli"]
4
4
 
5
5
  [workspace.package]
6
- version = "0.3.1"
6
+ version = "0.3.2"
7
7
  edition = "2021"
8
8
  license = "MIT"
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: duplicatecode
3
- Version: 0.3.1
3
+ Version: 0.3.2
4
4
  Classifier: Programming Language :: Rust
5
5
  Classifier: Environment :: Console
6
6
  Summary: Static (LLM-free) detection of duplicate/similar code in Python, TypeScript/JavaScript, SQL and C#
@@ -71,13 +71,15 @@ Two optional signals, both off by default (the detector stays static and offline
71
71
  > **Privacy warning.** `--embed openai`, `--embed cohere`, and `--embed-code` / `--embeddings` pointed at any
72
72
  > hosted endpoint (OpenAI, Cohere, Azure, ...) **upload the source text of your units** to that provider
73
73
  > (`--embeddings` sends identifier names only). `find` embeds **every unit under the given paths**, so
74
- > `find ... --embed openai` uploads all of it. The local presets (`minilm`, `qwen3`, `potion`) are meant for a
75
- > server on loopback (default `127.0.0.1:8099`; `DUPLICATECODE_EMBED_ENDPOINT` overrides it): pointing that
74
+ > `find ... --embed openai` uploads all of it. The local presets (`gemma`, `minilm`, `qwen3`, `potion`) are meant for a
75
+ > server on loopback (`gemma`: ollama on `127.0.0.1:11434`, the others `127.0.0.1:8099`; `DUPLICATECODE_EMBED_ENDPOINT` overrides it): pointing that
76
76
  > variable at a remote host sends text there too. Do not use hosted embeddings on code you may not share.
77
77
 
78
78
  - `--embeddings` embeds identifier *names* (cheap) as an extra name-similarity signal.
79
- - `--embed <preset>` is the short form (`minilm`, `qwen3`, `potion`, `openai`, `cohere`; `qwen3` is about 5x slower than `minilm` on CPU, so use it with a GPU or prefer `minilm`) and implies
80
- `--embed-code`. Model ids go into the cache key, so vectors of different models never mix.
79
+ - `--embed` is the short form and implies `--embed-code`. A bare `--embed` uses **`gemma`** (embeddinggemma via a local
80
+ [ollama](https://ollama.com): `ollama pull embeddinggemma`), the best model in the codebase benchmark (issue #12: re-implementation
81
+ retrieval R@1 0.94 vs 0.89 for the static detector alone, unchanged when function names are masked). Other presets:
82
+ `--embed=minilm|qwen3|potion|openai|cohere`; MiniLM is smaller but clearly weaker (0.85) and leans on names. Model ids go into the cache key, so vectors of different models never mix.
81
83
  - `--embed-code` embeds the *whole text of every unit* (function, class, file, SQL statement) and blends
82
84
  the cosine into the score (`--embed-weight`, default 0.35, others scaled by 1 - weight). It finds
83
85
  re-implementations that share no tokens. On function-level LLM re-implementations it lifted Python from
@@ -97,7 +99,8 @@ export OPENAI_BASE_URL=http://localhost:11434/v1 OPENAI_API_KEY=anything OPENAI_
97
99
  export AZURE_AI_FOUNDRY_ENDPOINT=https://<res>.services.ai.azure.com
98
100
  export AZURE_AI_FOUNDRY_API_KEY=... AZURE_AI_FOUNDRY_EMBEDDING_DEPLOYMENT=text-embedding-3-small
99
101
  # Azure OpenAI: AZURE_OPENAI_ENDPOINT / AZURE_OPENAI_API_KEY / AZURE_OPENAI_EMBEDDING_DEPLOYMENT
100
- # Fully local, no key: start the server once (CPU; it loads whichever model a request names)
102
+ # Fully local, no key, recommended: ollama pull embeddinggemma && duplicatecode scan . --embed (GPU: ~5 ms per unit)
103
+ # Other local models: start the server once (CPU; it loads whichever model a request names)
101
104
  # uv run --no-project --python 3.12 eval/embed_server.py
102
105
  # duplicatecode scan . --embed minilm # or: qwen3, potion (default endpoint 127.0.0.1:8099)
103
106
  # Hosted presets: --embed openai (OPENAI_API_KEY), --embed cohere (COHERE_API_KEY)
@@ -61,13 +61,15 @@ Two optional signals, both off by default (the detector stays static and offline
61
61
  > **Privacy warning.** `--embed openai`, `--embed cohere`, and `--embed-code` / `--embeddings` pointed at any
62
62
  > hosted endpoint (OpenAI, Cohere, Azure, ...) **upload the source text of your units** to that provider
63
63
  > (`--embeddings` sends identifier names only). `find` embeds **every unit under the given paths**, so
64
- > `find ... --embed openai` uploads all of it. The local presets (`minilm`, `qwen3`, `potion`) are meant for a
65
- > server on loopback (default `127.0.0.1:8099`; `DUPLICATECODE_EMBED_ENDPOINT` overrides it): pointing that
64
+ > `find ... --embed openai` uploads all of it. The local presets (`gemma`, `minilm`, `qwen3`, `potion`) are meant for a
65
+ > server on loopback (`gemma`: ollama on `127.0.0.1:11434`, the others `127.0.0.1:8099`; `DUPLICATECODE_EMBED_ENDPOINT` overrides it): pointing that
66
66
  > variable at a remote host sends text there too. Do not use hosted embeddings on code you may not share.
67
67
 
68
68
  - `--embeddings` embeds identifier *names* (cheap) as an extra name-similarity signal.
69
- - `--embed <preset>` is the short form (`minilm`, `qwen3`, `potion`, `openai`, `cohere`; `qwen3` is about 5x slower than `minilm` on CPU, so use it with a GPU or prefer `minilm`) and implies
70
- `--embed-code`. Model ids go into the cache key, so vectors of different models never mix.
69
+ - `--embed` is the short form and implies `--embed-code`. A bare `--embed` uses **`gemma`** (embeddinggemma via a local
70
+ [ollama](https://ollama.com): `ollama pull embeddinggemma`), the best model in the codebase benchmark (issue #12: re-implementation
71
+ retrieval R@1 0.94 vs 0.89 for the static detector alone, unchanged when function names are masked). Other presets:
72
+ `--embed=minilm|qwen3|potion|openai|cohere`; MiniLM is smaller but clearly weaker (0.85) and leans on names. Model ids go into the cache key, so vectors of different models never mix.
71
73
  - `--embed-code` embeds the *whole text of every unit* (function, class, file, SQL statement) and blends
72
74
  the cosine into the score (`--embed-weight`, default 0.35, others scaled by 1 - weight). It finds
73
75
  re-implementations that share no tokens. On function-level LLM re-implementations it lifted Python from
@@ -87,7 +89,8 @@ export OPENAI_BASE_URL=http://localhost:11434/v1 OPENAI_API_KEY=anything OPENAI_
87
89
  export AZURE_AI_FOUNDRY_ENDPOINT=https://<res>.services.ai.azure.com
88
90
  export AZURE_AI_FOUNDRY_API_KEY=... AZURE_AI_FOUNDRY_EMBEDDING_DEPLOYMENT=text-embedding-3-small
89
91
  # Azure OpenAI: AZURE_OPENAI_ENDPOINT / AZURE_OPENAI_API_KEY / AZURE_OPENAI_EMBEDDING_DEPLOYMENT
90
- # Fully local, no key: start the server once (CPU; it loads whichever model a request names)
92
+ # Fully local, no key, recommended: ollama pull embeddinggemma && duplicatecode scan . --embed (GPU: ~5 ms per unit)
93
+ # Other local models: start the server once (CPU; it loads whichever model a request names)
91
94
  # uv run --no-project --python 3.12 eval/embed_server.py
92
95
  # duplicatecode scan . --embed minilm # or: qwen3, potion (default endpoint 127.0.0.1:8099)
93
96
  # Hosted presets: --embed openai (OPENAI_API_KEY), --embed cohere (COHERE_API_KEY)
@@ -11,11 +11,14 @@ use duplicatecode_engine::{
11
11
  use std::io::Read;
12
12
  use std::path::PathBuf;
13
13
 
14
- /// Named embedding setups for `--embed`. Local presets talk to `eval/embed_server.py` (default
15
- /// http://127.0.0.1:8099/v1, override with DUPLICATECODE_EMBED_ENDPOINT), which loads the named model.
14
+ /// Named embedding setups for `--embed`. `gemma` talks to a local ollama (default
15
+ /// http://127.0.0.1:11434/v1); the other local presets talk to `eval/embed_server.py` (default
16
+ /// http://127.0.0.1:8099/v1). DUPLICATECODE_EMBED_ENDPOINT overrides either, loopback hosts only.
16
17
  #[derive(Clone, Copy, clap::ValueEnum)]
17
18
  enum Preset {
18
- /// all-MiniLM-L6-v2, local: fastest, tied for best quality in the bake-off
19
+ /// embeddinggemma via local ollama (`ollama pull embeddinggemma`): best in the codebase bake-off, the default of a bare `--embed`
20
+ Gemma,
21
+ /// all-MiniLM-L6-v2, local server: small and fast, but below `gemma` on re-implementation retrieval
19
22
  Minilm,
20
23
  /// Qwen3-Embedding-0.6B, local: 8k-token context, slower
21
24
  Qwen3,
@@ -30,9 +33,9 @@ enum Preset {
30
33
  impl Preset {
31
34
  fn config(self, dims: Option<u32>) -> Result<duplicatecode_engine::embed::EmbedConfig> {
32
35
  use duplicatecode_engine::embed::EmbedConfig;
33
- let local = |model: &str| -> Result<EmbedConfig> {
36
+ let local = |model: &str, default_endpoint: &str| -> Result<EmbedConfig> {
34
37
  let endpoint = std::env::var("DUPLICATECODE_EMBED_ENDPOINT")
35
- .unwrap_or_else(|_| "http://127.0.0.1:8099/v1".into());
38
+ .unwrap_or_else(|_| default_endpoint.into());
36
39
  // the local server returns full vectors whatever `dimensions` asks for
37
40
  let cfg = EmbedConfig::new(endpoint, model.into(), Some("local".into()), None, None);
38
41
  anyhow::ensure!(
@@ -43,13 +46,15 @@ impl Preset {
43
46
  );
44
47
  Ok(cfg)
45
48
  };
49
+ const SERVER: &str = "http://127.0.0.1:8099/v1";
46
50
  let key = |name: &str| {
47
51
  std::env::var(name).with_context(|| format!("this preset needs {name} to be set"))
48
52
  };
49
53
  Ok(match self {
50
- Preset::Minilm => local("sentence-transformers/all-MiniLM-L6-v2")?,
51
- Preset::Qwen3 => local("Qwen/Qwen3-Embedding-0.6B")?,
52
- Preset::Potion => local("minishlab/potion-base-8M")?,
54
+ Preset::Gemma => local("embeddinggemma", "http://127.0.0.1:11434/v1")?,
55
+ Preset::Minilm => local("sentence-transformers/all-MiniLM-L6-v2", SERVER)?,
56
+ Preset::Qwen3 => local("Qwen/Qwen3-Embedding-0.6B", SERVER)?,
57
+ Preset::Potion => local("minishlab/potion-base-8M", SERVER)?,
53
58
  Preset::Openai => EmbedConfig::new(
54
59
  "https://api.openai.com/v1".into(),
55
60
  "text-embedding-3-small".into(),
@@ -77,7 +82,8 @@ struct EmbedArgs {
77
82
  #[arg(long)]
78
83
  embeddings: bool,
79
84
  /// Embed every unit with a named setup (implies --embed-code); see `--help` for the presets.
80
- #[arg(long, value_enum)]
85
+ /// A bare `--embed` means `gemma`; write `--embed=minilm` etc. for the others.
86
+ #[arg(long, value_enum, num_args = 0..=1, require_equals = true, default_missing_value = "gemma")]
81
87
  embed: Option<Preset>,
82
88
  /// Embed the full text of every unit (function, class, file, statement) and blend the cosine of
83
89
  /// the two vectors into the score. Finds re-implementations that share no tokens.