duplicatecode 0.3.1__tar.gz → 0.3.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {duplicatecode-0.3.1 → duplicatecode-0.3.2}/Cargo.lock +2 -2
- {duplicatecode-0.3.1 → duplicatecode-0.3.2}/Cargo.toml +1 -1
- {duplicatecode-0.3.1 → duplicatecode-0.3.2}/PKG-INFO +9 -6
- {duplicatecode-0.3.1 → duplicatecode-0.3.2}/README.md +8 -5
- {duplicatecode-0.3.1 → duplicatecode-0.3.2}/crates/duplicatecode-cli/src/main.rs +15 -9
- {duplicatecode-0.3.1 → duplicatecode-0.3.2}/crates/duplicatecode-cli/Cargo.toml +0 -0
- {duplicatecode-0.3.1 → duplicatecode-0.3.2}/crates/duplicatecode-cli/src/bench.rs +0 -0
- {duplicatecode-0.3.1 → duplicatecode-0.3.2}/crates/duplicatecode-cli/src/groups.rs +0 -0
- {duplicatecode-0.3.1 → duplicatecode-0.3.2}/crates/duplicatecode-cli/src/review.rs +0 -0
- {duplicatecode-0.3.1 → duplicatecode-0.3.2}/crates/duplicatecode-engine/Cargo.toml +0 -0
- {duplicatecode-0.3.1 → duplicatecode-0.3.2}/crates/duplicatecode-engine/src/diff.rs +0 -0
- {duplicatecode-0.3.1 → duplicatecode-0.3.2}/crates/duplicatecode-engine/src/embed.rs +0 -0
- {duplicatecode-0.3.1 → duplicatecode-0.3.2}/crates/duplicatecode-engine/src/explain.rs +0 -0
- {duplicatecode-0.3.1 → duplicatecode-0.3.2}/crates/duplicatecode-engine/src/fingerprint.rs +0 -0
- {duplicatecode-0.3.1 → duplicatecode-0.3.2}/crates/duplicatecode-engine/src/fragments.rs +0 -0
- {duplicatecode-0.3.1 → duplicatecode-0.3.2}/crates/duplicatecode-engine/src/index.rs +0 -0
- {duplicatecode-0.3.1 → duplicatecode-0.3.2}/crates/duplicatecode-engine/src/lang.rs +0 -0
- {duplicatecode-0.3.1 → duplicatecode-0.3.2}/crates/duplicatecode-engine/src/lib.rs +0 -0
- {duplicatecode-0.3.1 → duplicatecode-0.3.2}/crates/duplicatecode-engine/src/mutate.rs +0 -0
- {duplicatecode-0.3.1 → duplicatecode-0.3.2}/crates/duplicatecode-engine/src/naming.rs +0 -0
- {duplicatecode-0.3.1 → duplicatecode-0.3.2}/crates/duplicatecode-engine/src/similarity.rs +0 -0
- {duplicatecode-0.3.1 → duplicatecode-0.3.2}/crates/duplicatecode-engine/src/units.rs +0 -0
- {duplicatecode-0.3.1 → duplicatecode-0.3.2}/pyproject.toml +0 -0
|
@@ -198,7 +198,7 @@ dependencies = [
|
|
|
198
198
|
|
|
199
199
|
[[package]]
|
|
200
200
|
name = "duplicatecode"
|
|
201
|
-
version = "0.3.
|
|
201
|
+
version = "0.3.2"
|
|
202
202
|
dependencies = [
|
|
203
203
|
"anyhow",
|
|
204
204
|
"clap",
|
|
@@ -212,7 +212,7 @@ dependencies = [
|
|
|
212
212
|
|
|
213
213
|
[[package]]
|
|
214
214
|
name = "duplicatecode-engine"
|
|
215
|
-
version = "0.3.
|
|
215
|
+
version = "0.3.2"
|
|
216
216
|
dependencies = [
|
|
217
217
|
"ignore",
|
|
218
218
|
"serde",
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: duplicatecode
|
|
3
|
-
Version: 0.3.
|
|
3
|
+
Version: 0.3.2
|
|
4
4
|
Classifier: Programming Language :: Rust
|
|
5
5
|
Classifier: Environment :: Console
|
|
6
6
|
Summary: Static (LLM-free) detection of duplicate/similar code in Python, TypeScript/JavaScript, SQL and C#
|
|
@@ -71,13 +71,15 @@ Two optional signals, both off by default (the detector stays static and offline
|
|
|
71
71
|
> **Privacy warning.** `--embed openai`, `--embed cohere`, and `--embed-code` / `--embeddings` pointed at any
|
|
72
72
|
> hosted endpoint (OpenAI, Cohere, Azure, ...) **upload the source text of your units** to that provider
|
|
73
73
|
> (`--embeddings` sends identifier names only). `find` embeds **every unit under the given paths**, so
|
|
74
|
-
> `find ... --embed openai` uploads all of it. The local presets (`minilm`, `qwen3`, `potion`) are meant for a
|
|
75
|
-
> server on loopback (
|
|
74
|
+
> `find ... --embed openai` uploads all of it. The local presets (`gemma`, `minilm`, `qwen3`, `potion`) are meant for a
|
|
75
|
+
> server on loopback (`gemma`: ollama on `127.0.0.1:11434`, the others `127.0.0.1:8099`; `DUPLICATECODE_EMBED_ENDPOINT` overrides it): pointing that
|
|
76
76
|
> variable at a remote host sends text there too. Do not use hosted embeddings on code you may not share.
|
|
77
77
|
|
|
78
78
|
- `--embeddings` embeds identifier *names* (cheap) as an extra name-similarity signal.
|
|
79
|
-
- `--embed
|
|
80
|
-
|
|
79
|
+
- `--embed` is the short form and implies `--embed-code`. A bare `--embed` uses **`gemma`** (embeddinggemma via a local
|
|
80
|
+
[ollama](https://ollama.com): `ollama pull embeddinggemma`), the best model in the codebase benchmark (issue #12: re-implementation
|
|
81
|
+
retrieval R@1 0.94 vs 0.89 for the static detector alone, unchanged when function names are masked). Other presets:
|
|
82
|
+
`--embed=minilm|qwen3|potion|openai|cohere`; MiniLM is smaller but clearly weaker (0.85) and leans on names. Model ids go into the cache key, so vectors of different models never mix.
|
|
81
83
|
- `--embed-code` embeds the *whole text of every unit* (function, class, file, SQL statement) and blends
|
|
82
84
|
the cosine into the score (`--embed-weight`, default 0.35, others scaled by 1 - weight). It finds
|
|
83
85
|
re-implementations that share no tokens. On function-level LLM re-implementations it lifted Python from
|
|
@@ -97,7 +99,8 @@ export OPENAI_BASE_URL=http://localhost:11434/v1 OPENAI_API_KEY=anything OPENAI_
|
|
|
97
99
|
export AZURE_AI_FOUNDRY_ENDPOINT=https://<res>.services.ai.azure.com
|
|
98
100
|
export AZURE_AI_FOUNDRY_API_KEY=... AZURE_AI_FOUNDRY_EMBEDDING_DEPLOYMENT=text-embedding-3-small
|
|
99
101
|
# Azure OpenAI: AZURE_OPENAI_ENDPOINT / AZURE_OPENAI_API_KEY / AZURE_OPENAI_EMBEDDING_DEPLOYMENT
|
|
100
|
-
# Fully local, no key:
|
|
102
|
+
# Fully local, no key, recommended: ollama pull embeddinggemma && duplicatecode scan . --embed (GPU: ~5 ms per unit)
|
|
103
|
+
# Other local models: start the server once (CPU; it loads whichever model a request names)
|
|
101
104
|
# uv run --no-project --python 3.12 eval/embed_server.py
|
|
102
105
|
# duplicatecode scan . --embed minilm # or: qwen3, potion (default endpoint 127.0.0.1:8099)
|
|
103
106
|
# Hosted presets: --embed openai (OPENAI_API_KEY), --embed cohere (COHERE_API_KEY)
|
|
@@ -61,13 +61,15 @@ Two optional signals, both off by default (the detector stays static and offline
|
|
|
61
61
|
> **Privacy warning.** `--embed openai`, `--embed cohere`, and `--embed-code` / `--embeddings` pointed at any
|
|
62
62
|
> hosted endpoint (OpenAI, Cohere, Azure, ...) **upload the source text of your units** to that provider
|
|
63
63
|
> (`--embeddings` sends identifier names only). `find` embeds **every unit under the given paths**, so
|
|
64
|
-
> `find ... --embed openai` uploads all of it. The local presets (`minilm`, `qwen3`, `potion`) are meant for a
|
|
65
|
-
> server on loopback (
|
|
64
|
+
> `find ... --embed openai` uploads all of it. The local presets (`gemma`, `minilm`, `qwen3`, `potion`) are meant for a
|
|
65
|
+
> server on loopback (`gemma`: ollama on `127.0.0.1:11434`, the others `127.0.0.1:8099`; `DUPLICATECODE_EMBED_ENDPOINT` overrides it): pointing that
|
|
66
66
|
> variable at a remote host sends text there too. Do not use hosted embeddings on code you may not share.
|
|
67
67
|
|
|
68
68
|
- `--embeddings` embeds identifier *names* (cheap) as an extra name-similarity signal.
|
|
69
|
-
- `--embed
|
|
70
|
-
|
|
69
|
+
- `--embed` is the short form and implies `--embed-code`. A bare `--embed` uses **`gemma`** (embeddinggemma via a local
|
|
70
|
+
[ollama](https://ollama.com): `ollama pull embeddinggemma`), the best model in the codebase benchmark (issue #12: re-implementation
|
|
71
|
+
retrieval R@1 0.94 vs 0.89 for the static detector alone, unchanged when function names are masked). Other presets:
|
|
72
|
+
`--embed=minilm|qwen3|potion|openai|cohere`; MiniLM is smaller but clearly weaker (0.85) and leans on names. Model ids go into the cache key, so vectors of different models never mix.
|
|
71
73
|
- `--embed-code` embeds the *whole text of every unit* (function, class, file, SQL statement) and blends
|
|
72
74
|
the cosine into the score (`--embed-weight`, default 0.35, others scaled by 1 - weight). It finds
|
|
73
75
|
re-implementations that share no tokens. On function-level LLM re-implementations it lifted Python from
|
|
@@ -87,7 +89,8 @@ export OPENAI_BASE_URL=http://localhost:11434/v1 OPENAI_API_KEY=anything OPENAI_
|
|
|
87
89
|
export AZURE_AI_FOUNDRY_ENDPOINT=https://<res>.services.ai.azure.com
|
|
88
90
|
export AZURE_AI_FOUNDRY_API_KEY=... AZURE_AI_FOUNDRY_EMBEDDING_DEPLOYMENT=text-embedding-3-small
|
|
89
91
|
# Azure OpenAI: AZURE_OPENAI_ENDPOINT / AZURE_OPENAI_API_KEY / AZURE_OPENAI_EMBEDDING_DEPLOYMENT
|
|
90
|
-
# Fully local, no key:
|
|
92
|
+
# Fully local, no key, recommended: ollama pull embeddinggemma && duplicatecode scan . --embed (GPU: ~5 ms per unit)
|
|
93
|
+
# Other local models: start the server once (CPU; it loads whichever model a request names)
|
|
91
94
|
# uv run --no-project --python 3.12 eval/embed_server.py
|
|
92
95
|
# duplicatecode scan . --embed minilm # or: qwen3, potion (default endpoint 127.0.0.1:8099)
|
|
93
96
|
# Hosted presets: --embed openai (OPENAI_API_KEY), --embed cohere (COHERE_API_KEY)
|
|
@@ -11,11 +11,14 @@ use duplicatecode_engine::{
|
|
|
11
11
|
use std::io::Read;
|
|
12
12
|
use std::path::PathBuf;
|
|
13
13
|
|
|
14
|
-
/// Named embedding setups for `--embed`.
|
|
15
|
-
/// http://127.0.0.1:
|
|
14
|
+
/// Named embedding setups for `--embed`. `gemma` talks to a local ollama (default
|
|
15
|
+
/// http://127.0.0.1:11434/v1); the other local presets talk to `eval/embed_server.py` (default
|
|
16
|
+
/// http://127.0.0.1:8099/v1). DUPLICATECODE_EMBED_ENDPOINT overrides either, loopback hosts only.
|
|
16
17
|
#[derive(Clone, Copy, clap::ValueEnum)]
|
|
17
18
|
enum Preset {
|
|
18
|
-
///
|
|
19
|
+
/// embeddinggemma via local ollama (`ollama pull embeddinggemma`): best in the codebase bake-off, the default of a bare `--embed`
|
|
20
|
+
Gemma,
|
|
21
|
+
/// all-MiniLM-L6-v2, local server: small and fast, but below `gemma` on re-implementation retrieval
|
|
19
22
|
Minilm,
|
|
20
23
|
/// Qwen3-Embedding-0.6B, local: 8k-token context, slower
|
|
21
24
|
Qwen3,
|
|
@@ -30,9 +33,9 @@ enum Preset {
|
|
|
30
33
|
impl Preset {
|
|
31
34
|
fn config(self, dims: Option<u32>) -> Result<duplicatecode_engine::embed::EmbedConfig> {
|
|
32
35
|
use duplicatecode_engine::embed::EmbedConfig;
|
|
33
|
-
let local = |model: &str| -> Result<EmbedConfig> {
|
|
36
|
+
let local = |model: &str, default_endpoint: &str| -> Result<EmbedConfig> {
|
|
34
37
|
let endpoint = std::env::var("DUPLICATECODE_EMBED_ENDPOINT")
|
|
35
|
-
.unwrap_or_else(|_|
|
|
38
|
+
.unwrap_or_else(|_| default_endpoint.into());
|
|
36
39
|
// the local server returns full vectors whatever `dimensions` asks for
|
|
37
40
|
let cfg = EmbedConfig::new(endpoint, model.into(), Some("local".into()), None, None);
|
|
38
41
|
anyhow::ensure!(
|
|
@@ -43,13 +46,15 @@ impl Preset {
|
|
|
43
46
|
);
|
|
44
47
|
Ok(cfg)
|
|
45
48
|
};
|
|
49
|
+
const SERVER: &str = "http://127.0.0.1:8099/v1";
|
|
46
50
|
let key = |name: &str| {
|
|
47
51
|
std::env::var(name).with_context(|| format!("this preset needs {name} to be set"))
|
|
48
52
|
};
|
|
49
53
|
Ok(match self {
|
|
50
|
-
Preset::
|
|
51
|
-
Preset::
|
|
52
|
-
Preset::
|
|
54
|
+
Preset::Gemma => local("embeddinggemma", "http://127.0.0.1:11434/v1")?,
|
|
55
|
+
Preset::Minilm => local("sentence-transformers/all-MiniLM-L6-v2", SERVER)?,
|
|
56
|
+
Preset::Qwen3 => local("Qwen/Qwen3-Embedding-0.6B", SERVER)?,
|
|
57
|
+
Preset::Potion => local("minishlab/potion-base-8M", SERVER)?,
|
|
53
58
|
Preset::Openai => EmbedConfig::new(
|
|
54
59
|
"https://api.openai.com/v1".into(),
|
|
55
60
|
"text-embedding-3-small".into(),
|
|
@@ -77,7 +82,8 @@ struct EmbedArgs {
|
|
|
77
82
|
#[arg(long)]
|
|
78
83
|
embeddings: bool,
|
|
79
84
|
/// Embed every unit with a named setup (implies --embed-code); see `--help` for the presets.
|
|
80
|
-
|
|
85
|
+
/// A bare `--embed` means `gemma`; write `--embed=minilm` etc. for the others.
|
|
86
|
+
#[arg(long, value_enum, num_args = 0..=1, require_equals = true, default_missing_value = "gemma")]
|
|
81
87
|
embed: Option<Preset>,
|
|
82
88
|
/// Embed the full text of every unit (function, class, file, statement) and blend the cosine of
|
|
83
89
|
/// the two vectors into the score. Finds re-implementations that share no tokens.
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|