duplicatecode 0.3.0__tar.gz → 0.3.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {duplicatecode-0.3.0 → duplicatecode-0.3.2}/Cargo.lock +2 -2
- {duplicatecode-0.3.0 → duplicatecode-0.3.2}/Cargo.toml +1 -1
- {duplicatecode-0.3.0 → duplicatecode-0.3.2}/PKG-INFO +20 -10
- {duplicatecode-0.3.0 → duplicatecode-0.3.2}/README.md +19 -9
- {duplicatecode-0.3.0 → duplicatecode-0.3.2}/crates/duplicatecode-cli/src/groups.rs +2 -0
- {duplicatecode-0.3.0 → duplicatecode-0.3.2}/crates/duplicatecode-cli/src/main.rs +100 -32
- {duplicatecode-0.3.0 → duplicatecode-0.3.2}/crates/duplicatecode-engine/src/embed.rs +412 -83
- {duplicatecode-0.3.0 → duplicatecode-0.3.2}/crates/duplicatecode-engine/src/index.rs +190 -26
- {duplicatecode-0.3.0 → duplicatecode-0.3.2}/crates/duplicatecode-engine/src/lang.rs +3 -2
- {duplicatecode-0.3.0 → duplicatecode-0.3.2}/crates/duplicatecode-engine/src/units.rs +142 -12
- {duplicatecode-0.3.0 → duplicatecode-0.3.2}/crates/duplicatecode-cli/Cargo.toml +0 -0
- {duplicatecode-0.3.0 → duplicatecode-0.3.2}/crates/duplicatecode-cli/src/bench.rs +0 -0
- {duplicatecode-0.3.0 → duplicatecode-0.3.2}/crates/duplicatecode-cli/src/review.rs +0 -0
- {duplicatecode-0.3.0 → duplicatecode-0.3.2}/crates/duplicatecode-engine/Cargo.toml +0 -0
- {duplicatecode-0.3.0 → duplicatecode-0.3.2}/crates/duplicatecode-engine/src/diff.rs +0 -0
- {duplicatecode-0.3.0 → duplicatecode-0.3.2}/crates/duplicatecode-engine/src/explain.rs +0 -0
- {duplicatecode-0.3.0 → duplicatecode-0.3.2}/crates/duplicatecode-engine/src/fingerprint.rs +0 -0
- {duplicatecode-0.3.0 → duplicatecode-0.3.2}/crates/duplicatecode-engine/src/fragments.rs +0 -0
- {duplicatecode-0.3.0 → duplicatecode-0.3.2}/crates/duplicatecode-engine/src/lib.rs +0 -0
- {duplicatecode-0.3.0 → duplicatecode-0.3.2}/crates/duplicatecode-engine/src/mutate.rs +0 -0
- {duplicatecode-0.3.0 → duplicatecode-0.3.2}/crates/duplicatecode-engine/src/naming.rs +0 -0
- {duplicatecode-0.3.0 → duplicatecode-0.3.2}/crates/duplicatecode-engine/src/similarity.rs +0 -0
- {duplicatecode-0.3.0 → duplicatecode-0.3.2}/pyproject.toml +0 -0
|
@@ -198,7 +198,7 @@ dependencies = [
|
|
|
198
198
|
|
|
199
199
|
[[package]]
|
|
200
200
|
name = "duplicatecode"
|
|
201
|
-
version = "0.3.
|
|
201
|
+
version = "0.3.2"
|
|
202
202
|
dependencies = [
|
|
203
203
|
"anyhow",
|
|
204
204
|
"clap",
|
|
@@ -212,7 +212,7 @@ dependencies = [
|
|
|
212
212
|
|
|
213
213
|
[[package]]
|
|
214
214
|
name = "duplicatecode-engine"
|
|
215
|
-
version = "0.3.
|
|
215
|
+
version = "0.3.2"
|
|
216
216
|
dependencies = [
|
|
217
217
|
"ignore",
|
|
218
218
|
"serde",
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: duplicatecode
|
|
3
|
-
Version: 0.3.
|
|
3
|
+
Version: 0.3.2
|
|
4
4
|
Classifier: Programming Language :: Rust
|
|
5
5
|
Classifier: Environment :: Console
|
|
6
6
|
Summary: Static (LLM-free) detection of duplicate/similar code in Python, TypeScript/JavaScript, SQL and C#
|
|
@@ -10,7 +10,7 @@ Description-Content-Type: text/markdown; charset=UTF-8; variant=GFM
|
|
|
10
10
|
|
|
11
11
|
# duplicatecode
|
|
12
12
|
|
|
13
|
-
Static (LLM-free) detection of duplicate / similar code in Python, TypeScript/JavaScript (TSX/JSX), SQL and C#,
|
|
13
|
+
Static by default (LLM-free, no network unless `--embed*` is used) detection of duplicate / similar code in Python, TypeScript/JavaScript (TSX/JSX), SQL and C#,
|
|
14
14
|
aimed at catching an LLM re-implementing something that already exists. Input can be a git diff
|
|
15
15
|
checked against existing source.
|
|
16
16
|
|
|
@@ -66,11 +66,20 @@ duplicatecode find "retry with exponential backoff" src/ # does something like
|
|
|
66
66
|
|
|
67
67
|
## Embeddings (bring your own key)
|
|
68
68
|
|
|
69
|
-
Two optional signals, both off by default (the detector stays
|
|
69
|
+
Two optional signals, both off by default (the detector stays static and offline unless you ask).
|
|
70
|
+
|
|
71
|
+
> **Privacy warning.** `--embed openai`, `--embed cohere`, and `--embed-code` / `--embeddings` pointed at any
|
|
72
|
+
> hosted endpoint (OpenAI, Cohere, Azure, ...) **upload the source text of your units** to that provider
|
|
73
|
+
> (`--embeddings` sends identifier names only). `find` embeds **every unit under the given paths**, so
|
|
74
|
+
> `find ... --embed openai` uploads all of it. The local presets (`gemma`, `minilm`, `qwen3`, `potion`) are meant for a
|
|
75
|
+
> server on loopback (`gemma`: ollama on `127.0.0.1:11434`, the others `127.0.0.1:8099`; `DUPLICATECODE_EMBED_ENDPOINT` overrides it): pointing that
|
|
76
|
+
> variable at a remote host sends text there too. Do not use hosted embeddings on code you may not share.
|
|
70
77
|
|
|
71
78
|
- `--embeddings` embeds identifier *names* (cheap) as an extra name-similarity signal.
|
|
72
|
-
- `--embed
|
|
73
|
-
|
|
79
|
+
- `--embed` is the short form and implies `--embed-code`. A bare `--embed` uses **`gemma`** (embeddinggemma via a local
|
|
80
|
+
[ollama](https://ollama.com): `ollama pull embeddinggemma`), the best model in the codebase benchmark (issue #12: re-implementation
|
|
81
|
+
retrieval R@1 0.94 vs 0.89 for the static detector alone, unchanged when function names are masked). Other presets:
|
|
82
|
+
`--embed=minilm|qwen3|potion|openai|cohere`; MiniLM is smaller but clearly weaker (0.85) and leans on names. Model ids go into the cache key, so vectors of different models never mix.
|
|
74
83
|
- `--embed-code` embeds the *whole text of every unit* (function, class, file, SQL statement) and blends
|
|
75
84
|
the cosine into the score (`--embed-weight`, default 0.35, others scaled by 1 - weight). It finds
|
|
76
85
|
re-implementations that share no tokens. On function-level LLM re-implementations it lifted Python from
|
|
@@ -90,7 +99,8 @@ export OPENAI_BASE_URL=http://localhost:11434/v1 OPENAI_API_KEY=anything OPENAI_
|
|
|
90
99
|
export AZURE_AI_FOUNDRY_ENDPOINT=https://<res>.services.ai.azure.com
|
|
91
100
|
export AZURE_AI_FOUNDRY_API_KEY=... AZURE_AI_FOUNDRY_EMBEDDING_DEPLOYMENT=text-embedding-3-small
|
|
92
101
|
# Azure OpenAI: AZURE_OPENAI_ENDPOINT / AZURE_OPENAI_API_KEY / AZURE_OPENAI_EMBEDDING_DEPLOYMENT
|
|
93
|
-
# Fully local, no key:
|
|
102
|
+
# Fully local, no key, recommended: ollama pull embeddinggemma && duplicatecode scan . --embed (GPU: ~5 ms per unit)
|
|
103
|
+
# Other local models: start the server once (CPU; it loads whichever model a request names)
|
|
94
104
|
# uv run --no-project --python 3.12 eval/embed_server.py
|
|
95
105
|
# duplicatecode scan . --embed minilm # or: qwen3, potion (default endpoint 127.0.0.1:8099)
|
|
96
106
|
# Hosted presets: --embed openai (OPENAI_API_KEY), --embed cohere (COHERE_API_KEY)
|
|
@@ -215,9 +225,9 @@ Self-scans of three internal repositories (Python + TypeScript), 189 pairs judge
|
|
|
215
225
|
before trusting them. A structure-heavy `--profile reimpl` exists for renamed re-implementations
|
|
216
226
|
(use with `--min-name 0`) but has no real-repo precision data yet.
|
|
217
227
|
|
|
218
|
-
### Fresh held-out check (copies profile, threshold 0.6)
|
|
228
|
+
### Fresh held-out check (copies profile, threshold 0.6 = the `copies` default for `scan`)
|
|
219
229
|
|
|
220
|
-
A third, untouched sample of 57 pairs (nothing was tuned on it): 30% true duplicates, 47% incl. partial
|
|
230
|
+
A third, untouched sample of 57 pairs (nothing was tuned on it; the `reimpl` profile has different defaults, 0.35 for `scan`, and is not covered by this sample): 30% true duplicates, 47% incl. partial
|
|
221
231
|
(OneSales 0/24, MDMApp 11/24, CCMT2 6/9 true). The 50%/76% above was optimistic because it was measured on the
|
|
222
232
|
pairs used to choose the weights. Test code is about half of the noise but also holds real copies
|
|
223
233
|
(25% true either way), so it is kept by default; `--skip-tests` drops it.
|
|
@@ -235,7 +245,7 @@ by Haiku and Sonnet without seeing the repo. Fraction where the original is foun
|
|
|
235
245
|
| 0.5 | 45% | 14% | 30% | 1% |
|
|
236
246
|
| 0.6 | 24% | 5% | 14% | 0% |
|
|
237
247
|
|
|
238
|
-
So `diff` (checking new code) defaults to threshold 0.4 while `scan` (existing copies) defaults to 0.6. About half of
|
|
248
|
+
So with the `copies` profile `diff` (checking new code) defaults to threshold 0.4 while `scan` (existing copies) defaults to 0.6; the `reimpl` profile has its own lower defaults (see above). About half of
|
|
239
249
|
independent re-implementations are caught; the rest are genuinely different code. The structure-heavy `reimpl`
|
|
240
250
|
profile is not better on this test.
|
|
241
251
|
|
|
@@ -252,7 +262,7 @@ read/grep, two with this CLI. All distinct groups (82) were then judged blind by
|
|
|
252
262
|
| OneSales, with CLI | 21 | 10 | 48% / 86% | 67% | 9 |
|
|
253
263
|
|
|
254
264
|
Only 18 of 82 groups were found by both, so the approaches are complementary. The agents' reports are capped at 30
|
|
255
|
-
groups, so the raw tool is a better measure of recall: `scan --threshold 0.6` (defaults) finds 38 of the 40 judged
|
|
265
|
+
groups, so the raw tool is a better measure of recall: `scan --threshold 0.6` (the `copies` defaults) finds 38 of the 40 judged
|
|
256
266
|
true duplicates (25/25 MDMApp, 13/15 OneSales) and 23 of the 25 that the LLM-only agents found independently.
|
|
257
267
|
What it still misses: a differently-written picker function (same purpose, different code) and a formatFileSize
|
|
258
268
|
variant with different constants. Fixed after this test: tiny same-name exact copies (`min_tokens` 20 -> 8, near-exact
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
# duplicatecode
|
|
2
2
|
|
|
3
|
-
Static (LLM-free) detection of duplicate / similar code in Python, TypeScript/JavaScript (TSX/JSX), SQL and C#,
|
|
3
|
+
Static by default (LLM-free, no network unless `--embed*` is used) detection of duplicate / similar code in Python, TypeScript/JavaScript (TSX/JSX), SQL and C#,
|
|
4
4
|
aimed at catching an LLM re-implementing something that already exists. Input can be a git diff
|
|
5
5
|
checked against existing source.
|
|
6
6
|
|
|
@@ -56,11 +56,20 @@ duplicatecode find "retry with exponential backoff" src/ # does something like
|
|
|
56
56
|
|
|
57
57
|
## Embeddings (bring your own key)
|
|
58
58
|
|
|
59
|
-
Two optional signals, both off by default (the detector stays
|
|
59
|
+
Two optional signals, both off by default (the detector stays static and offline unless you ask).
|
|
60
|
+
|
|
61
|
+
> **Privacy warning.** `--embed openai`, `--embed cohere`, and `--embed-code` / `--embeddings` pointed at any
|
|
62
|
+
> hosted endpoint (OpenAI, Cohere, Azure, ...) **upload the source text of your units** to that provider
|
|
63
|
+
> (`--embeddings` sends identifier names only). `find` embeds **every unit under the given paths**, so
|
|
64
|
+
> `find ... --embed openai` uploads all of it. The local presets (`gemma`, `minilm`, `qwen3`, `potion`) are meant for a
|
|
65
|
+
> server on loopback (`gemma`: ollama on `127.0.0.1:11434`, the others `127.0.0.1:8099`; `DUPLICATECODE_EMBED_ENDPOINT` overrides it): pointing that
|
|
66
|
+
> variable at a remote host sends text there too. Do not use hosted embeddings on code you may not share.
|
|
60
67
|
|
|
61
68
|
- `--embeddings` embeds identifier *names* (cheap) as an extra name-similarity signal.
|
|
62
|
-
- `--embed
|
|
63
|
-
|
|
69
|
+
- `--embed` is the short form and implies `--embed-code`. A bare `--embed` uses **`gemma`** (embeddinggemma via a local
|
|
70
|
+
[ollama](https://ollama.com): `ollama pull embeddinggemma`), the best model in the codebase benchmark (issue #12: re-implementation
|
|
71
|
+
retrieval R@1 0.94 vs 0.89 for the static detector alone, unchanged when function names are masked). Other presets:
|
|
72
|
+
`--embed=minilm|qwen3|potion|openai|cohere`; MiniLM is smaller but clearly weaker (0.85) and leans on names. Model ids go into the cache key, so vectors of different models never mix.
|
|
64
73
|
- `--embed-code` embeds the *whole text of every unit* (function, class, file, SQL statement) and blends
|
|
65
74
|
the cosine into the score (`--embed-weight`, default 0.35, others scaled by 1 - weight). It finds
|
|
66
75
|
re-implementations that share no tokens. On function-level LLM re-implementations it lifted Python from
|
|
@@ -80,7 +89,8 @@ export OPENAI_BASE_URL=http://localhost:11434/v1 OPENAI_API_KEY=anything OPENAI_
|
|
|
80
89
|
export AZURE_AI_FOUNDRY_ENDPOINT=https://<res>.services.ai.azure.com
|
|
81
90
|
export AZURE_AI_FOUNDRY_API_KEY=... AZURE_AI_FOUNDRY_EMBEDDING_DEPLOYMENT=text-embedding-3-small
|
|
82
91
|
# Azure OpenAI: AZURE_OPENAI_ENDPOINT / AZURE_OPENAI_API_KEY / AZURE_OPENAI_EMBEDDING_DEPLOYMENT
|
|
83
|
-
# Fully local, no key:
|
|
92
|
+
# Fully local, no key, recommended: ollama pull embeddinggemma && duplicatecode scan . --embed (GPU: ~5 ms per unit)
|
|
93
|
+
# Other local models: start the server once (CPU; it loads whichever model a request names)
|
|
84
94
|
# uv run --no-project --python 3.12 eval/embed_server.py
|
|
85
95
|
# duplicatecode scan . --embed minilm # or: qwen3, potion (default endpoint 127.0.0.1:8099)
|
|
86
96
|
# Hosted presets: --embed openai (OPENAI_API_KEY), --embed cohere (COHERE_API_KEY)
|
|
@@ -205,9 +215,9 @@ Self-scans of three internal repositories (Python + TypeScript), 189 pairs judge
|
|
|
205
215
|
before trusting them. A structure-heavy `--profile reimpl` exists for renamed re-implementations
|
|
206
216
|
(use with `--min-name 0`) but has no real-repo precision data yet.
|
|
207
217
|
|
|
208
|
-
### Fresh held-out check (copies profile, threshold 0.6)
|
|
218
|
+
### Fresh held-out check (copies profile, threshold 0.6 = the `copies` default for `scan`)
|
|
209
219
|
|
|
210
|
-
A third, untouched sample of 57 pairs (nothing was tuned on it): 30% true duplicates, 47% incl. partial
|
|
220
|
+
A third, untouched sample of 57 pairs (nothing was tuned on it; the `reimpl` profile has different defaults, 0.35 for `scan`, and is not covered by this sample): 30% true duplicates, 47% incl. partial
|
|
211
221
|
(OneSales 0/24, MDMApp 11/24, CCMT2 6/9 true). The 50%/76% above was optimistic because it was measured on the
|
|
212
222
|
pairs used to choose the weights. Test code is about half of the noise but also holds real copies
|
|
213
223
|
(25% true either way), so it is kept by default; `--skip-tests` drops it.
|
|
@@ -225,7 +235,7 @@ by Haiku and Sonnet without seeing the repo. Fraction where the original is foun
|
|
|
225
235
|
| 0.5 | 45% | 14% | 30% | 1% |
|
|
226
236
|
| 0.6 | 24% | 5% | 14% | 0% |
|
|
227
237
|
|
|
228
|
-
So `diff` (checking new code) defaults to threshold 0.4 while `scan` (existing copies) defaults to 0.6. About half of
|
|
238
|
+
So with the `copies` profile `diff` (checking new code) defaults to threshold 0.4 while `scan` (existing copies) defaults to 0.6; the `reimpl` profile has its own lower defaults (see above). About half of
|
|
229
239
|
independent re-implementations are caught; the rest are genuinely different code. The structure-heavy `reimpl`
|
|
230
240
|
profile is not better on this test.
|
|
231
241
|
|
|
@@ -242,7 +252,7 @@ read/grep, two with this CLI. All distinct groups (82) were then judged blind by
|
|
|
242
252
|
| OneSales, with CLI | 21 | 10 | 48% / 86% | 67% | 9 |
|
|
243
253
|
|
|
244
254
|
Only 18 of 82 groups were found by both, so the approaches are complementary. The agents' reports are capped at 30
|
|
245
|
-
groups, so the raw tool is a better measure of recall: `scan --threshold 0.6` (defaults) finds 38 of the 40 judged
|
|
255
|
+
groups, so the raw tool is a better measure of recall: `scan --threshold 0.6` (the `copies` defaults) finds 38 of the 40 judged
|
|
246
256
|
true duplicates (25/25 MDMApp, 13/15 OneSales) and 23 of the 25 that the LLM-only agents found independently.
|
|
247
257
|
What it still misses: a differently-written picker function (same purpose, different code) and a formatFileSize
|
|
248
258
|
variant with different constants. Fixed after this test: tiny same-name exact copies (`min_tokens` 20 -> 8, near-exact
|
|
@@ -246,6 +246,8 @@ pub fn make_mutation_groups(
|
|
|
246
246
|
for (k, p) in files.iter().skip(skip).take(n).enumerate() {
|
|
247
247
|
let text = std::fs::read_to_string(p)?;
|
|
248
248
|
let dir = out.join(format!("c{:03}", skip + k));
|
|
249
|
+
// a leftover variant of a previous component would be mislabeled as a clone of this one
|
|
250
|
+
let _ = std::fs::remove_dir_all(&dir);
|
|
249
251
|
std::fs::create_dir_all(&dir)?;
|
|
250
252
|
std::fs::write(dir.join("orig.tsx"), &text)?;
|
|
251
253
|
for m in Mutation::ALL {
|
|
@@ -11,11 +11,14 @@ use duplicatecode_engine::{
|
|
|
11
11
|
use std::io::Read;
|
|
12
12
|
use std::path::PathBuf;
|
|
13
13
|
|
|
14
|
-
/// Named embedding setups for `--embed`.
|
|
15
|
-
/// http://127.0.0.1:
|
|
14
|
+
/// Named embedding setups for `--embed`. `gemma` talks to a local ollama (default
|
|
15
|
+
/// http://127.0.0.1:11434/v1); the other local presets talk to `eval/embed_server.py` (default
|
|
16
|
+
/// http://127.0.0.1:8099/v1). DUPLICATECODE_EMBED_ENDPOINT overrides either, loopback hosts only.
|
|
16
17
|
#[derive(Clone, Copy, clap::ValueEnum)]
|
|
17
18
|
enum Preset {
|
|
18
|
-
///
|
|
19
|
+
/// embeddinggemma via local ollama (`ollama pull embeddinggemma`): best in the codebase bake-off, the default of a bare `--embed`
|
|
20
|
+
Gemma,
|
|
21
|
+
/// all-MiniLM-L6-v2, local server: small and fast, but below `gemma` on re-implementation retrieval
|
|
19
22
|
Minilm,
|
|
20
23
|
/// Qwen3-Embedding-0.6B, local: 8k-token context, slower
|
|
21
24
|
Qwen3,
|
|
@@ -30,19 +33,28 @@ enum Preset {
|
|
|
30
33
|
impl Preset {
|
|
31
34
|
fn config(self, dims: Option<u32>) -> Result<duplicatecode_engine::embed::EmbedConfig> {
|
|
32
35
|
use duplicatecode_engine::embed::EmbedConfig;
|
|
33
|
-
let local = |model: &str| {
|
|
36
|
+
let local = |model: &str, default_endpoint: &str| -> Result<EmbedConfig> {
|
|
34
37
|
let endpoint = std::env::var("DUPLICATECODE_EMBED_ENDPOINT")
|
|
35
|
-
.unwrap_or_else(|_|
|
|
38
|
+
.unwrap_or_else(|_| default_endpoint.into());
|
|
36
39
|
// the local server returns full vectors whatever `dimensions` asks for
|
|
37
|
-
EmbedConfig::new(endpoint, model.into(), Some("local".into()), None, None)
|
|
40
|
+
let cfg = EmbedConfig::new(endpoint, model.into(), Some("local".into()), None, None);
|
|
41
|
+
anyhow::ensure!(
|
|
42
|
+
cfg.is_loopback(),
|
|
43
|
+
"the local presets only talk to a server on this machine, but DUPLICATECODE_EMBED_ENDPOINT points at {}; \
|
|
44
|
+
unset it, or use --embed openai|cohere (or --embed-code with OPENAI_API_KEY etc.) to send code to a hosted provider on purpose",
|
|
45
|
+
cfg.host()
|
|
46
|
+
);
|
|
47
|
+
Ok(cfg)
|
|
38
48
|
};
|
|
49
|
+
const SERVER: &str = "http://127.0.0.1:8099/v1";
|
|
39
50
|
let key = |name: &str| {
|
|
40
51
|
std::env::var(name).with_context(|| format!("this preset needs {name} to be set"))
|
|
41
52
|
};
|
|
42
53
|
Ok(match self {
|
|
43
|
-
Preset::
|
|
44
|
-
Preset::
|
|
45
|
-
Preset::
|
|
54
|
+
Preset::Gemma => local("embeddinggemma", "http://127.0.0.1:11434/v1")?,
|
|
55
|
+
Preset::Minilm => local("sentence-transformers/all-MiniLM-L6-v2", SERVER)?,
|
|
56
|
+
Preset::Qwen3 => local("Qwen/Qwen3-Embedding-0.6B", SERVER)?,
|
|
57
|
+
Preset::Potion => local("minishlab/potion-base-8M", SERVER)?,
|
|
46
58
|
Preset::Openai => EmbedConfig::new(
|
|
47
59
|
"https://api.openai.com/v1".into(),
|
|
48
60
|
"text-embedding-3-small".into(),
|
|
@@ -70,7 +82,8 @@ struct EmbedArgs {
|
|
|
70
82
|
#[arg(long)]
|
|
71
83
|
embeddings: bool,
|
|
72
84
|
/// Embed every unit with a named setup (implies --embed-code); see `--help` for the presets.
|
|
73
|
-
|
|
85
|
+
/// A bare `--embed` means `gemma`; write `--embed=minilm` etc. for the others.
|
|
86
|
+
#[arg(long, value_enum, num_args = 0..=1, require_equals = true, default_missing_value = "gemma")]
|
|
74
87
|
embed: Option<Preset>,
|
|
75
88
|
/// Embed the full text of every unit (function, class, file, statement) and blend the cosine of
|
|
76
89
|
/// the two vectors into the score. Finds re-implementations that share no tokens.
|
|
@@ -94,6 +107,11 @@ struct EmbedArgs {
|
|
|
94
107
|
}
|
|
95
108
|
|
|
96
109
|
impl EmbedArgs {
|
|
110
|
+
/// Any embedding feature is on.
|
|
111
|
+
fn enabled(&self) -> bool {
|
|
112
|
+
self.embeddings || self.code_enabled()
|
|
113
|
+
}
|
|
114
|
+
|
|
97
115
|
/// Whole-unit embeddings are on for `--embed-code` and for any `--embed <preset>`.
|
|
98
116
|
fn code_enabled(&self) -> bool {
|
|
99
117
|
self.embed_code || self.embed.is_some()
|
|
@@ -113,6 +131,27 @@ impl EmbedArgs {
|
|
|
113
131
|
return Ok(());
|
|
114
132
|
}
|
|
115
133
|
let cfg = self.config()?;
|
|
134
|
+
// say where the text goes BEFORE anything is sent
|
|
135
|
+
if cfg.is_loopback() {
|
|
136
|
+
eprintln!(
|
|
137
|
+
"embedding {} units via {} (this machine)",
|
|
138
|
+
units.len(),
|
|
139
|
+
cfg.host()
|
|
140
|
+
);
|
|
141
|
+
} else {
|
|
142
|
+
eprintln!(
|
|
143
|
+
"embedding {} units: unit text is sent to {} (model {})",
|
|
144
|
+
units.len(),
|
|
145
|
+
cfg.host(),
|
|
146
|
+
cfg.deployment
|
|
147
|
+
);
|
|
148
|
+
if cfg.is_cleartext_remote() {
|
|
149
|
+
eprintln!(
|
|
150
|
+
"warning: {} is plain http, so your API key and source text travel unencrypted",
|
|
151
|
+
cfg.endpoint
|
|
152
|
+
);
|
|
153
|
+
}
|
|
154
|
+
}
|
|
116
155
|
let path = self
|
|
117
156
|
.embed_cache
|
|
118
157
|
.clone()
|
|
@@ -163,6 +202,30 @@ impl EmbedArgs {
|
|
|
163
202
|
}
|
|
164
203
|
}
|
|
165
204
|
|
|
205
|
+
/// Units under the options' size/kind filters can never be reported, so there is no reason to embed
|
|
206
|
+
/// (upload) them.
|
|
207
|
+
fn retain_matchable(
|
|
208
|
+
units: &mut Vec<duplicatecode_engine::Unit>,
|
|
209
|
+
min_tokens: usize,
|
|
210
|
+
skip_tests: bool,
|
|
211
|
+
) {
|
|
212
|
+
units.retain(|u| u.token_count() >= min_tokens && !u.boilerplate && !(skip_tests && u.is_test));
|
|
213
|
+
}
|
|
214
|
+
|
|
215
|
+
/// Units of every root; with several roots the file names carry the root so equal names stay apart.
|
|
216
|
+
fn load_roots(paths: &[PathBuf], exclude: &[String]) -> Vec<duplicatecode_engine::Unit> {
|
|
217
|
+
let mut units = Vec::new();
|
|
218
|
+
for p in paths {
|
|
219
|
+
for mut u in load_units_with(p, exclude) {
|
|
220
|
+
if paths.len() > 1 {
|
|
221
|
+
u.file = format!("{}/{}", p.display(), u.file);
|
|
222
|
+
}
|
|
223
|
+
units.push(u);
|
|
224
|
+
}
|
|
225
|
+
}
|
|
226
|
+
units
|
|
227
|
+
}
|
|
228
|
+
|
|
166
229
|
#[derive(Clone, Copy, clap::ValueEnum)]
|
|
167
230
|
enum Profile {
|
|
168
231
|
Copies,
|
|
@@ -397,7 +460,7 @@ enum Cmd {
|
|
|
397
460
|
EvalGroups {
|
|
398
461
|
#[arg(long, default_value = "eval/data/codenet")]
|
|
399
462
|
root: PathBuf,
|
|
400
|
-
/// Print only the
|
|
463
|
+
/// Print only the summary lines (`dataset_score`, `dataset_threshold`, `score=`).
|
|
401
464
|
#[arg(long)]
|
|
402
465
|
quiet: bool,
|
|
403
466
|
/// Write every pair's feature vector (TSV) for offline analysis.
|
|
@@ -548,6 +611,10 @@ fn main() -> Result<()> {
|
|
|
548
611
|
}));
|
|
549
612
|
}
|
|
550
613
|
let mut corpus_units = load_units(&repo);
|
|
614
|
+
if embed.enabled() {
|
|
615
|
+
retain_matchable(&mut corpus_units, min_tokens, false);
|
|
616
|
+
retain_matchable(&mut queries, min_tokens, false);
|
|
617
|
+
}
|
|
551
618
|
embed.apply(&mut corpus_units)?;
|
|
552
619
|
embed.apply(&mut queries)?;
|
|
553
620
|
let corpus = Corpus::new(corpus_units);
|
|
@@ -616,6 +683,9 @@ fn main() -> Result<()> {
|
|
|
616
683
|
units.push(u);
|
|
617
684
|
}
|
|
618
685
|
}
|
|
686
|
+
if embed.enabled() {
|
|
687
|
+
units.retain(|u| !u.boilerplate && !(skip_tests && u.is_test));
|
|
688
|
+
}
|
|
619
689
|
embed.apply(&mut units)?;
|
|
620
690
|
let o = review::ReviewOptions {
|
|
621
691
|
threshold,
|
|
@@ -711,10 +781,7 @@ fn main() -> Result<()> {
|
|
|
711
781
|
cross_file,
|
|
712
782
|
json,
|
|
713
783
|
} => {
|
|
714
|
-
let mut units =
|
|
715
|
-
for p in &paths {
|
|
716
|
-
units.extend(load_units_with(p, &exclude));
|
|
717
|
-
}
|
|
784
|
+
let mut units = load_roots(&paths, &exclude);
|
|
718
785
|
units.retain(|u| !u.boilerplate && !(skip_tests && u.is_test));
|
|
719
786
|
let mut found = duplicatecode_engine::fragments::find_fragments(
|
|
720
787
|
&units,
|
|
@@ -760,10 +827,7 @@ fn main() -> Result<()> {
|
|
|
760
827
|
json,
|
|
761
828
|
} => {
|
|
762
829
|
embed.embed_code = true; // searching by description needs unit embeddings
|
|
763
|
-
let mut units =
|
|
764
|
-
for p in &paths {
|
|
765
|
-
units.extend(load_units_with(p, &exclude));
|
|
766
|
-
}
|
|
830
|
+
let mut units = load_roots(&paths, &exclude);
|
|
767
831
|
units.retain(|u| u.token_count() >= min_tokens && !u.boilerplate);
|
|
768
832
|
embed.apply(&mut units)?;
|
|
769
833
|
let cfg = embed.config()?;
|
|
@@ -814,14 +878,9 @@ fn main() -> Result<()> {
|
|
|
814
878
|
} => {
|
|
815
879
|
let threshold =
|
|
816
880
|
threshold.unwrap_or(profile.default_threshold(Command::Scan, embed.code_enabled()));
|
|
817
|
-
let mut units =
|
|
818
|
-
|
|
819
|
-
|
|
820
|
-
if paths.len() > 1 {
|
|
821
|
-
u.file = format!("{}/{}", p.display(), u.file);
|
|
822
|
-
}
|
|
823
|
-
units.push(u);
|
|
824
|
-
}
|
|
881
|
+
let mut units = load_roots(&paths, &exclude);
|
|
882
|
+
if embed.enabled() {
|
|
883
|
+
retain_matchable(&mut units, min_tokens, skip_tests);
|
|
825
884
|
}
|
|
826
885
|
embed.apply(&mut units)?;
|
|
827
886
|
let corpus = Corpus::new(units.clone());
|
|
@@ -851,14 +910,23 @@ fn main() -> Result<()> {
|
|
|
851
910
|
.collect();
|
|
852
911
|
pairs.sort_by(|a, b| b.scores.combined.total_cmp(&a.scores.combined));
|
|
853
912
|
let groups = group_pairs(&pairs);
|
|
854
|
-
|
|
913
|
+
// keyed by the full place: units that start on the same line (minified code, one-line
|
|
914
|
+
// classes) must not be mixed up
|
|
915
|
+
let place = |r: &duplicatecode_engine::index::UnitRef| {
|
|
916
|
+
(r.file.clone(), r.start_line, r.end_line, r.name.clone())
|
|
917
|
+
};
|
|
918
|
+
let by_place: std::collections::HashMap<_, &duplicatecode_engine::Unit> = units
|
|
855
919
|
.iter()
|
|
856
|
-
.map(|u|
|
|
920
|
+
.map(|u| {
|
|
921
|
+
(
|
|
922
|
+
(u.file.clone(), u.start_line, u.end_line, u.name.clone()),
|
|
923
|
+
u,
|
|
924
|
+
)
|
|
925
|
+
})
|
|
857
926
|
.collect();
|
|
858
927
|
let explanation = |m: &duplicatecode_engine::Match| {
|
|
859
|
-
let a = by_place.get(&
|
|
860
|
-
let b =
|
|
861
|
-
by_place.get(&format!("{}:{}", m.candidate.file, m.candidate.start_line))?;
|
|
928
|
+
let a = by_place.get(&place(&m.query))?;
|
|
929
|
+
let b = by_place.get(&place(&m.candidate))?;
|
|
862
930
|
Some(duplicatecode_engine::explain::explain(a, b))
|
|
863
931
|
};
|
|
864
932
|
if pairs_out {
|