duplicatecode 0.2.0__tar.gz → 0.3.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (24) hide show
  1. {duplicatecode-0.2.0 → duplicatecode-0.3.1}/Cargo.lock +15 -4
  2. {duplicatecode-0.2.0 → duplicatecode-0.3.1}/Cargo.toml +1 -1
  3. {duplicatecode-0.2.0 → duplicatecode-0.3.1}/PKG-INFO +142 -7
  4. {duplicatecode-0.2.0 → duplicatecode-0.3.1}/README.md +140 -5
  5. {duplicatecode-0.2.0 → duplicatecode-0.3.1}/crates/duplicatecode-cli/src/bench.rs +4 -1
  6. duplicatecode-0.3.1/crates/duplicatecode-cli/src/groups.rs +269 -0
  7. {duplicatecode-0.2.0 → duplicatecode-0.3.1}/crates/duplicatecode-cli/src/main.rs +446 -34
  8. {duplicatecode-0.2.0 → duplicatecode-0.3.1}/crates/duplicatecode-engine/Cargo.toml +2 -1
  9. duplicatecode-0.3.1/crates/duplicatecode-engine/src/embed.rs +1105 -0
  10. duplicatecode-0.3.1/crates/duplicatecode-engine/src/explain.rs +130 -0
  11. duplicatecode-0.3.1/crates/duplicatecode-engine/src/fragments.rs +201 -0
  12. {duplicatecode-0.2.0 → duplicatecode-0.3.1}/crates/duplicatecode-engine/src/index.rs +340 -36
  13. {duplicatecode-0.2.0 → duplicatecode-0.3.1}/crates/duplicatecode-engine/src/lang.rs +6 -1
  14. {duplicatecode-0.2.0 → duplicatecode-0.3.1}/crates/duplicatecode-engine/src/lib.rs +2 -0
  15. {duplicatecode-0.2.0 → duplicatecode-0.3.1}/crates/duplicatecode-engine/src/similarity.rs +32 -0
  16. {duplicatecode-0.2.0 → duplicatecode-0.3.1}/crates/duplicatecode-engine/src/units.rs +699 -9
  17. {duplicatecode-0.2.0 → duplicatecode-0.3.1}/pyproject.toml +1 -1
  18. duplicatecode-0.2.0/crates/duplicatecode-engine/src/embed.rs +0 -552
  19. {duplicatecode-0.2.0 → duplicatecode-0.3.1}/crates/duplicatecode-cli/Cargo.toml +0 -0
  20. {duplicatecode-0.2.0 → duplicatecode-0.3.1}/crates/duplicatecode-cli/src/review.rs +0 -0
  21. {duplicatecode-0.2.0 → duplicatecode-0.3.1}/crates/duplicatecode-engine/src/diff.rs +0 -0
  22. {duplicatecode-0.2.0 → duplicatecode-0.3.1}/crates/duplicatecode-engine/src/fingerprint.rs +0 -0
  23. {duplicatecode-0.2.0 → duplicatecode-0.3.1}/crates/duplicatecode-engine/src/mutate.rs +0 -0
  24. {duplicatecode-0.2.0 → duplicatecode-0.3.1}/crates/duplicatecode-engine/src/naming.rs +0 -0
@@ -91,9 +91,9 @@ dependencies = [
91
91
 
92
92
  [[package]]
93
93
  name = "cc"
94
- version = "1.5.1"
94
+ version = "1.2.67"
95
95
  source = "registry+https://github.com/rust-lang/crates.io-index"
96
- checksum = "f360145194ee8e21db5ee7f3fcd4fe52210864c75c985dae33218202c8bbe040"
96
+ checksum = "e17dd265a7d0f31ef544e1b20e03add05d3b45b491b633b10d67145d2acc1a38"
97
97
  dependencies = [
98
98
  "find-msvc-tools",
99
99
  "shlex",
@@ -198,7 +198,7 @@ dependencies = [
198
198
 
199
199
  [[package]]
200
200
  name = "duplicatecode"
201
- version = "0.2.0"
201
+ version = "0.3.1"
202
202
  dependencies = [
203
203
  "anyhow",
204
204
  "clap",
@@ -212,7 +212,7 @@ dependencies = [
212
212
 
213
213
  [[package]]
214
214
  name = "duplicatecode-engine"
215
- version = "0.2.0"
215
+ version = "0.3.1"
216
216
  dependencies = [
217
217
  "ignore",
218
218
  "serde",
@@ -220,6 +220,7 @@ dependencies = [
220
220
  "tree-sitter",
221
221
  "tree-sitter-c-sharp",
222
222
  "tree-sitter-python",
223
+ "tree-sitter-sequel",
223
224
  "tree-sitter-typescript",
224
225
  "ureq",
225
226
  "walkdir",
@@ -790,6 +791,16 @@ dependencies = [
790
791
  "tree-sitter-language",
791
792
  ]
792
793
 
794
+ [[package]]
795
+ name = "tree-sitter-sequel"
796
+ version = "0.3.11"
797
+ source = "registry+https://github.com/rust-lang/crates.io-index"
798
+ checksum = "9d198ad3c319c02e43c21efa1ec796b837afcb96ffaef1a40c1978fbdcec7d17"
799
+ dependencies = [
800
+ "cc",
801
+ "tree-sitter-language",
802
+ ]
803
+
793
804
  [[package]]
794
805
  name = "tree-sitter-typescript"
795
806
  version = "0.23.2"
@@ -3,6 +3,6 @@ resolver = "2"
3
3
  members = ["crates/duplicatecode-engine", "crates/duplicatecode-cli"]
4
4
 
5
5
  [workspace.package]
6
- version = "0.2.0"
6
+ version = "0.3.1"
7
7
  edition = "2021"
8
8
  license = "MIT"
@@ -1,16 +1,16 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: duplicatecode
3
- Version: 0.2.0
3
+ Version: 0.3.1
4
4
  Classifier: Programming Language :: Rust
5
5
  Classifier: Environment :: Console
6
- Summary: Static (LLM-free) detection of duplicate/similar code in Python, TypeScript and C#
6
+ Summary: Static (LLM-free) detection of duplicate/similar code in Python, TypeScript/JavaScript, SQL and C#
7
7
  License: MIT
8
8
  Requires-Python: >=3.9
9
9
  Description-Content-Type: text/markdown; charset=UTF-8; variant=GFM
10
10
 
11
11
  # duplicatecode
12
12
 
13
- Static (LLM-free) detection of duplicate / similar code in Python, TypeScript (TSX) and C#,
13
+ Static by default (LLM-free, no network unless `--embed*` is used) detection of duplicate / similar code in Python, TypeScript/JavaScript (TSX/JSX), SQL and C#,
14
14
  aimed at catching an LLM re-implementing something that already exists. Input can be a git diff
15
15
  checked against existing source.
16
16
 
@@ -42,6 +42,141 @@ duplicatecode scan . --pairs --json # machine-readable pairs in
42
42
  duplicatecode bench --dataset dataset [--file-level] [--mutations] [--negatives <other repo>]
43
43
  ```
44
44
 
45
+ ## More ways to look
46
+
47
+ ```sh
48
+ duplicatecode fragments packages/ # copied blocks inside different functions (4+ identical statements)
49
+ duplicatecode scan . --pairs --explain # say what differs: literals, calls, lines (what to parameterize)
50
+ duplicatecode find "retry with exponential backoff" src/ # does something like this already exist? (needs embeddings)
51
+ ```
52
+
53
+ - `fragments` indexes windows of consecutive normalized statements, so a block pasted into another
54
+ function is found even when the surrounding functions differ. On an injection benchmark (renamed
55
+ blocks of real functions pasted into others) it finds 93-97% of blocks of 6+ statements; see
56
+ `eval/REPORT.md`. Tune with `--min-stmts` / `--min-tokens`; constructors and dunders are ignored.
57
+ - `--explain` adds, per pair, the literals and calls only one side has and the source lines with no
58
+ counterpart ("16 of 18/17 statements shared").
59
+ - `find` embeds the query and every unit and ranks by similarity. With MiniLM, 50 task descriptions
60
+ against 168 implementation units rank the right one first 94% of the time; with 1,379 unrelated real
61
+ units mixed in, 82% first and 96% in the top 3 (every query within the top 10).
62
+ - Default thresholds depend on the profile: `copies` keeps its tuned 0.6 (`scan`), 0.4 (`diff`), 0.45
63
+ (`review`); `reimpl` scores live lower, so it defaults to 0.35 / 0.28 / 0.30, which correspond to roughly
64
+ 0.1% / 1% false-positive rates on unrelated code in the benchmarks. With `--embed` the `reimpl`
65
+ defaults rise by 0.07 because blended scores of unrelated code rise too.
66
+
67
+ ## Embeddings (bring your own key)
68
+
69
+ Two optional signals, both off by default (the detector stays static and offline unless you ask).
70
+
71
+ > **Privacy warning.** `--embed openai`, `--embed cohere`, and `--embed-code` / `--embeddings` pointed at any
72
+ > hosted endpoint (OpenAI, Cohere, Azure, ...) **upload the source text of your units** to that provider
73
+ > (`--embeddings` sends identifier names only). `find` embeds **every unit under the given paths**, so
74
+ > `find ... --embed openai` uploads all of it. The local presets (`minilm`, `qwen3`, `potion`) are meant for a
75
+ > server on loopback (default `127.0.0.1:8099`; `DUPLICATECODE_EMBED_ENDPOINT` overrides it): pointing that
76
+ > variable at a remote host sends text there too. Do not use hosted embeddings on code you may not share.
77
+
78
+ - `--embeddings` embeds identifier *names* (cheap) as an extra name-similarity signal.
79
+ - `--embed <preset>` is the short form (`minilm`, `qwen3`, `potion`, `openai`, `cohere`; `qwen3` is about 5x slower than `minilm` on CPU, so use it with a GPU or prefer `minilm`) and implies
80
+ `--embed-code`. Model ids go into the cache key, so vectors of different models never mix.
81
+ - `--embed-code` embeds the *whole text of every unit* (function, class, file, SQL statement) and blends
82
+ the cosine into the score (`--embed-weight`, default 0.35, others scaled by 1 - weight). It finds
83
+ re-implementations that share no tokens. On function-level LLM re-implementations it lifted Python from
84
+ 0.70 to 0.89 on a held-out half; see `eval/REPORT.md` for the numbers and caveats. Vectors are cached per
85
+ unit text in `~/.cache/duplicatecode/embeddings.bin`, so a rescan only embeds what changed.
86
+
87
+ Credentials come from the environment:
88
+
89
+ ```sh
90
+ # OpenAI
91
+ export OPENAI_API_KEY=sk-... # optional: OPENAI_EMBEDDING_MODEL (default text-embedding-3-small)
92
+ # Cohere (native /v2/embed)
93
+ export COHERE_API_KEY=... # optional: COHERE_EMBEDDING_MODEL (default embed-v4.0)
94
+ # OpenAI-compatible server (vLLM, Ollama, LiteLLM, ...)
95
+ export OPENAI_BASE_URL=http://localhost:11434/v1 OPENAI_API_KEY=anything OPENAI_EMBEDDING_MODEL=nomic-embed-text
96
+ # Azure AI Foundry (key, or `az login` if no key is set)
97
+ export AZURE_AI_FOUNDRY_ENDPOINT=https://<res>.services.ai.azure.com
98
+ export AZURE_AI_FOUNDRY_API_KEY=... AZURE_AI_FOUNDRY_EMBEDDING_DEPLOYMENT=text-embedding-3-small
99
+ # Azure OpenAI: AZURE_OPENAI_ENDPOINT / AZURE_OPENAI_API_KEY / AZURE_OPENAI_EMBEDDING_DEPLOYMENT
100
+ # Fully local, no key: start the server once (CPU; it loads whichever model a request names)
101
+ # uv run --no-project --python 3.12 eval/embed_server.py
102
+ # duplicatecode scan . --embed minilm # or: qwen3, potion (default endpoint 127.0.0.1:8099)
103
+ # Hosted presets: --embed openai (OPENAI_API_KEY), --embed cohere (COHERE_API_KEY)
104
+
105
+ duplicatecode embed-test fetchUser getUser # check credentials
106
+ duplicatecode scan . --embed-code # whole-unit embeddings
107
+ duplicatecode scan . --embeddings # names only
108
+ ```
109
+
110
+ ## API
111
+
112
+ ### Command line
113
+
114
+ | command | what it does |
115
+ | --- | --- |
116
+ | `scan <paths>` | similar units inside one or more folders; `--pairs`, `--explain`, `--json`, `--fail-on-found` |
117
+ | `diff` | check code added in a git diff (stdin or `--diff`) against the repo (`--repo`) |
118
+ | `review` | ranked candidates for a diff, for a human or an agent to judge |
119
+ | `fragments <paths>` | copied blocks of identical statements inside different functions |
120
+ | `find "<description>" <paths>` | existing units closest to a plain-language description (needs embeddings) |
121
+ | `units <path>` / `show file:10-40` | list extracted units / print a unit's source |
122
+ | `embed-test <names...>` | check the embedding provider and calibrate `--embed-floor` |
123
+ | `eval-groups`, `bench`, `make-mutation-groups` | evaluation (see `eval/README.md`) |
124
+
125
+ Shared options: `--profile copies|reimpl`, `--threshold`, `--min-tokens`, `--min-lines`, `--exclude`, `--skip-tests`,
126
+ `--cross-file`, `--json`. Embedding options (all off by default): `--embed <preset>`, `--embed-code`,
127
+ `--embed-weight`, `--embed-max-chars`, `--embeddings`, `--embed-cache`, `--embed-dims`.
128
+
129
+ ```sh
130
+ duplicatecode scan . --profile reimpl --min-name 0 --embed minilm --pairs --explain
131
+ duplicatecode fragments src/ --min-stmts 4 --min-tokens 30 --cross-file --json
132
+ duplicatecode find "retry with exponential backoff" src/ --top 5 --json
133
+ duplicatecode diff --repo . < change.diff --embed openai
134
+ ```
135
+
136
+ JSON output:
137
+
138
+ - `find --json`: `[{score, file, name, kind, start_line, end_line}]`
139
+ - `fragments --json`: `[{a, b, statements, tokens}]` with `a`/`b` = `{file, unit, start_line, end_line, coverage}`
140
+ - `scan --pairs --json`: `[{query, candidate, scores}]`; with `--explain`: `[{match, explanation}]`
141
+ - `scan --json` (groups): `[{score, units: [{file, name, kind, start_line, end_line}]}]`
142
+
143
+ `scan --fail-on-found` exits with status 1 when anything is reported, for CI.
144
+
145
+ ### Rust library
146
+
147
+ The engine crate `duplicatecode-engine` is what the CLI uses:
148
+
149
+ ```rust
150
+ use duplicatecode_engine::embed::{embed_unit_code, EmbedConfig, EmbeddingCache};
151
+ use duplicatecode_engine::explain::explain;
152
+ use duplicatecode_engine::fragments::{find_fragments, FragmentOptions};
153
+ use duplicatecode_engine::index::Weights;
154
+ use duplicatecode_engine::{find_matches, load_units_with, Corpus, MatchOptions};
155
+
156
+ let mut units = load_units_with(Path::new("src/"), &[]);
157
+
158
+ // optional: embed every unit; units without a vector simply skip the embedding term
159
+ let cfg = EmbedConfig::from_env(None).ok_or("no embedding provider configured")?;
160
+ let mut cache = EmbeddingCache::load(&EmbeddingCache::default_path());
161
+ embed_unit_code(&mut units, &cfg, &mut cache, 3000)?;
162
+
163
+ // pairs: queries against a corpus
164
+ let weights = Weights::default().with_embed(0.35); // 0.0 = static only
165
+ let corpus = Corpus::new(units.clone());
166
+ let pairs = find_matches(&units, &corpus, MatchOptions { weights, threshold: 0.42, ..Default::default() });
167
+
168
+ // copied blocks, and what differs inside a pair
169
+ let blocks = find_fragments(&units, FragmentOptions::default());
170
+ println!("{}", explain(&units[0], &units[1]).summary());
171
+ ```
172
+
173
+ ### Not built yet
174
+
175
+ Planned, not available today: a `.duplicatecode.toml` with an `[embed]` section; `--embed-optional` (warn and
176
+ fall back to the static score when the provider is unreachable; today an unreachable provider is an error);
177
+ `--allow-upload` (required for the hosted presets, which send source text out); a `duplicatecode[embed]` Python
178
+ extra that starts the local server on demand; an `Embedder` trait so providers can plug in without HTTP.
179
+
45
180
  ## How it works
46
181
 
47
182
  Units (functions, methods, classes, arrow-function components) are extracted with tree-sitter and
@@ -87,9 +222,9 @@ Self-scans of three internal repositories (Python + TypeScript), 189 pairs judge
87
222
  before trusting them. A structure-heavy `--profile reimpl` exists for renamed re-implementations
88
223
  (use with `--min-name 0`) but has no real-repo precision data yet.
89
224
 
90
- ### Fresh held-out check (copies profile, threshold 0.6)
225
+ ### Fresh held-out check (copies profile, threshold 0.6 = the `copies` default for `scan`)
91
226
 
92
- A third, untouched sample of 57 pairs (nothing was tuned on it): 30% true duplicates, 47% incl. partial
227
+ A third, untouched sample of 57 pairs (nothing was tuned on it; the `reimpl` profile has different defaults, 0.35 for `scan`, and is not covered by this sample): 30% true duplicates, 47% incl. partial
93
228
  (OneSales 0/24, MDMApp 11/24, CCMT2 6/9 true). The 50%/76% above was optimistic because it was measured on the
94
229
  pairs used to choose the weights. Test code is about half of the noise but also holds real copies
95
230
  (25% true either way), so it is kept by default; `--skip-tests` drops it.
@@ -107,7 +242,7 @@ by Haiku and Sonnet without seeing the repo. Fraction where the original is foun
107
242
  | 0.5 | 45% | 14% | 30% | 1% |
108
243
  | 0.6 | 24% | 5% | 14% | 0% |
109
244
 
110
- So `diff` (checking new code) defaults to threshold 0.4 while `scan` (existing copies) defaults to 0.6. About half of
245
+ So with the `copies` profile `diff` (checking new code) defaults to threshold 0.4 while `scan` (existing copies) defaults to 0.6; the `reimpl` profile has its own lower defaults (see above). About half of
111
246
  independent re-implementations are caught; the rest are genuinely different code. The structure-heavy `reimpl`
112
247
  profile is not better on this test.
113
248
 
@@ -124,7 +259,7 @@ read/grep, two with this CLI. All distinct groups (82) were then judged blind by
124
259
  | OneSales, with CLI | 21 | 10 | 48% / 86% | 67% | 9 |
125
260
 
126
261
  Only 18 of 82 groups were found by both, so the approaches are complementary. The agents' reports are capped at 30
127
- groups, so the raw tool is a better measure of recall: `scan --threshold 0.6` (defaults) finds 38 of the 40 judged
262
+ groups, so the raw tool is a better measure of recall: `scan --threshold 0.6` (the `copies` defaults) finds 38 of the 40 judged
128
263
  true duplicates (25/25 MDMApp, 13/15 OneSales) and 23 of the 25 that the LLM-only agents found independently.
129
264
  What it still misses: a differently-written picker function (same purpose, different code) and a formatFileSize
130
265
  variant with different constants. Fixed after this test: tiny same-name exact copies (`min_tokens` 20 -> 8, near-exact
@@ -1,6 +1,6 @@
1
1
  # duplicatecode
2
2
 
3
- Static (LLM-free) detection of duplicate / similar code in Python, TypeScript (TSX) and C#,
3
+ Static by default (LLM-free, no network unless `--embed*` is used) detection of duplicate / similar code in Python, TypeScript/JavaScript (TSX/JSX), SQL and C#,
4
4
  aimed at catching an LLM re-implementing something that already exists. Input can be a git diff
5
5
  checked against existing source.
6
6
 
@@ -32,6 +32,141 @@ duplicatecode scan . --pairs --json # machine-readable pairs in
32
32
  duplicatecode bench --dataset dataset [--file-level] [--mutations] [--negatives <other repo>]
33
33
  ```
34
34
 
35
+ ## More ways to look
36
+
37
+ ```sh
38
+ duplicatecode fragments packages/ # copied blocks inside different functions (4+ identical statements)
39
+ duplicatecode scan . --pairs --explain # say what differs: literals, calls, lines (what to parameterize)
40
+ duplicatecode find "retry with exponential backoff" src/ # does something like this already exist? (needs embeddings)
41
+ ```
42
+
43
+ - `fragments` indexes windows of consecutive normalized statements, so a block pasted into another
44
+ function is found even when the surrounding functions differ. On an injection benchmark (renamed
45
+ blocks of real functions pasted into others) it finds 93-97% of blocks of 6+ statements; see
46
+ `eval/REPORT.md`. Tune with `--min-stmts` / `--min-tokens`; constructors and dunders are ignored.
47
+ - `--explain` adds, per pair, the literals and calls only one side has and the source lines with no
48
+ counterpart ("16 of 18/17 statements shared").
49
+ - `find` embeds the query and every unit and ranks by similarity. With MiniLM, 50 task descriptions
50
+ against 168 implementation units rank the right one first 94% of the time; with 1,379 unrelated real
51
+ units mixed in, 82% first and 96% in the top 3 (every query within the top 10).
52
+ - Default thresholds depend on the profile: `copies` keeps its tuned 0.6 (`scan`), 0.4 (`diff`), 0.45
53
+ (`review`); `reimpl` scores live lower, so it defaults to 0.35 / 0.28 / 0.30, which correspond to roughly
54
+ 0.1% / 1% false-positive rates on unrelated code in the benchmarks. With `--embed` the `reimpl`
55
+ defaults rise by 0.07 because blended scores of unrelated code rise too.
56
+
57
+ ## Embeddings (bring your own key)
58
+
59
+ Two optional signals, both off by default (the detector stays static and offline unless you ask).
60
+
61
+ > **Privacy warning.** `--embed openai`, `--embed cohere`, and `--embed-code` / `--embeddings` pointed at any
62
+ > hosted endpoint (OpenAI, Cohere, Azure, ...) **upload the source text of your units** to that provider
63
+ > (`--embeddings` sends identifier names only). `find` embeds **every unit under the given paths**, so
64
+ > `find ... --embed openai` uploads all of it. The local presets (`minilm`, `qwen3`, `potion`) are meant for a
65
+ > server on loopback (default `127.0.0.1:8099`; `DUPLICATECODE_EMBED_ENDPOINT` overrides it): pointing that
66
+ > variable at a remote host sends text there too. Do not use hosted embeddings on code you may not share.
67
+
68
+ - `--embeddings` embeds identifier *names* (cheap) as an extra name-similarity signal.
69
+ - `--embed <preset>` is the short form (`minilm`, `qwen3`, `potion`, `openai`, `cohere`; `qwen3` is about 5x slower than `minilm` on CPU, so use it with a GPU or prefer `minilm`) and implies
70
+ `--embed-code`. Model ids go into the cache key, so vectors of different models never mix.
71
+ - `--embed-code` embeds the *whole text of every unit* (function, class, file, SQL statement) and blends
72
+ the cosine into the score (`--embed-weight`, default 0.35, others scaled by 1 - weight). It finds
73
+ re-implementations that share no tokens. On function-level LLM re-implementations it lifted Python from
74
+ 0.70 to 0.89 on a held-out half; see `eval/REPORT.md` for the numbers and caveats. Vectors are cached per
75
+ unit text in `~/.cache/duplicatecode/embeddings.bin`, so a rescan only embeds what changed.
76
+
77
+ Credentials come from the environment:
78
+
79
+ ```sh
80
+ # OpenAI
81
+ export OPENAI_API_KEY=sk-... # optional: OPENAI_EMBEDDING_MODEL (default text-embedding-3-small)
82
+ # Cohere (native /v2/embed)
83
+ export COHERE_API_KEY=... # optional: COHERE_EMBEDDING_MODEL (default embed-v4.0)
84
+ # OpenAI-compatible server (vLLM, Ollama, LiteLLM, ...)
85
+ export OPENAI_BASE_URL=http://localhost:11434/v1 OPENAI_API_KEY=anything OPENAI_EMBEDDING_MODEL=nomic-embed-text
86
+ # Azure AI Foundry (key, or `az login` if no key is set)
87
+ export AZURE_AI_FOUNDRY_ENDPOINT=https://<res>.services.ai.azure.com
88
+ export AZURE_AI_FOUNDRY_API_KEY=... AZURE_AI_FOUNDRY_EMBEDDING_DEPLOYMENT=text-embedding-3-small
89
+ # Azure OpenAI: AZURE_OPENAI_ENDPOINT / AZURE_OPENAI_API_KEY / AZURE_OPENAI_EMBEDDING_DEPLOYMENT
90
+ # Fully local, no key: start the server once (CPU; it loads whichever model a request names)
91
+ # uv run --no-project --python 3.12 eval/embed_server.py
92
+ # duplicatecode scan . --embed minilm # or: qwen3, potion (default endpoint 127.0.0.1:8099)
93
+ # Hosted presets: --embed openai (OPENAI_API_KEY), --embed cohere (COHERE_API_KEY)
94
+
95
+ duplicatecode embed-test fetchUser getUser # check credentials
96
+ duplicatecode scan . --embed-code # whole-unit embeddings
97
+ duplicatecode scan . --embeddings # names only
98
+ ```
99
+
100
+ ## API
101
+
102
+ ### Command line
103
+
104
+ | command | what it does |
105
+ | --- | --- |
106
+ | `scan <paths>` | similar units inside one or more folders; `--pairs`, `--explain`, `--json`, `--fail-on-found` |
107
+ | `diff` | check code added in a git diff (stdin or `--diff`) against the repo (`--repo`) |
108
+ | `review` | ranked candidates for a diff, for a human or an agent to judge |
109
+ | `fragments <paths>` | copied blocks of identical statements inside different functions |
110
+ | `find "<description>" <paths>` | existing units closest to a plain-language description (needs embeddings) |
111
+ | `units <path>` / `show file:10-40` | list extracted units / print a unit's source |
112
+ | `embed-test <names...>` | check the embedding provider and calibrate `--embed-floor` |
113
+ | `eval-groups`, `bench`, `make-mutation-groups` | evaluation (see `eval/README.md`) |
114
+
115
+ Shared options: `--profile copies|reimpl`, `--threshold`, `--min-tokens`, `--min-lines`, `--exclude`, `--skip-tests`,
116
+ `--cross-file`, `--json`. Embedding options (all off by default): `--embed <preset>`, `--embed-code`,
117
+ `--embed-weight`, `--embed-max-chars`, `--embeddings`, `--embed-cache`, `--embed-dims`.
118
+
119
+ ```sh
120
+ duplicatecode scan . --profile reimpl --min-name 0 --embed minilm --pairs --explain
121
+ duplicatecode fragments src/ --min-stmts 4 --min-tokens 30 --cross-file --json
122
+ duplicatecode find "retry with exponential backoff" src/ --top 5 --json
123
+ duplicatecode diff --repo . < change.diff --embed openai
124
+ ```
125
+
126
+ JSON output:
127
+
128
+ - `find --json`: `[{score, file, name, kind, start_line, end_line}]`
129
+ - `fragments --json`: `[{a, b, statements, tokens}]` with `a`/`b` = `{file, unit, start_line, end_line, coverage}`
130
+ - `scan --pairs --json`: `[{query, candidate, scores}]`; with `--explain`: `[{match, explanation}]`
131
+ - `scan --json` (groups): `[{score, units: [{file, name, kind, start_line, end_line}]}]`
132
+
133
+ `scan --fail-on-found` exits with status 1 when anything is reported, for CI.
134
+
135
+ ### Rust library
136
+
137
+ The engine crate `duplicatecode-engine` is what the CLI uses:
138
+
139
+ ```rust
140
+ use duplicatecode_engine::embed::{embed_unit_code, EmbedConfig, EmbeddingCache};
141
+ use duplicatecode_engine::explain::explain;
142
+ use duplicatecode_engine::fragments::{find_fragments, FragmentOptions};
143
+ use duplicatecode_engine::index::Weights;
144
+ use duplicatecode_engine::{find_matches, load_units_with, Corpus, MatchOptions};
145
+
146
+ let mut units = load_units_with(Path::new("src/"), &[]);
147
+
148
+ // optional: embed every unit; units without a vector simply skip the embedding term
149
+ let cfg = EmbedConfig::from_env(None).ok_or("no embedding provider configured")?;
150
+ let mut cache = EmbeddingCache::load(&EmbeddingCache::default_path());
151
+ embed_unit_code(&mut units, &cfg, &mut cache, 3000)?;
152
+
153
+ // pairs: queries against a corpus
154
+ let weights = Weights::default().with_embed(0.35); // 0.0 = static only
155
+ let corpus = Corpus::new(units.clone());
156
+ let pairs = find_matches(&units, &corpus, MatchOptions { weights, threshold: 0.42, ..Default::default() });
157
+
158
+ // copied blocks, and what differs inside a pair
159
+ let blocks = find_fragments(&units, FragmentOptions::default());
160
+ println!("{}", explain(&units[0], &units[1]).summary());
161
+ ```
162
+
163
+ ### Not built yet
164
+
165
+ Planned, not available today: a `.duplicatecode.toml` with an `[embed]` section; `--embed-optional` (warn and
166
+ fall back to the static score when the provider is unreachable; today an unreachable provider is an error);
167
+ `--allow-upload` (required for the hosted presets, which send source text out); a `duplicatecode[embed]` Python
168
+ extra that starts the local server on demand; an `Embedder` trait so providers can plug in without HTTP.
169
+
35
170
  ## How it works
36
171
 
37
172
  Units (functions, methods, classes, arrow-function components) are extracted with tree-sitter and
@@ -77,9 +212,9 @@ Self-scans of three internal repositories (Python + TypeScript), 189 pairs judge
77
212
  before trusting them. A structure-heavy `--profile reimpl` exists for renamed re-implementations
78
213
  (use with `--min-name 0`) but has no real-repo precision data yet.
79
214
 
80
- ### Fresh held-out check (copies profile, threshold 0.6)
215
+ ### Fresh held-out check (copies profile, threshold 0.6 = the `copies` default for `scan`)
81
216
 
82
- A third, untouched sample of 57 pairs (nothing was tuned on it): 30% true duplicates, 47% incl. partial
217
+ A third, untouched sample of 57 pairs (nothing was tuned on it; the `reimpl` profile has different defaults, 0.35 for `scan`, and is not covered by this sample): 30% true duplicates, 47% incl. partial
83
218
  (OneSales 0/24, MDMApp 11/24, CCMT2 6/9 true). The 50%/76% above was optimistic because it was measured on the
84
219
  pairs used to choose the weights. Test code is about half of the noise but also holds real copies
85
220
  (25% true either way), so it is kept by default; `--skip-tests` drops it.
@@ -97,7 +232,7 @@ by Haiku and Sonnet without seeing the repo. Fraction where the original is foun
97
232
  | 0.5 | 45% | 14% | 30% | 1% |
98
233
  | 0.6 | 24% | 5% | 14% | 0% |
99
234
 
100
- So `diff` (checking new code) defaults to threshold 0.4 while `scan` (existing copies) defaults to 0.6. About half of
235
+ So with the `copies` profile `diff` (checking new code) defaults to threshold 0.4 while `scan` (existing copies) defaults to 0.6; the `reimpl` profile has its own lower defaults (see above). About half of
101
236
  independent re-implementations are caught; the rest are genuinely different code. The structure-heavy `reimpl`
102
237
  profile is not better on this test.
103
238
 
@@ -114,7 +249,7 @@ read/grep, two with this CLI. All distinct groups (82) were then judged blind by
114
249
  | OneSales, with CLI | 21 | 10 | 48% / 86% | 67% | 9 |
115
250
 
116
251
  Only 18 of 82 groups were found by both, so the approaches are complementary. The agents' reports are capped at 30
117
- groups, so the raw tool is a better measure of recall: `scan --threshold 0.6` (defaults) finds 38 of the 40 judged
252
+ groups, so the raw tool is a better measure of recall: `scan --threshold 0.6` (the `copies` defaults) finds 38 of the 40 judged
118
253
  true duplicates (25/25 MDMApp, 13/15 OneSales) and 23 of the 25 that the LLM-only agents found independently.
119
254
  What it still misses: a differently-written picker function (same purpose, different code) and a formatFileSize
120
255
  variant with different constants. Fixed after this test: tiny same-name exact copies (`min_tokens` 20 -> 8, near-exact
@@ -389,6 +389,9 @@ fn fit(samples: &[&Sample]) -> Weights {
389
389
  let mut w = Weights {
390
390
  bias: 0.0,
391
391
  w: [0.0; N_FEATURES],
392
+ sql: None,
393
+ file: None,
394
+ renormalize: false,
392
395
  name_floor: 0.5,
393
396
  };
394
397
  for _ in 0..2500 {
@@ -415,7 +418,7 @@ fn apply(w: &Weights, x: &[f64; N_FEATURES]) -> f64 {
415
418
  }
416
419
 
417
420
  /// Fraction of positives scoring above the (1 - fpr) quantile of the negatives.
418
- fn tpr_at_fpr(scores: &[(f64, bool)], fpr: f64) -> f64 {
421
+ pub(crate) fn tpr_at_fpr(scores: &[(f64, bool)], fpr: f64) -> f64 {
419
422
  let mut neg: Vec<f64> = scores.iter().filter(|s| !s.1).map(|s| s.0).collect();
420
423
  let pos: Vec<f64> = scores.iter().filter(|s| s.1).map(|s| s.0).collect();
421
424
  if neg.is_empty() || pos.is_empty() {