duplicatecode 0.1.0__tar.gz → 0.3.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (23) hide show
  1. {duplicatecode-0.1.0 → duplicatecode-0.3.0}/Cargo.lock +26 -4
  2. {duplicatecode-0.1.0 → duplicatecode-0.3.0}/Cargo.toml +1 -1
  3. {duplicatecode-0.1.0 → duplicatecode-0.3.0}/PKG-INFO +131 -3
  4. {duplicatecode-0.1.0 → duplicatecode-0.3.0}/README.md +129 -1
  5. {duplicatecode-0.1.0 → duplicatecode-0.3.0}/crates/duplicatecode-cli/src/bench.rs +4 -1
  6. duplicatecode-0.3.0/crates/duplicatecode-cli/src/groups.rs +267 -0
  7. {duplicatecode-0.1.0 → duplicatecode-0.3.0}/crates/duplicatecode-cli/src/main.rs +377 -27
  8. {duplicatecode-0.1.0 → duplicatecode-0.3.0}/crates/duplicatecode-engine/Cargo.toml +3 -1
  9. {duplicatecode-0.1.0 → duplicatecode-0.3.0}/crates/duplicatecode-engine/src/embed.rs +259 -35
  10. duplicatecode-0.3.0/crates/duplicatecode-engine/src/explain.rs +130 -0
  11. duplicatecode-0.3.0/crates/duplicatecode-engine/src/fragments.rs +201 -0
  12. {duplicatecode-0.1.0 → duplicatecode-0.3.0}/crates/duplicatecode-engine/src/index.rs +166 -23
  13. {duplicatecode-0.1.0 → duplicatecode-0.3.0}/crates/duplicatecode-engine/src/lang.rs +12 -3
  14. {duplicatecode-0.1.0 → duplicatecode-0.3.0}/crates/duplicatecode-engine/src/lib.rs +2 -0
  15. {duplicatecode-0.1.0 → duplicatecode-0.3.0}/crates/duplicatecode-engine/src/similarity.rs +32 -0
  16. {duplicatecode-0.1.0 → duplicatecode-0.3.0}/crates/duplicatecode-engine/src/units.rs +701 -22
  17. {duplicatecode-0.1.0 → duplicatecode-0.3.0}/pyproject.toml +1 -1
  18. {duplicatecode-0.1.0 → duplicatecode-0.3.0}/crates/duplicatecode-cli/Cargo.toml +0 -0
  19. {duplicatecode-0.1.0 → duplicatecode-0.3.0}/crates/duplicatecode-cli/src/review.rs +0 -0
  20. {duplicatecode-0.1.0 → duplicatecode-0.3.0}/crates/duplicatecode-engine/src/diff.rs +0 -0
  21. {duplicatecode-0.1.0 → duplicatecode-0.3.0}/crates/duplicatecode-engine/src/fingerprint.rs +0 -0
  22. {duplicatecode-0.1.0 → duplicatecode-0.3.0}/crates/duplicatecode-engine/src/mutate.rs +0 -0
  23. {duplicatecode-0.1.0 → duplicatecode-0.3.0}/crates/duplicatecode-engine/src/naming.rs +0 -0
@@ -91,9 +91,9 @@ dependencies = [
91
91
 
92
92
  [[package]]
93
93
  name = "cc"
94
- version = "1.5.1"
94
+ version = "1.2.67"
95
95
  source = "registry+https://github.com/rust-lang/crates.io-index"
96
- checksum = "f360145194ee8e21db5ee7f3fcd4fe52210864c75c985dae33218202c8bbe040"
96
+ checksum = "e17dd265a7d0f31ef544e1b20e03add05d3b45b491b633b10d67145d2acc1a38"
97
97
  dependencies = [
98
98
  "find-msvc-tools",
99
99
  "shlex",
@@ -198,7 +198,7 @@ dependencies = [
198
198
 
199
199
  [[package]]
200
200
  name = "duplicatecode"
201
- version = "0.1.0"
201
+ version = "0.3.0"
202
202
  dependencies = [
203
203
  "anyhow",
204
204
  "clap",
@@ -212,13 +212,15 @@ dependencies = [
212
212
 
213
213
  [[package]]
214
214
  name = "duplicatecode-engine"
215
- version = "0.1.0"
215
+ version = "0.3.0"
216
216
  dependencies = [
217
217
  "ignore",
218
218
  "serde",
219
219
  "serde_json",
220
220
  "tree-sitter",
221
+ "tree-sitter-c-sharp",
221
222
  "tree-sitter-python",
223
+ "tree-sitter-sequel",
222
224
  "tree-sitter-typescript",
223
225
  "ureq",
224
226
  "walkdir",
@@ -763,6 +765,16 @@ dependencies = [
763
765
  "tree-sitter-language",
764
766
  ]
765
767
 
768
+ [[package]]
769
+ name = "tree-sitter-c-sharp"
770
+ version = "0.23.5"
771
+ source = "registry+https://github.com/rust-lang/crates.io-index"
772
+ checksum = "c1aac67f1ad71de1d6d39708d34811081c26dfa495658de6c14c34200849357c"
773
+ dependencies = [
774
+ "cc",
775
+ "tree-sitter-language",
776
+ ]
777
+
766
778
  [[package]]
767
779
  name = "tree-sitter-language"
768
780
  version = "0.1.8"
@@ -779,6 +791,16 @@ dependencies = [
779
791
  "tree-sitter-language",
780
792
  ]
781
793
 
794
+ [[package]]
795
+ name = "tree-sitter-sequel"
796
+ version = "0.3.11"
797
+ source = "registry+https://github.com/rust-lang/crates.io-index"
798
+ checksum = "9d198ad3c319c02e43c21efa1ec796b837afcb96ffaef1a40c1978fbdcec7d17"
799
+ dependencies = [
800
+ "cc",
801
+ "tree-sitter-language",
802
+ ]
803
+
782
804
  [[package]]
783
805
  name = "tree-sitter-typescript"
784
806
  version = "0.23.2"
@@ -3,6 +3,6 @@ resolver = "2"
3
3
  members = ["crates/duplicatecode-engine", "crates/duplicatecode-cli"]
4
4
 
5
5
  [workspace.package]
6
- version = "0.1.0"
6
+ version = "0.3.0"
7
7
  edition = "2021"
8
8
  license = "MIT"
@@ -1,16 +1,16 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: duplicatecode
3
- Version: 0.1.0
3
+ Version: 0.3.0
4
4
  Classifier: Programming Language :: Rust
5
5
  Classifier: Environment :: Console
6
- Summary: Static (LLM-free) detection of duplicate/similar code in Python and TypeScript
6
+ Summary: Static (LLM-free) detection of duplicate/similar code in Python, TypeScript/JavaScript, SQL and C#
7
7
  License: MIT
8
8
  Requires-Python: >=3.9
9
9
  Description-Content-Type: text/markdown; charset=UTF-8; variant=GFM
10
10
 
11
11
  # duplicatecode
12
12
 
13
- Static (LLM-free) detection of duplicate / similar code in Python and TypeScript (TSX),
13
+ Static (LLM-free) detection of duplicate / similar code in Python, TypeScript/JavaScript (TSX/JSX), SQL and C#,
14
14
  aimed at catching an LLM re-implementing something that already exists. Input can be a git diff
15
15
  checked against existing source.
16
16
 
@@ -42,6 +42,134 @@ duplicatecode scan . --pairs --json # machine-readable pairs in
42
42
  duplicatecode bench --dataset dataset [--file-level] [--mutations] [--negatives <other repo>]
43
43
  ```
44
44
 
45
+ ## More ways to look
46
+
47
+ ```sh
48
+ duplicatecode fragments packages/ # copied blocks inside different functions (4+ identical statements)
49
+ duplicatecode scan . --pairs --explain # say what differs: literals, calls, lines (what to parameterize)
50
+ duplicatecode find "retry with exponential backoff" src/ # does something like this already exist? (needs embeddings)
51
+ ```
52
+
53
+ - `fragments` indexes windows of consecutive normalized statements, so a block pasted into another
54
+ function is found even when the surrounding functions differ. On an injection benchmark (renamed
55
+ blocks of real functions pasted into others) it finds 93-97% of blocks of 6+ statements; see
56
+ `eval/REPORT.md`. Tune with `--min-stmts` / `--min-tokens`; constructors and dunders are ignored.
57
+ - `--explain` adds, per pair, the literals and calls only one side has and the source lines with no
58
+ counterpart ("16 of 18/17 statements shared").
59
+ - `find` embeds the query and every unit and ranks by similarity. With MiniLM, 50 task descriptions
60
+ against 168 implementation units rank the right one first 94% of the time; with 1,379 unrelated real
61
+ units mixed in, 82% first and 96% in the top 3 (every query within the top 10).
62
+ - Default thresholds depend on the profile: `copies` keeps its tuned 0.6 (`scan`), 0.4 (`diff`), 0.45
63
+ (`review`); `reimpl` scores live lower, so it defaults to 0.35 / 0.28 / 0.30, which correspond to roughly
64
+ 0.1% / 1% false-positive rates on unrelated code in the benchmarks. With `--embed` the `reimpl`
65
+ defaults rise by 0.07 because blended scores of unrelated code rise too.
66
+
67
+ ## Embeddings (bring your own key)
68
+
69
+ Two optional signals, both off by default (the detector stays LLM-free unless you ask):
70
+
71
+ - `--embeddings` embeds identifier *names* (cheap) as an extra name-similarity signal.
72
+ - `--embed <preset>` is the short form (`minilm`, `qwen3`, `potion`, `openai`, `cohere`; `qwen3` is about 5x slower than `minilm` on CPU, so use it with a GPU or prefer `minilm`) and implies
73
+ `--embed-code`. Model ids go into the cache key, so vectors of different models never mix.
74
+ - `--embed-code` embeds the *whole text of every unit* (function, class, file, SQL statement) and blends
75
+ the cosine into the score (`--embed-weight`, default 0.35, others scaled by 1 - weight). It finds
76
+ re-implementations that share no tokens. On function-level LLM re-implementations it lifted Python from
77
+ 0.70 to 0.89 on a held-out half; see `eval/REPORT.md` for the numbers and caveats. Vectors are cached per
78
+ unit text in `~/.cache/duplicatecode/embeddings.bin`, so a rescan only embeds what changed.
79
+
80
+ Credentials come from the environment:
81
+
82
+ ```sh
83
+ # OpenAI
84
+ export OPENAI_API_KEY=sk-... # optional: OPENAI_EMBEDDING_MODEL (default text-embedding-3-small)
85
+ # Cohere (native /v2/embed)
86
+ export COHERE_API_KEY=... # optional: COHERE_EMBEDDING_MODEL (default embed-v4.0)
87
+ # OpenAI-compatible server (vLLM, Ollama, LiteLLM, ...)
88
+ export OPENAI_BASE_URL=http://localhost:11434/v1 OPENAI_API_KEY=anything OPENAI_EMBEDDING_MODEL=nomic-embed-text
89
+ # Azure AI Foundry (key, or `az login` if no key is set)
90
+ export AZURE_AI_FOUNDRY_ENDPOINT=https://<res>.services.ai.azure.com
91
+ export AZURE_AI_FOUNDRY_API_KEY=... AZURE_AI_FOUNDRY_EMBEDDING_DEPLOYMENT=text-embedding-3-small
92
+ # Azure OpenAI: AZURE_OPENAI_ENDPOINT / AZURE_OPENAI_API_KEY / AZURE_OPENAI_EMBEDDING_DEPLOYMENT
93
+ # Fully local, no key: start the server once (CPU; it loads whichever model a request names)
94
+ # uv run --no-project --python 3.12 eval/embed_server.py
95
+ # duplicatecode scan . --embed minilm # or: qwen3, potion (default endpoint 127.0.0.1:8099)
96
+ # Hosted presets: --embed openai (OPENAI_API_KEY), --embed cohere (COHERE_API_KEY)
97
+
98
+ duplicatecode embed-test fetchUser getUser # check credentials
99
+ duplicatecode scan . --embed-code # whole-unit embeddings
100
+ duplicatecode scan . --embeddings # names only
101
+ ```
102
+
103
+ ## API
104
+
105
+ ### Command line
106
+
107
+ | command | what it does |
108
+ | --- | --- |
109
+ | `scan <paths>` | similar units inside one or more folders; `--pairs`, `--explain`, `--json`, `--fail-on-found` |
110
+ | `diff` | check code added in a git diff (stdin or `--diff`) against the repo (`--repo`) |
111
+ | `review` | ranked candidates for a diff, for a human or an agent to judge |
112
+ | `fragments <paths>` | copied blocks of identical statements inside different functions |
113
+ | `find "<description>" <paths>` | existing units closest to a plain-language description (needs embeddings) |
114
+ | `units <path>` / `show file:10-40` | list extracted units / print a unit's source |
115
+ | `embed-test <names...>` | check the embedding provider and calibrate `--embed-floor` |
116
+ | `eval-groups`, `bench`, `make-mutation-groups` | evaluation (see `eval/README.md`) |
117
+
118
+ Shared options: `--profile copies|reimpl`, `--threshold`, `--min-tokens`, `--min-lines`, `--exclude`, `--skip-tests`,
119
+ `--cross-file`, `--json`. Embedding options (all off by default): `--embed <preset>`, `--embed-code`,
120
+ `--embed-weight`, `--embed-max-chars`, `--embeddings`, `--embed-cache`, `--embed-dims`.
121
+
122
+ ```sh
123
+ duplicatecode scan . --profile reimpl --min-name 0 --embed minilm --pairs --explain
124
+ duplicatecode fragments src/ --min-stmts 4 --min-tokens 30 --cross-file --json
125
+ duplicatecode find "retry with exponential backoff" src/ --top 5 --json
126
+ duplicatecode diff --repo . < change.diff --embed openai
127
+ ```
128
+
129
+ JSON output:
130
+
131
+ - `find --json`: `[{score, file, name, kind, start_line, end_line}]`
132
+ - `fragments --json`: `[{a, b, statements, tokens}]` with `a`/`b` = `{file, unit, start_line, end_line, coverage}`
133
+ - `scan --pairs --json`: `[{query, candidate, scores}]`; with `--explain`: `[{match, explanation}]`
134
+ - `scan --json` (groups): `[{score, units: [{file, name, kind, start_line, end_line}]}]`
135
+
136
+ `scan --fail-on-found` exits with status 1 when anything is reported, for CI.
137
+
138
+ ### Rust library
139
+
140
+ The engine crate `duplicatecode-engine` is what the CLI uses:
141
+
142
+ ```rust
143
+ use duplicatecode_engine::embed::{embed_unit_code, EmbedConfig, EmbeddingCache};
144
+ use duplicatecode_engine::explain::explain;
145
+ use duplicatecode_engine::fragments::{find_fragments, FragmentOptions};
146
+ use duplicatecode_engine::index::Weights;
147
+ use duplicatecode_engine::{find_matches, load_units_with, Corpus, MatchOptions};
148
+
149
+ let mut units = load_units_with(Path::new("src/"), &[]);
150
+
151
+ // optional: embed every unit; units without a vector simply skip the embedding term
152
+ let cfg = EmbedConfig::from_env(None).ok_or("no embedding provider configured")?;
153
+ let mut cache = EmbeddingCache::load(&EmbeddingCache::default_path());
154
+ embed_unit_code(&mut units, &cfg, &mut cache, 3000)?;
155
+
156
+ // pairs: queries against a corpus
157
+ let weights = Weights::default().with_embed(0.35); // 0.0 = static only
158
+ let corpus = Corpus::new(units.clone());
159
+ let pairs = find_matches(&units, &corpus, MatchOptions { weights, threshold: 0.42, ..Default::default() });
160
+
161
+ // copied blocks, and what differs inside a pair
162
+ let blocks = find_fragments(&units, FragmentOptions::default());
163
+ println!("{}", explain(&units[0], &units[1]).summary());
164
+ ```
165
+
166
+ ### Not built yet
167
+
168
+ Planned, not available today: a `.duplicatecode.toml` with an `[embed]` section; `--embed-optional` (warn and
169
+ fall back to the static score when the provider is unreachable; today an unreachable provider is an error);
170
+ `--allow-upload` (required for the hosted presets, which send source text out); a `duplicatecode[embed]` Python
171
+ extra that starts the local server on demand; an `Embedder` trait so providers can plug in without HTTP.
172
+
45
173
  ## How it works
46
174
 
47
175
  Units (functions, methods, classes, arrow-function components) are extracted with tree-sitter and
@@ -1,6 +1,6 @@
1
1
  # duplicatecode
2
2
 
3
- Static (LLM-free) detection of duplicate / similar code in Python and TypeScript (TSX),
3
+ Static (LLM-free) detection of duplicate / similar code in Python, TypeScript/JavaScript (TSX/JSX), SQL and C#,
4
4
  aimed at catching an LLM re-implementing something that already exists. Input can be a git diff
5
5
  checked against existing source.
6
6
 
@@ -32,6 +32,134 @@ duplicatecode scan . --pairs --json # machine-readable pairs in
32
32
  duplicatecode bench --dataset dataset [--file-level] [--mutations] [--negatives <other repo>]
33
33
  ```
34
34
 
35
+ ## More ways to look
36
+
37
+ ```sh
38
+ duplicatecode fragments packages/ # copied blocks inside different functions (4+ identical statements)
39
+ duplicatecode scan . --pairs --explain # say what differs: literals, calls, lines (what to parameterize)
40
+ duplicatecode find "retry with exponential backoff" src/ # does something like this already exist? (needs embeddings)
41
+ ```
42
+
43
+ - `fragments` indexes windows of consecutive normalized statements, so a block pasted into another
44
+ function is found even when the surrounding functions differ. On an injection benchmark (renamed
45
+ blocks of real functions pasted into others) it finds 93-97% of blocks of 6+ statements; see
46
+ `eval/REPORT.md`. Tune with `--min-stmts` / `--min-tokens`; constructors and dunders are ignored.
47
+ - `--explain` adds, per pair, the literals and calls only one side has and the source lines with no
48
+ counterpart ("16 of 18/17 statements shared").
49
+ - `find` embeds the query and every unit and ranks by similarity. With MiniLM, 50 task descriptions
50
+ against 168 implementation units rank the right one first 94% of the time; with 1,379 unrelated real
51
+ units mixed in, 82% first and 96% in the top 3 (every query within the top 10).
52
+ - Default thresholds depend on the profile: `copies` keeps its tuned 0.6 (`scan`), 0.4 (`diff`), 0.45
53
+ (`review`); `reimpl` scores live lower, so it defaults to 0.35 / 0.28 / 0.30, which correspond to roughly
54
+ 0.1% / 1% false-positive rates on unrelated code in the benchmarks. With `--embed` the `reimpl`
55
+ defaults rise by 0.07 because blended scores of unrelated code rise too.
56
+
57
+ ## Embeddings (bring your own key)
58
+
59
+ Two optional signals, both off by default (the detector stays LLM-free unless you ask):
60
+
61
+ - `--embeddings` embeds identifier *names* (cheap) as an extra name-similarity signal.
62
+ - `--embed <preset>` is the short form (`minilm`, `qwen3`, `potion`, `openai`, `cohere`; `qwen3` is about 5x slower than `minilm` on CPU, so use it with a GPU or prefer `minilm`) and implies
63
+ `--embed-code`. Model ids go into the cache key, so vectors of different models never mix.
64
+ - `--embed-code` embeds the *whole text of every unit* (function, class, file, SQL statement) and blends
65
+ the cosine into the score (`--embed-weight`, default 0.35, others scaled by 1 - weight). It finds
66
+ re-implementations that share no tokens. On function-level LLM re-implementations it lifted Python from
67
+ 0.70 to 0.89 on a held-out half; see `eval/REPORT.md` for the numbers and caveats. Vectors are cached per
68
+ unit text in `~/.cache/duplicatecode/embeddings.bin`, so a rescan only embeds what changed.
69
+
70
+ Credentials come from the environment:
71
+
72
+ ```sh
73
+ # OpenAI
74
+ export OPENAI_API_KEY=sk-... # optional: OPENAI_EMBEDDING_MODEL (default text-embedding-3-small)
75
+ # Cohere (native /v2/embed)
76
+ export COHERE_API_KEY=... # optional: COHERE_EMBEDDING_MODEL (default embed-v4.0)
77
+ # OpenAI-compatible server (vLLM, Ollama, LiteLLM, ...)
78
+ export OPENAI_BASE_URL=http://localhost:11434/v1 OPENAI_API_KEY=anything OPENAI_EMBEDDING_MODEL=nomic-embed-text
79
+ # Azure AI Foundry (key, or `az login` if no key is set)
80
+ export AZURE_AI_FOUNDRY_ENDPOINT=https://<res>.services.ai.azure.com
81
+ export AZURE_AI_FOUNDRY_API_KEY=... AZURE_AI_FOUNDRY_EMBEDDING_DEPLOYMENT=text-embedding-3-small
82
+ # Azure OpenAI: AZURE_OPENAI_ENDPOINT / AZURE_OPENAI_API_KEY / AZURE_OPENAI_EMBEDDING_DEPLOYMENT
83
+ # Fully local, no key: start the server once (CPU; it loads whichever model a request names)
84
+ # uv run --no-project --python 3.12 eval/embed_server.py
85
+ # duplicatecode scan . --embed minilm # or: qwen3, potion (default endpoint 127.0.0.1:8099)
86
+ # Hosted presets: --embed openai (OPENAI_API_KEY), --embed cohere (COHERE_API_KEY)
87
+
88
+ duplicatecode embed-test fetchUser getUser # check credentials
89
+ duplicatecode scan . --embed-code # whole-unit embeddings
90
+ duplicatecode scan . --embeddings # names only
91
+ ```
92
+
93
+ ## API
94
+
95
+ ### Command line
96
+
97
+ | command | what it does |
98
+ | --- | --- |
99
+ | `scan <paths>` | similar units inside one or more folders; `--pairs`, `--explain`, `--json`, `--fail-on-found` |
100
+ | `diff` | check code added in a git diff (stdin or `--diff`) against the repo (`--repo`) |
101
+ | `review` | ranked candidates for a diff, for a human or an agent to judge |
102
+ | `fragments <paths>` | copied blocks of identical statements inside different functions |
103
+ | `find "<description>" <paths>` | existing units closest to a plain-language description (needs embeddings) |
104
+ | `units <path>` / `show file:10-40` | list extracted units / print a unit's source |
105
+ | `embed-test <names...>` | check the embedding provider and calibrate `--embed-floor` |
106
+ | `eval-groups`, `bench`, `make-mutation-groups` | evaluation (see `eval/README.md`) |
107
+
108
+ Shared options: `--profile copies|reimpl`, `--threshold`, `--min-tokens`, `--min-lines`, `--exclude`, `--skip-tests`,
109
+ `--cross-file`, `--json`. Embedding options (all off by default): `--embed <preset>`, `--embed-code`,
110
+ `--embed-weight`, `--embed-max-chars`, `--embeddings`, `--embed-cache`, `--embed-dims`.
111
+
112
+ ```sh
113
+ duplicatecode scan . --profile reimpl --min-name 0 --embed minilm --pairs --explain
114
+ duplicatecode fragments src/ --min-stmts 4 --min-tokens 30 --cross-file --json
115
+ duplicatecode find "retry with exponential backoff" src/ --top 5 --json
116
+ duplicatecode diff --repo . < change.diff --embed openai
117
+ ```
118
+
119
+ JSON output:
120
+
121
+ - `find --json`: `[{score, file, name, kind, start_line, end_line}]`
122
+ - `fragments --json`: `[{a, b, statements, tokens}]` with `a`/`b` = `{file, unit, start_line, end_line, coverage}`
123
+ - `scan --pairs --json`: `[{query, candidate, scores}]`; with `--explain`: `[{match, explanation}]`
124
+ - `scan --json` (groups): `[{score, units: [{file, name, kind, start_line, end_line}]}]`
125
+
126
+ `scan --fail-on-found` exits with status 1 when anything is reported, for CI.
127
+
128
+ ### Rust library
129
+
130
+ The engine crate `duplicatecode-engine` is what the CLI uses:
131
+
132
+ ```rust
133
+ use duplicatecode_engine::embed::{embed_unit_code, EmbedConfig, EmbeddingCache};
134
+ use duplicatecode_engine::explain::explain;
135
+ use duplicatecode_engine::fragments::{find_fragments, FragmentOptions};
136
+ use duplicatecode_engine::index::Weights;
137
+ use duplicatecode_engine::{find_matches, load_units_with, Corpus, MatchOptions};
138
+
139
+ let mut units = load_units_with(Path::new("src/"), &[]);
140
+
141
+ // optional: embed every unit; units without a vector simply skip the embedding term
142
+ let cfg = EmbedConfig::from_env(None).ok_or("no embedding provider configured")?;
143
+ let mut cache = EmbeddingCache::load(&EmbeddingCache::default_path());
144
+ embed_unit_code(&mut units, &cfg, &mut cache, 3000)?;
145
+
146
+ // pairs: queries against a corpus
147
+ let weights = Weights::default().with_embed(0.35); // 0.0 = static only
148
+ let corpus = Corpus::new(units.clone());
149
+ let pairs = find_matches(&units, &corpus, MatchOptions { weights, threshold: 0.42, ..Default::default() });
150
+
151
+ // copied blocks, and what differs inside a pair
152
+ let blocks = find_fragments(&units, FragmentOptions::default());
153
+ println!("{}", explain(&units[0], &units[1]).summary());
154
+ ```
155
+
156
+ ### Not built yet
157
+
158
+ Planned, not available today: a `.duplicatecode.toml` with an `[embed]` section; `--embed-optional` (warn and
159
+ fall back to the static score when the provider is unreachable; today an unreachable provider is an error);
160
+ `--allow-upload` (required for the hosted presets, which send source text out); a `duplicatecode[embed]` Python
161
+ extra that starts the local server on demand; an `Embedder` trait so providers can plug in without HTTP.
162
+
35
163
  ## How it works
36
164
 
37
165
  Units (functions, methods, classes, arrow-function components) are extracted with tree-sitter and
@@ -389,6 +389,9 @@ fn fit(samples: &[&Sample]) -> Weights {
389
389
  let mut w = Weights {
390
390
  bias: 0.0,
391
391
  w: [0.0; N_FEATURES],
392
+ sql: None,
393
+ file: None,
394
+ renormalize: false,
392
395
  name_floor: 0.5,
393
396
  };
394
397
  for _ in 0..2500 {
@@ -415,7 +418,7 @@ fn apply(w: &Weights, x: &[f64; N_FEATURES]) -> f64 {
415
418
  }
416
419
 
417
420
  /// Fraction of positives scoring above the (1 - fpr) quantile of the negatives.
418
- fn tpr_at_fpr(scores: &[(f64, bool)], fpr: f64) -> f64 {
421
+ pub(crate) fn tpr_at_fpr(scores: &[(f64, bool)], fpr: f64) -> f64 {
419
422
  let mut neg: Vec<f64> = scores.iter().filter(|s| !s.1).map(|s| s.0).collect();
420
423
  let pos: Vec<f64> = scores.iter().filter(|s| s.1).map(|s| s.0).collect();
421
424
  if neg.is_empty() || pos.is_empty() {
@@ -0,0 +1,267 @@
1
+ //! Labeled-group benchmark: files inside one group are clones, files of different groups are not.
2
+ //!
3
+ //! Layout `<root>/<dataset>/<group>/<file>`. Every file is one whole-file unit; all pairs inside a
4
+ //! dataset are scored. Reports AUC, recall at fixed false-positive rates and top-1 retrieval.
5
+
6
+ use crate::bench::tpr_at_fpr;
7
+ use anyhow::{bail, Result};
8
+ use duplicatecode_engine::index::{score_with_idf, Idf, Scores, Weights};
9
+ use duplicatecode_engine::{load_file_units, load_units, Unit};
10
+ use rayon::prelude::*;
11
+ use std::path::Path;
12
+
13
+ type Pick = fn(&Scores) -> f64;
14
+ const SIGNALS: [(&str, Pick); 7] = [
15
+ ("structure", |s| s.structural),
16
+ ("loose", |s| s.loose),
17
+ ("kinds", |s| s.kinds),
18
+ ("stmt_exact", |s| s.stmt_exact),
19
+ ("stmt_shape", |s| s.stmt_shape),
20
+ ("stmt_lcs", |s| s.stmt_lcs),
21
+ ("combined", |s| s.combined),
22
+ ];
23
+
24
+ /// Area under the ROC curve with tie handling (Mann-Whitney U).
25
+ fn auc(scores: &[(f64, bool)]) -> f64 {
26
+ let mut v: Vec<(f64, bool)> = scores.to_vec();
27
+ v.sort_by(|a, b| a.0.total_cmp(&b.0));
28
+ let (mut rank_sum, mut n_pos, mut i) = (0.0, 0usize, 0usize);
29
+ while i < v.len() {
30
+ let mut j = i;
31
+ while j < v.len() && v[j].0 == v[i].0 {
32
+ j += 1;
33
+ }
34
+ let avg_rank = (i + j + 1) as f64 / 2.0; // 1-based average rank of the tie block
35
+ let pos = v[i..j].iter().filter(|x| x.1).count();
36
+ rank_sum += avg_rank * pos as f64;
37
+ n_pos += pos;
38
+ i = j;
39
+ }
40
+ let n_neg = v.len() - n_pos;
41
+ if n_pos == 0 || n_neg == 0 {
42
+ return f64::NAN;
43
+ }
44
+ (rank_sum - (n_pos * (n_pos + 1)) as f64 / 2.0) / (n_pos * n_neg) as f64
45
+ }
46
+
47
+ fn group_of(u: &Unit) -> &str {
48
+ u.file.split('/').next().unwrap_or("")
49
+ }
50
+
51
+ /// Score above which only `fpr` of the negative pairs lie.
52
+ fn score_at_fpr(scores: &[(f64, bool)], fpr: f64) -> f64 {
53
+ let mut neg: Vec<f64> = scores.iter().filter(|s| !s.1).map(|s| s.0).collect();
54
+ neg.sort_by(|a, b| a.total_cmp(b));
55
+ neg.get(((1.0 - fpr) * neg.len() as f64) as usize)
56
+ .or(neg.last())
57
+ .copied()
58
+ .unwrap_or(f64::NAN)
59
+ }
60
+
61
+ struct Report {
62
+ thr1: f64,
63
+ thr01: f64,
64
+ auc: f64,
65
+ tpr1: f64,
66
+ tpr5: f64,
67
+ top1: f64,
68
+ }
69
+
70
+ fn evaluate(units: &[Unit], pick: Pick, all: &[Vec<Scores>]) -> Report {
71
+ let n = units.len();
72
+ let mut pairs = Vec::with_capacity(n * n / 2);
73
+ let mut hit = 0usize;
74
+ for i in 0..n {
75
+ let mut best: Option<(f64, bool)> = None;
76
+ for j in 0..n {
77
+ if i == j || units[i].file == units[j].file {
78
+ continue;
79
+ }
80
+ let v = pick(&all[i][j]);
81
+ let same = group_of(&units[i]) == group_of(&units[j]);
82
+ if j > i {
83
+ pairs.push((v, same));
84
+ }
85
+ if best.is_none_or(|b| v > b.0) {
86
+ best = Some((v, same));
87
+ }
88
+ }
89
+ hit += best.is_some_and(|b| b.1) as usize;
90
+ }
91
+ Report {
92
+ thr1: score_at_fpr(&pairs, 0.01),
93
+ thr01: score_at_fpr(&pairs, 0.001),
94
+ auc: auc(&pairs),
95
+ tpr1: tpr_at_fpr(&pairs, 0.01),
96
+ tpr5: tpr_at_fpr(&pairs, 0.05),
97
+ top1: hit as f64 / n as f64,
98
+ }
99
+ }
100
+
101
+ pub fn run(
102
+ root: &Path,
103
+ quiet: bool,
104
+ dump: Option<&Path>,
105
+ prepare: &dyn Fn(&mut [Unit]) -> Result<()>,
106
+ weights: Weights,
107
+ ) -> Result<()> {
108
+ let mut dump_out = dump.map(std::fs::File::create).transpose()?;
109
+ let mut datasets: Vec<_> = std::fs::read_dir(root)?
110
+ .filter_map(Result::ok)
111
+ .map(|e| e.path())
112
+ .filter(|p| p.is_dir() && !p.file_name().unwrap().to_string_lossy().starts_with('_'))
113
+ .collect();
114
+ datasets.sort();
115
+ if datasets.is_empty() {
116
+ bail!(
117
+ "no datasets below {} (run eval/fetch_codenet.py)",
118
+ root.display()
119
+ );
120
+ }
121
+ let mut headline = Vec::new();
122
+ for dir in datasets {
123
+ let name = dir.file_name().unwrap().to_string_lossy().to_string();
124
+ // `fn-*` datasets compare functions/classes (product path); the others whole files
125
+ let functions = name.starts_with("fn-");
126
+ let mut units = if functions {
127
+ load_units(&dir)
128
+ .into_iter()
129
+ .filter(|u| !u.boilerplate && u.token_count() >= 8)
130
+ .collect()
131
+ } else {
132
+ load_file_units(&dir)
133
+ };
134
+ units.sort_by(|a, b| (&a.file, a.start_line).cmp(&(&b.file, b.start_line)));
135
+ prepare(&mut units)?;
136
+ let n = units.len();
137
+ let idf = Idf::from_units(&units);
138
+ // full score matrix once; each signal is a projection of it
139
+ let all: Vec<Vec<Scores>> = (0..n)
140
+ .into_par_iter()
141
+ .map(|i| {
142
+ (0..n)
143
+ .map(|j| score_with_idf(&units[i], &units[j], &weights, Some(&idf)))
144
+ .collect()
145
+ })
146
+ .collect();
147
+ if let Some(f) = dump_out.as_mut() {
148
+ use std::io::Write;
149
+ for i in 0..n {
150
+ for j in i + 1..n {
151
+ if units[i].file == units[j].file {
152
+ continue;
153
+ }
154
+ let feats: Vec<String> = all[i][j]
155
+ .features()
156
+ .iter()
157
+ .map(|x| format!("{x:.5}"))
158
+ .collect();
159
+ writeln!(
160
+ f,
161
+ "{name}\t{}:{}\t{}:{}\t{}\t{}\t{}\t{}\t{}",
162
+ units[i].file,
163
+ units[i].start_line,
164
+ units[j].file,
165
+ units[j].start_line,
166
+ group_of(&units[i]),
167
+ group_of(&units[j]),
168
+ units[i].tokens.len().min(units[j].tokens.len()),
169
+ units[i].tokens.len().max(units[j].tokens.len()),
170
+ feats.join("\t")
171
+ )?;
172
+ }
173
+ }
174
+ }
175
+ let groups: std::collections::BTreeSet<&str> = units.iter().map(group_of).collect();
176
+ if !quiet {
177
+ println!("== {name}: {n} files, {} groups", groups.len());
178
+ println!(
179
+ "{:<12} {:>6} {:>8} {:>8} {:>7}",
180
+ "signal", "AUC", "TPR@1%", "TPR@5%", "top1"
181
+ );
182
+ }
183
+ for (label, pick) in SIGNALS {
184
+ let r = evaluate(&units, pick, &all);
185
+ if !quiet {
186
+ println!(
187
+ "{label:<12} {:>6.3} {:>8.3} {:>8.3} {:>7.3}",
188
+ r.auc, r.tpr1, r.tpr5, r.top1
189
+ );
190
+ }
191
+ if label == "combined" {
192
+ headline.push((name.clone(), (r.auc + r.tpr1) / 2.0));
193
+ println!(
194
+ "dataset_threshold {name} score@FPR1%={:.3} score@FPR0.1%={:.3}",
195
+ r.thr1, r.thr01
196
+ );
197
+ }
198
+ }
199
+ }
200
+ for (name, s) in &headline {
201
+ println!("dataset_score {name}={s:.4}");
202
+ }
203
+ let mean = headline.iter().map(|h| h.1).sum::<f64>() / headline.len() as f64;
204
+ println!("score={mean:.4}");
205
+ Ok(())
206
+ }
207
+
208
+ /// Build `<out>/<id>/<variant>.tsx` groups from real components: the original plus mechanically
209
+ /// rewritten variants (renames, statement swaps, temp variables, logging, dead code, all combined).
210
+ /// Other components act as negatives. `skip` lets a holdout set use disjoint files.
211
+ pub fn make_mutation_groups(
212
+ src: &Path,
213
+ out: &Path,
214
+ n: usize,
215
+ seed: u64,
216
+ skip: usize,
217
+ ) -> Result<()> {
218
+ use duplicatecode_engine::mutate::{apply, Mutation};
219
+ use duplicatecode_engine::{walk_files, Lang};
220
+ let mut files: Vec<_> = walk_files(src, &[])
221
+ .into_iter()
222
+ .filter(|p| {
223
+ let s = p.to_string_lossy();
224
+ s.ends_with(".tsx")
225
+ && !s.contains(".test.")
226
+ && !s.contains(".stories.")
227
+ && !s.contains("/generated/")
228
+ })
229
+ .filter(|p| {
230
+ std::fs::read_to_string(p)
231
+ .map(|t| (30..=300).contains(&t.lines().count()) && t.contains("</"))
232
+ .unwrap_or(false)
233
+ })
234
+ .collect();
235
+ files.sort();
236
+ let mut state = seed.wrapping_mul(0x9E37_79B9_7F4A_7C15) | 1;
237
+ let mut next = || {
238
+ state ^= state << 13;
239
+ state ^= state >> 7;
240
+ state ^= state << 17;
241
+ state
242
+ };
243
+ for i in (1..files.len()).rev() {
244
+ files.swap(i, (next() % (i as u64 + 1)) as usize);
245
+ }
246
+ for (k, p) in files.iter().skip(skip).take(n).enumerate() {
247
+ let text = std::fs::read_to_string(p)?;
248
+ let dir = out.join(format!("c{:03}", skip + k));
249
+ std::fs::create_dir_all(&dir)?;
250
+ std::fs::write(dir.join("orig.tsx"), &text)?;
251
+ for m in Mutation::ALL {
252
+ if m == Mutation::LoopToComprehension {
253
+ continue; // Python only
254
+ }
255
+ let v = apply(m, Lang::Tsx, &text);
256
+ if v != text {
257
+ std::fs::write(dir.join(format!("{}.tsx", m.name().replace(' ', "_"))), v)?;
258
+ }
259
+ }
260
+ }
261
+ println!(
262
+ "wrote {} component groups to {}",
263
+ files.len().saturating_sub(skip).min(n),
264
+ out.display()
265
+ );
266
+ Ok(())
267
+ }