semantic-dedup 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- semantic_dedup-0.1.0/.gitignore +30 -0
- semantic_dedup-0.1.0/LICENSE +21 -0
- semantic_dedup-0.1.0/PKG-INFO +170 -0
- semantic_dedup-0.1.0/README.md +146 -0
- semantic_dedup-0.1.0/pyproject.toml +49 -0
- semantic_dedup-0.1.0/src/semantic_dedup/__init__.py +36 -0
- semantic_dedup-0.1.0/src/semantic_dedup/__main__.py +7 -0
- semantic_dedup-0.1.0/src/semantic_dedup/_core.py +495 -0
- semantic_dedup-0.1.0/src/semantic_dedup/_io.py +180 -0
- semantic_dedup-0.1.0/src/semantic_dedup/_lexicon.py +199 -0
- semantic_dedup-0.1.0/src/semantic_dedup/_minhash.py +147 -0
- semantic_dedup-0.1.0/src/semantic_dedup/_text.py +193 -0
- semantic_dedup-0.1.0/src/semantic_dedup/_vectors.py +406 -0
- semantic_dedup-0.1.0/src/semantic_dedup/cli.py +114 -0
- semantic_dedup-0.1.0/tests/test_api.py +214 -0
- semantic_dedup-0.1.0/tests/test_cli.py +173 -0
- semantic_dedup-0.1.0/tests/test_dedupe.py +44 -0
- semantic_dedup-0.1.0/tests/test_edge_cases.py +236 -0
- semantic_dedup-0.1.0/tests/test_io.py +127 -0
- semantic_dedup-0.1.0/tests/test_regressions.py +220 -0
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
# Virtual environments and scratch space used while building. Never committed:
|
|
2
|
+
# they are large, machine-specific, and rebuilt from pyproject.toml anyway.
|
|
3
|
+
_venvs/
|
|
4
|
+
_proof/
|
|
5
|
+
.venv*/
|
|
6
|
+
|
|
7
|
+
# Build output. Wheels are built by the release workflow from this source, so a
|
|
8
|
+
# wheel in git could silently differ from the code beside it.
|
|
9
|
+
dist/
|
|
10
|
+
build/
|
|
11
|
+
*.egg-info/
|
|
12
|
+
src/*.egg-info/
|
|
13
|
+
|
|
14
|
+
# Python noise
|
|
15
|
+
__pycache__/
|
|
16
|
+
*.py[cod]
|
|
17
|
+
.pytest_cache/
|
|
18
|
+
.mypy_cache/
|
|
19
|
+
.ruff_cache/
|
|
20
|
+
|
|
21
|
+
# Local verification state, not source
|
|
22
|
+
verify_results.json
|
|
23
|
+
publish-log.txt
|
|
24
|
+
published.json
|
|
25
|
+
|
|
26
|
+
# Credentials. None of these belong here, and this line is the backstop.
|
|
27
|
+
.pypirc
|
|
28
|
+
.env
|
|
29
|
+
*.pem
|
|
30
|
+
*.key
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Pranay Mahendrakar
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,170 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: semantic-dedup
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Remove passages that repeat the same meaning, not just the same words
|
|
5
|
+
Project-URL: Homepage, https://pypi.org/project/semantic-dedup/
|
|
6
|
+
Project-URL: Author, https://pypi.org/user/pranaymahendrakar/
|
|
7
|
+
Author: Pranay Mahendrakar
|
|
8
|
+
License-Expression: MIT
|
|
9
|
+
License-File: LICENSE
|
|
10
|
+
Keywords: deduplication,embeddings,minhash,nlp,paraphrase,semantic-similarity,text-cleaning,tfidf
|
|
11
|
+
Classifier: Development Status :: 4 - Beta
|
|
12
|
+
Classifier: Intended Audience :: Developers
|
|
13
|
+
Classifier: Intended Audience :: Science/Research
|
|
14
|
+
Classifier: Operating System :: OS Independent
|
|
15
|
+
Classifier: Programming Language :: Python :: 3
|
|
16
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
17
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
18
|
+
Classifier: Topic :: Text Processing :: Linguistic
|
|
19
|
+
Requires-Python: >=3.9
|
|
20
|
+
Requires-Dist: numpy>=1.23
|
|
21
|
+
Provides-Extra: dev
|
|
22
|
+
Requires-Dist: pytest>=7; extra == 'dev'
|
|
23
|
+
Description-Content-Type: text/markdown
|
|
24
|
+
|
|
25
|
+
# semantic-dedup
|
|
26
|
+
|
|
27
|
+
Remove passages that repeat the same meaning, not just the same words - "the meeting was postponed"
|
|
28
|
+
and "we moved the meeting to a later date" are one idea, and you only need to keep one of them.
|
|
29
|
+
|
|
30
|
+
## Install
|
|
31
|
+
|
|
32
|
+
```
|
|
33
|
+
pip install semantic-dedup
|
|
34
|
+
```
|
|
35
|
+
|
|
36
|
+
The only dependency is numpy. Nothing is downloaded at runtime, and no model is ever fetched.
|
|
37
|
+
|
|
38
|
+
## Quickstart
|
|
39
|
+
|
|
40
|
+
```python
|
|
41
|
+
import semantic_dedup
|
|
42
|
+
|
|
43
|
+
notes = ["The meeting was postponed.", "the meeting was postponed", "We moved the meeting to a later date.", "Lunch is at noon."]
|
|
44
|
+
result = semantic_dedup.dedupe(notes)
|
|
45
|
+
print(result.summary())
|
|
46
|
+
print(result.texts)
|
|
47
|
+
```
|
|
48
|
+
|
|
49
|
+
```
|
|
50
|
+
semantic-dedup: 4 texts, 1 duplicate group, 2 removed, 50.0% smaller
|
|
51
|
+
method tfidf, threshold 0.82, keep longest
|
|
52
|
+
group 1 (3 texts, similarity 1.00 to 1.00)
|
|
53
|
+
removed [0] "The meeting was postponed."
|
|
54
|
+
removed [1] "the meeting was postponed"
|
|
55
|
+
kept [2] "We moved the meeting to a later date."
|
|
56
|
+
['We moved the meeting to a later date.', 'Lunch is at noon.']
|
|
57
|
+
```
|
|
58
|
+
|
|
59
|
+
`semantic_dedup.dedupe("notes.txt")` works the same way on a `.txt`, `.csv` or `.jsonl` file.
|
|
60
|
+
|
|
61
|
+
## What it does
|
|
62
|
+
|
|
63
|
+
- **Canonicalizes the wording first.** Text is NFKC-normalized, casefolded, split into words, and then
|
|
64
|
+
run through a small built-in English lexicon: a few dozen paraphrase families collapse onto one token
|
|
65
|
+
each (`postponed`, `delayed`, `pushed back`, `at a later date` all become `postpone`), stopwords go, and
|
|
66
|
+
a light suffix stemmer finishes the job. Negations (`not`, `never`, `no`) are deliberately *kept*, so a
|
|
67
|
+
sentence and its opposite do not look alike.
|
|
68
|
+
- **Then scores the canonical text.** Word 1- and 2-grams plus character 3- and 4-grams of the canonical
|
|
69
|
+
string are turned into a sublinear TF-IDF vector; similarity is the cosine between two vectors, in
|
|
70
|
+
`[0, 1]`. Because the characters are taken from the *canonical* string, two differently worded
|
|
71
|
+
sentences that canonicalize alike also share characters, and typos still land close together.
|
|
72
|
+
- **Numbers and identifiers are evidence, not noise.** `1000`, `2022` and `SKU12` are never stemmed and
|
|
73
|
+
are weighted above ordinary words, because an amount or a ticket number is the most distinguishing
|
|
74
|
+
thing in a passage. "Refund issued for 1000 rupees." and "Refund issued for 100 rupees." are two
|
|
75
|
+
different passages and stay apart; "Ticket 1199 was escalated." and "Ticket 1199 has been escalated."
|
|
76
|
+
are one, and group.
|
|
77
|
+
- **Groups and keeps one.** Pairs at or above `threshold` become edges; a duplicate group is a connected
|
|
78
|
+
component; `keep` decides which member survives.
|
|
79
|
+
- **How honest is this about "meaning"?** TF-IDF does not understand language. What it has is a lexicon
|
|
80
|
+
of common paraphrases and a robust surface metric, which handles the everyday cases - reworded tickets,
|
|
81
|
+
restated notes, copies with edits - and *will* miss anything whose paraphrase is not in the lexicon
|
|
82
|
+
("the sprint slipped" vs "we are behind schedule" scores near zero). When you need real semantics, pass
|
|
83
|
+
`embed=` and this package will use your vectors instead; that hook is the honest answer, and it is why
|
|
84
|
+
nothing here depends on a model.
|
|
85
|
+
- **Exact duplicates are free.** Texts that are identical after normalization are collapsed before any
|
|
86
|
+
scoring, so they group at *every* threshold and a file full of copies is fast.
|
|
87
|
+
- **Indices are the caller's.** `kept`, `removed`, `groups` and `pairs` always refer to positions in the
|
|
88
|
+
input list, whatever happens internally.
|
|
89
|
+
- **Deterministic.** The same input gives the same output, every run; the MinHash permutations come from
|
|
90
|
+
a fixed `random_state`.
|
|
91
|
+
|
|
92
|
+
## API
|
|
93
|
+
|
|
94
|
+
```python
|
|
95
|
+
semantic_dedup.dedupe(texts, *, threshold=0.82, method="auto", keep="longest", embed=None) -> DedupeResult
|
|
96
|
+
semantic_dedup.find_duplicates(texts, **kw) -> DedupeResult # same analysis, drops nothing
|
|
97
|
+
semantic_dedup.similarity(a, b, *, method="tfidf") -> float # 0.0 to 1.0
|
|
98
|
+
```
|
|
99
|
+
|
|
100
|
+
- `texts` - a list of strings, or a path to a `.txt` (one text per line), `.csv`/`.tsv` (a text column is
|
|
101
|
+
picked automatically, or name it with `column=`) or `.jsonl` file.
|
|
102
|
+
- `threshold` - minimum similarity, in `(0, 1]`. `1.0` means exact matches only: the similarity search is
|
|
103
|
+
skipped and only texts identical after normalization group.
|
|
104
|
+
- `method` -
|
|
105
|
+
`"tfidf"` compares every pair exactly;
|
|
106
|
+
`"minhash"` uses MinHash + LSH banding to propose candidate pairs and then scores each candidate with
|
|
107
|
+
the exact same cosine, so the numbers you get back are never estimates;
|
|
108
|
+
`"embed"` uses the vectors from your `embed` callable;
|
|
109
|
+
`"auto"` (default) is `embed` when you pass one, else `minhash` above 5000 distinct texts, else `tfidf`.
|
|
110
|
+
- `keep` - `"longest"` (default), `"first"`, `"last"`, or `"most_complete"`: the member whose words cover
|
|
111
|
+
the rest of the group best, which is usually the one that says everything the others say.
|
|
112
|
+
- `embed` - `callable(list[str]) -> ndarray` with one row per text. Rows are L2-normalized for you.
|
|
113
|
+
|
|
114
|
+
`similarity(a, b)` scores one pair **on its own**, while `dedupe(texts)` weights every word by how rare
|
|
115
|
+
it is **across the corpus you passed in** (that is what the IDF in TF-IDF means). The two numbers are
|
|
116
|
+
close but not identical - `similarity("The cat sat.", "The cat sat on the mat.")` is `0.65` alone and
|
|
117
|
+
`0.63` inside a three-note corpus - so use `similarity()` to get a feel for the scale, and calibrate the
|
|
118
|
+
threshold you ship on a sample of the real corpus.
|
|
119
|
+
|
|
120
|
+
```python
|
|
121
|
+
import semantic_dedup
|
|
122
|
+
from sentence_transformers import SentenceTransformer # not a dependency of this package
|
|
123
|
+
|
|
124
|
+
model = SentenceTransformer("all-MiniLM-L6-v2")
|
|
125
|
+
result = semantic_dedup.dedupe(texts, method="embed", embed=lambda batch: model.encode(batch))
|
|
126
|
+
```
|
|
127
|
+
|
|
128
|
+
`Deduper(threshold=0.82, method="auto", keep="longest", embed=None, word_ngram=(1, 2), char_ngram=(3, 4),
|
|
129
|
+
num_perm=128, random_state=0, column=None)` is the class underneath, with `.run(texts, drop=True)`.
|
|
130
|
+
|
|
131
|
+
`DedupeResult`
|
|
132
|
+
|
|
133
|
+
- `.kept` - `list[int]`, input positions that survived; `.removed` - what was dropped
|
|
134
|
+
- `.groups` - `list[list[int]]`, one ascending list per duplicate group (size 2 or more)
|
|
135
|
+
- `.pairs` - `list[(i, j, similarity)]` with `i < j`, the edges that formed the groups
|
|
136
|
+
- `.texts` - the surviving texts; `.all_texts` - the input as it was read
|
|
137
|
+
- `.n_removed`, `.n_kept`, `.n_groups`, `.n_duplicates`, `.n_texts`, `.reduction` (fraction removed)
|
|
138
|
+
- `.threshold`, `.method`, `.keep`, `.warnings`
|
|
139
|
+
- `.summary()` - the human-readable report shown above; `.to_dict()` - a JSON-safe dict
|
|
140
|
+
|
|
141
|
+
```python
|
|
142
|
+
semantic_dedup.similarity("Our prices went up.", "Costs increased.") # 1.0
|
|
143
|
+
semantic_dedup.similarity("Please fix the login bug.", "Lunch is at noon.") # 0.0
|
|
144
|
+
semantic_dedup.find_duplicates(texts).groups # report only, nothing dropped
|
|
145
|
+
semantic_dedup.dedupe(texts, threshold=1.0) # exact duplicates only
|
|
146
|
+
```
|
|
147
|
+
|
|
148
|
+
## CLI
|
|
149
|
+
|
|
150
|
+
```
|
|
151
|
+
semantic-dedup INPUT [--threshold 0.82] [--method auto|tfidf|minhash|embed] [--keep longest|first|last|most_complete]
|
|
152
|
+
[--column NAME] [--report] [--max-groups 5] [--json] [--output PATH]
|
|
153
|
+
```
|
|
154
|
+
|
|
155
|
+
- `semantic-dedup notes.txt` prints the summary.
|
|
156
|
+
- `semantic-dedup faq.csv --column question --threshold 0.9` reads one column of a table.
|
|
157
|
+
- `--report` finds duplicates without choosing anything to remove.
|
|
158
|
+
- `--json` prints `to_dict()` as JSON (UTF-8, never escaped); `--output PATH` writes the surviving texts,
|
|
159
|
+
one per line.
|
|
160
|
+
|
|
161
|
+
`--json` is an **index report**: `kept`, `removed`, `groups` and `pairs` are positions in the input file,
|
|
162
|
+
not the texts, so join it back to that file (line *n* of the `.txt`, row *n* of the `.csv`) to read it.
|
|
163
|
+
`pairs` carries one entry per matching pair and can be long on a large, repetitive corpus; `groups` is the
|
|
164
|
+
compact view. Use `--output PATH` when what you want is the cleaned text itself.
|
|
165
|
+
|
|
166
|
+
Output is UTF-8 whatever the console is set to, so piping a summary full of non-ASCII text is safe.
|
|
167
|
+
|
|
168
|
+
## License
|
|
169
|
+
|
|
170
|
+
MIT
|
|
@@ -0,0 +1,146 @@
|
|
|
1
|
+
# semantic-dedup
|
|
2
|
+
|
|
3
|
+
Remove passages that repeat the same meaning, not just the same words - "the meeting was postponed"
|
|
4
|
+
and "we moved the meeting to a later date" are one idea, and you only need to keep one of them.
|
|
5
|
+
|
|
6
|
+
## Install
|
|
7
|
+
|
|
8
|
+
```
|
|
9
|
+
pip install semantic-dedup
|
|
10
|
+
```
|
|
11
|
+
|
|
12
|
+
The only dependency is numpy. Nothing is downloaded at runtime, and no model is ever fetched.
|
|
13
|
+
|
|
14
|
+
## Quickstart
|
|
15
|
+
|
|
16
|
+
```python
|
|
17
|
+
import semantic_dedup
|
|
18
|
+
|
|
19
|
+
notes = ["The meeting was postponed.", "the meeting was postponed", "We moved the meeting to a later date.", "Lunch is at noon."]
|
|
20
|
+
result = semantic_dedup.dedupe(notes)
|
|
21
|
+
print(result.summary())
|
|
22
|
+
print(result.texts)
|
|
23
|
+
```
|
|
24
|
+
|
|
25
|
+
```
|
|
26
|
+
semantic-dedup: 4 texts, 1 duplicate group, 2 removed, 50.0% smaller
|
|
27
|
+
method tfidf, threshold 0.82, keep longest
|
|
28
|
+
group 1 (3 texts, similarity 1.00 to 1.00)
|
|
29
|
+
removed [0] "The meeting was postponed."
|
|
30
|
+
removed [1] "the meeting was postponed"
|
|
31
|
+
kept [2] "We moved the meeting to a later date."
|
|
32
|
+
['We moved the meeting to a later date.', 'Lunch is at noon.']
|
|
33
|
+
```
|
|
34
|
+
|
|
35
|
+
`semantic_dedup.dedupe("notes.txt")` works the same way on a `.txt`, `.csv` or `.jsonl` file.
|
|
36
|
+
|
|
37
|
+
## What it does
|
|
38
|
+
|
|
39
|
+
- **Canonicalizes the wording first.** Text is NFKC-normalized, casefolded, split into words, and then
|
|
40
|
+
run through a small built-in English lexicon: a few dozen paraphrase families collapse onto one token
|
|
41
|
+
each (`postponed`, `delayed`, `pushed back`, `at a later date` all become `postpone`), stopwords go, and
|
|
42
|
+
a light suffix stemmer finishes the job. Negations (`not`, `never`, `no`) are deliberately *kept*, so a
|
|
43
|
+
sentence and its opposite do not look alike.
|
|
44
|
+
- **Then scores the canonical text.** Word 1- and 2-grams plus character 3- and 4-grams of the canonical
|
|
45
|
+
string are turned into a sublinear TF-IDF vector; similarity is the cosine between two vectors, in
|
|
46
|
+
`[0, 1]`. Because the characters are taken from the *canonical* string, two differently worded
|
|
47
|
+
sentences that canonicalize alike also share characters, and typos still land close together.
|
|
48
|
+
- **Numbers and identifiers are evidence, not noise.** `1000`, `2022` and `SKU12` are never stemmed and
|
|
49
|
+
are weighted above ordinary words, because an amount or a ticket number is the most distinguishing
|
|
50
|
+
thing in a passage. "Refund issued for 1000 rupees." and "Refund issued for 100 rupees." are two
|
|
51
|
+
different passages and stay apart; "Ticket 1199 was escalated." and "Ticket 1199 has been escalated."
|
|
52
|
+
are one, and group.
|
|
53
|
+
- **Groups and keeps one.** Pairs at or above `threshold` become edges; a duplicate group is a connected
|
|
54
|
+
component; `keep` decides which member survives.
|
|
55
|
+
- **How honest is this about "meaning"?** TF-IDF does not understand language. What it has is a lexicon
|
|
56
|
+
of common paraphrases and a robust surface metric, which handles the everyday cases - reworded tickets,
|
|
57
|
+
restated notes, copies with edits - and *will* miss anything whose paraphrase is not in the lexicon
|
|
58
|
+
("the sprint slipped" vs "we are behind schedule" scores near zero). When you need real semantics, pass
|
|
59
|
+
`embed=` and this package will use your vectors instead; that hook is the honest answer, and it is why
|
|
60
|
+
nothing here depends on a model.
|
|
61
|
+
- **Exact duplicates are free.** Texts that are identical after normalization are collapsed before any
|
|
62
|
+
scoring, so they group at *every* threshold and a file full of copies is fast.
|
|
63
|
+
- **Indices are the caller's.** `kept`, `removed`, `groups` and `pairs` always refer to positions in the
|
|
64
|
+
input list, whatever happens internally.
|
|
65
|
+
- **Deterministic.** The same input gives the same output, every run; the MinHash permutations come from
|
|
66
|
+
a fixed `random_state`.
|
|
67
|
+
|
|
68
|
+
## API
|
|
69
|
+
|
|
70
|
+
```python
|
|
71
|
+
semantic_dedup.dedupe(texts, *, threshold=0.82, method="auto", keep="longest", embed=None) -> DedupeResult
|
|
72
|
+
semantic_dedup.find_duplicates(texts, **kw) -> DedupeResult # same analysis, drops nothing
|
|
73
|
+
semantic_dedup.similarity(a, b, *, method="tfidf") -> float # 0.0 to 1.0
|
|
74
|
+
```
|
|
75
|
+
|
|
76
|
+
- `texts` - a list of strings, or a path to a `.txt` (one text per line), `.csv`/`.tsv` (a text column is
|
|
77
|
+
picked automatically, or name it with `column=`) or `.jsonl` file.
|
|
78
|
+
- `threshold` - minimum similarity, in `(0, 1]`. `1.0` means exact matches only: the similarity search is
|
|
79
|
+
skipped and only texts identical after normalization group.
|
|
80
|
+
- `method` -
|
|
81
|
+
`"tfidf"` compares every pair exactly;
|
|
82
|
+
`"minhash"` uses MinHash + LSH banding to propose candidate pairs and then scores each candidate with
|
|
83
|
+
the exact same cosine, so the numbers you get back are never estimates;
|
|
84
|
+
`"embed"` uses the vectors from your `embed` callable;
|
|
85
|
+
`"auto"` (default) is `embed` when you pass one, else `minhash` above 5000 distinct texts, else `tfidf`.
|
|
86
|
+
- `keep` - `"longest"` (default), `"first"`, `"last"`, or `"most_complete"`: the member whose words cover
|
|
87
|
+
the rest of the group best, which is usually the one that says everything the others say.
|
|
88
|
+
- `embed` - `callable(list[str]) -> ndarray` with one row per text. Rows are L2-normalized for you.
|
|
89
|
+
|
|
90
|
+
`similarity(a, b)` scores one pair **on its own**, while `dedupe(texts)` weights every word by how rare
|
|
91
|
+
it is **across the corpus you passed in** (that is what the IDF in TF-IDF means). The two numbers are
|
|
92
|
+
close but not identical - `similarity("The cat sat.", "The cat sat on the mat.")` is `0.65` alone and
|
|
93
|
+
`0.63` inside a three-note corpus - so use `similarity()` to get a feel for the scale, and calibrate the
|
|
94
|
+
threshold you ship on a sample of the real corpus.
|
|
95
|
+
|
|
96
|
+
```python
|
|
97
|
+
import semantic_dedup
|
|
98
|
+
from sentence_transformers import SentenceTransformer # not a dependency of this package
|
|
99
|
+
|
|
100
|
+
model = SentenceTransformer("all-MiniLM-L6-v2")
|
|
101
|
+
result = semantic_dedup.dedupe(texts, method="embed", embed=lambda batch: model.encode(batch))
|
|
102
|
+
```
|
|
103
|
+
|
|
104
|
+
`Deduper(threshold=0.82, method="auto", keep="longest", embed=None, word_ngram=(1, 2), char_ngram=(3, 4),
|
|
105
|
+
num_perm=128, random_state=0, column=None)` is the class underneath, with `.run(texts, drop=True)`.
|
|
106
|
+
|
|
107
|
+
`DedupeResult`
|
|
108
|
+
|
|
109
|
+
- `.kept` - `list[int]`, input positions that survived; `.removed` - what was dropped
|
|
110
|
+
- `.groups` - `list[list[int]]`, one ascending list per duplicate group (size 2 or more)
|
|
111
|
+
- `.pairs` - `list[(i, j, similarity)]` with `i < j`, the edges that formed the groups
|
|
112
|
+
- `.texts` - the surviving texts; `.all_texts` - the input as it was read
|
|
113
|
+
- `.n_removed`, `.n_kept`, `.n_groups`, `.n_duplicates`, `.n_texts`, `.reduction` (fraction removed)
|
|
114
|
+
- `.threshold`, `.method`, `.keep`, `.warnings`
|
|
115
|
+
- `.summary()` - the human-readable report shown above; `.to_dict()` - a JSON-safe dict
|
|
116
|
+
|
|
117
|
+
```python
|
|
118
|
+
semantic_dedup.similarity("Our prices went up.", "Costs increased.") # 1.0
|
|
119
|
+
semantic_dedup.similarity("Please fix the login bug.", "Lunch is at noon.") # 0.0
|
|
120
|
+
semantic_dedup.find_duplicates(texts).groups # report only, nothing dropped
|
|
121
|
+
semantic_dedup.dedupe(texts, threshold=1.0) # exact duplicates only
|
|
122
|
+
```
|
|
123
|
+
|
|
124
|
+
## CLI
|
|
125
|
+
|
|
126
|
+
```
|
|
127
|
+
semantic-dedup INPUT [--threshold 0.82] [--method auto|tfidf|minhash|embed] [--keep longest|first|last|most_complete]
|
|
128
|
+
[--column NAME] [--report] [--max-groups 5] [--json] [--output PATH]
|
|
129
|
+
```
|
|
130
|
+
|
|
131
|
+
- `semantic-dedup notes.txt` prints the summary.
|
|
132
|
+
- `semantic-dedup faq.csv --column question --threshold 0.9` reads one column of a table.
|
|
133
|
+
- `--report` finds duplicates without choosing anything to remove.
|
|
134
|
+
- `--json` prints `to_dict()` as JSON (UTF-8, never escaped); `--output PATH` writes the surviving texts,
|
|
135
|
+
one per line.
|
|
136
|
+
|
|
137
|
+
`--json` is an **index report**: `kept`, `removed`, `groups` and `pairs` are positions in the input file,
|
|
138
|
+
not the texts, so join it back to that file (line *n* of the `.txt`, row *n* of the `.csv`) to read it.
|
|
139
|
+
`pairs` carries one entry per matching pair and can be long on a large, repetitive corpus; `groups` is the
|
|
140
|
+
compact view. Use `--output PATH` when what you want is the cleaned text itself.
|
|
141
|
+
|
|
142
|
+
Output is UTF-8 whatever the console is set to, so piping a summary full of non-ASCII text is safe.
|
|
143
|
+
|
|
144
|
+
## License
|
|
145
|
+
|
|
146
|
+
MIT
|
|
@@ -0,0 +1,49 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["hatchling>=1.27"]
|
|
3
|
+
build-backend = "hatchling.build"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "semantic-dedup"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "Remove passages that repeat the same meaning, not just the same words"
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.9"
|
|
11
|
+
license = "MIT"
|
|
12
|
+
license-files = ["LICENSE"]
|
|
13
|
+
authors = [{ name = "Pranay Mahendrakar" }]
|
|
14
|
+
keywords = [
|
|
15
|
+
"deduplication",
|
|
16
|
+
"semantic-similarity",
|
|
17
|
+
"paraphrase",
|
|
18
|
+
"tfidf",
|
|
19
|
+
"minhash",
|
|
20
|
+
"embeddings",
|
|
21
|
+
"text-cleaning",
|
|
22
|
+
"nlp",
|
|
23
|
+
]
|
|
24
|
+
classifiers = [
|
|
25
|
+
"Development Status :: 4 - Beta",
|
|
26
|
+
"Intended Audience :: Developers",
|
|
27
|
+
"Intended Audience :: Science/Research",
|
|
28
|
+
"Programming Language :: Python :: 3",
|
|
29
|
+
"Programming Language :: Python :: 3 :: Only",
|
|
30
|
+
"Operating System :: OS Independent",
|
|
31
|
+
"Topic :: Scientific/Engineering :: Artificial Intelligence",
|
|
32
|
+
"Topic :: Text Processing :: Linguistic",
|
|
33
|
+
]
|
|
34
|
+
dependencies = [
|
|
35
|
+
"numpy>=1.23",
|
|
36
|
+
]
|
|
37
|
+
|
|
38
|
+
[project.optional-dependencies]
|
|
39
|
+
dev = ["pytest>=7"]
|
|
40
|
+
|
|
41
|
+
[project.scripts]
|
|
42
|
+
semantic-dedup = "semantic_dedup.cli:main"
|
|
43
|
+
|
|
44
|
+
[project.urls]
|
|
45
|
+
Homepage = "https://pypi.org/project/semantic-dedup/"
|
|
46
|
+
Author = "https://pypi.org/user/pranaymahendrakar/"
|
|
47
|
+
|
|
48
|
+
[tool.hatch.build.targets.wheel]
|
|
49
|
+
packages = ["src/semantic_dedup"]
|
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
"""semantic-dedup: remove passages that repeat the same meaning, not just the same words.
|
|
2
|
+
|
|
3
|
+
Quick use::
|
|
4
|
+
|
|
5
|
+
import semantic_dedup
|
|
6
|
+
result = semantic_dedup.dedupe(texts) # or a .txt / .csv / .jsonl path
|
|
7
|
+
print(result.summary())
|
|
8
|
+
clean = result.texts
|
|
9
|
+
"""
|
|
10
|
+
from ._core import (
|
|
11
|
+
AUTO_MINHASH_ABOVE,
|
|
12
|
+
KEEP_MODES,
|
|
13
|
+
METHODS,
|
|
14
|
+
DedupeResult,
|
|
15
|
+
Deduper,
|
|
16
|
+
dedupe,
|
|
17
|
+
find_duplicates,
|
|
18
|
+
similarity,
|
|
19
|
+
)
|
|
20
|
+
from ._text import canonical_tokens, normalize
|
|
21
|
+
|
|
22
|
+
__version__ = "0.1.0"
|
|
23
|
+
|
|
24
|
+
__all__ = [
|
|
25
|
+
"DedupeResult",
|
|
26
|
+
"Deduper",
|
|
27
|
+
"dedupe",
|
|
28
|
+
"find_duplicates",
|
|
29
|
+
"similarity",
|
|
30
|
+
"canonical_tokens",
|
|
31
|
+
"normalize",
|
|
32
|
+
"METHODS",
|
|
33
|
+
"KEEP_MODES",
|
|
34
|
+
"AUTO_MINHASH_ABOVE",
|
|
35
|
+
"__version__",
|
|
36
|
+
]
|