semantic-dedup 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,30 @@
1
+ # Virtual environments and scratch space used while building. Never committed:
2
+ # they are large, machine-specific, and rebuilt from pyproject.toml anyway.
3
+ _venvs/
4
+ _proof/
5
+ .venv*/
6
+
7
+ # Build output. Wheels are built by the release workflow from this source, so a
8
+ # wheel in git could silently differ from the code beside it.
9
+ dist/
10
+ build/
11
+ *.egg-info/
12
+ src/*.egg-info/
13
+
14
+ # Python noise
15
+ __pycache__/
16
+ *.py[cod]
17
+ .pytest_cache/
18
+ .mypy_cache/
19
+ .ruff_cache/
20
+
21
+ # Local verification state, not source
22
+ verify_results.json
23
+ publish-log.txt
24
+ published.json
25
+
26
+ # Credentials. None of these belong here, and this line is the backstop.
27
+ .pypirc
28
+ .env
29
+ *.pem
30
+ *.key
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Pranay Mahendrakar
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,170 @@
1
+ Metadata-Version: 2.5
2
+ Name: semantic-dedup
3
+ Version: 0.1.0
4
+ Summary: Remove passages that repeat the same meaning, not just the same words
5
+ Project-URL: Homepage, https://pypi.org/project/semantic-dedup/
6
+ Project-URL: Author, https://pypi.org/user/pranaymahendrakar/
7
+ Author: Pranay Mahendrakar
8
+ License-Expression: MIT
9
+ License-File: LICENSE
10
+ Keywords: deduplication,embeddings,minhash,nlp,paraphrase,semantic-similarity,text-cleaning,tfidf
11
+ Classifier: Development Status :: 4 - Beta
12
+ Classifier: Intended Audience :: Developers
13
+ Classifier: Intended Audience :: Science/Research
14
+ Classifier: Operating System :: OS Independent
15
+ Classifier: Programming Language :: Python :: 3
16
+ Classifier: Programming Language :: Python :: 3 :: Only
17
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
18
+ Classifier: Topic :: Text Processing :: Linguistic
19
+ Requires-Python: >=3.9
20
+ Requires-Dist: numpy>=1.23
21
+ Provides-Extra: dev
22
+ Requires-Dist: pytest>=7; extra == 'dev'
23
+ Description-Content-Type: text/markdown
24
+
25
+ # semantic-dedup
26
+
27
+ Remove passages that repeat the same meaning, not just the same words - "the meeting was postponed"
28
+ and "we moved the meeting to a later date" are one idea, and you only need to keep one of them.
29
+
30
+ ## Install
31
+
32
+ ```
33
+ pip install semantic-dedup
34
+ ```
35
+
36
+ The only dependency is numpy. Nothing is downloaded at runtime, and no model is ever fetched.
37
+
38
+ ## Quickstart
39
+
40
+ ```python
41
+ import semantic_dedup
42
+
43
+ notes = ["The meeting was postponed.", "the meeting was postponed", "We moved the meeting to a later date.", "Lunch is at noon."]
44
+ result = semantic_dedup.dedupe(notes)
45
+ print(result.summary())
46
+ print(result.texts)
47
+ ```
48
+
49
+ ```
50
+ semantic-dedup: 4 texts, 1 duplicate group, 2 removed, 50.0% smaller
51
+ method tfidf, threshold 0.82, keep longest
52
+ group 1 (3 texts, similarity 1.00 to 1.00)
53
+ removed [0] "The meeting was postponed."
54
+ removed [1] "the meeting was postponed"
55
+ kept [2] "We moved the meeting to a later date."
56
+ ['We moved the meeting to a later date.', 'Lunch is at noon.']
57
+ ```
58
+
59
+ `semantic_dedup.dedupe("notes.txt")` works the same way on a `.txt`, `.csv` or `.jsonl` file.
60
+
61
+ ## What it does
62
+
63
+ - **Canonicalizes the wording first.** Text is NFKC-normalized, casefolded, split into words, and then
64
+ run through a small built-in English lexicon: a few dozen paraphrase families collapse onto one token
65
+ each (`postponed`, `delayed`, `pushed back`, `at a later date` all become `postpone`), stopwords go, and
66
+ a light suffix stemmer finishes the job. Negations (`not`, `never`, `no`) are deliberately *kept*, so a
67
+ sentence and its opposite do not look alike.
68
+ - **Then scores the canonical text.** Word 1- and 2-grams plus character 3- and 4-grams of the canonical
69
+ string are turned into a sublinear TF-IDF vector; similarity is the cosine between two vectors, in
70
+ `[0, 1]`. Because the characters are taken from the *canonical* string, two differently worded
71
+ sentences that canonicalize alike also share characters, and typos still land close together.
72
+ - **Numbers and identifiers are evidence, not noise.** `1000`, `2022` and `SKU12` are never stemmed and
73
+ are weighted above ordinary words, because an amount or a ticket number is the most distinguishing
74
+ thing in a passage. "Refund issued for 1000 rupees." and "Refund issued for 100 rupees." are two
75
+ different passages and stay apart; "Ticket 1199 was escalated." and "Ticket 1199 has been escalated."
76
+ are one, and group.
77
+ - **Groups and keeps one.** Pairs at or above `threshold` become edges; a duplicate group is a connected
78
+ component; `keep` decides which member survives.
79
+ - **How honest is this about "meaning"?** TF-IDF does not understand language. What it has is a lexicon
80
+ of common paraphrases and a robust surface metric, which handles the everyday cases - reworded tickets,
81
+ restated notes, copies with edits - and *will* miss anything whose paraphrase is not in the lexicon
82
+ ("the sprint slipped" vs "we are behind schedule" scores near zero). When you need real semantics, pass
83
+ `embed=` and this package will use your vectors instead; that hook is the honest answer, and it is why
84
+ nothing here depends on a model.
85
+ - **Exact duplicates are free.** Texts that are identical after normalization are collapsed before any
86
+ scoring, so they group at *every* threshold and a file full of copies is fast.
87
+ - **Indices are the caller's.** `kept`, `removed`, `groups` and `pairs` always refer to positions in the
88
+ input list, whatever happens internally.
89
+ - **Deterministic.** The same input gives the same output, every run; the MinHash permutations come from
90
+ a fixed `random_state`.
91
+
92
+ ## API
93
+
94
+ ```python
95
+ semantic_dedup.dedupe(texts, *, threshold=0.82, method="auto", keep="longest", embed=None) -> DedupeResult
96
+ semantic_dedup.find_duplicates(texts, **kw) -> DedupeResult # same analysis, drops nothing
97
+ semantic_dedup.similarity(a, b, *, method="tfidf") -> float # 0.0 to 1.0
98
+ ```
99
+
100
+ - `texts` - a list of strings, or a path to a `.txt` (one text per line), `.csv`/`.tsv` (a text column is
101
+ picked automatically, or name it with `column=`) or `.jsonl` file.
102
+ - `threshold` - minimum similarity, in `(0, 1]`. `1.0` means exact matches only: the similarity search is
103
+ skipped and only texts identical after normalization group.
104
+ - `method` -
105
+ `"tfidf"` compares every pair exactly;
106
+ `"minhash"` uses MinHash + LSH banding to propose candidate pairs and then scores each candidate with
107
+ the exact same cosine, so the numbers you get back are never estimates;
108
+ `"embed"` uses the vectors from your `embed` callable;
109
+ `"auto"` (default) is `embed` when you pass one, else `minhash` above 5000 distinct texts, else `tfidf`.
110
+ - `keep` - `"longest"` (default), `"first"`, `"last"`, or `"most_complete"`: the member whose words cover
111
+ the rest of the group best, which is usually the one that says everything the others say.
112
+ - `embed` - `callable(list[str]) -> ndarray` with one row per text. Rows are L2-normalized for you.
113
+
114
+ `similarity(a, b)` scores one pair **on its own**, while `dedupe(texts)` weights every word by how rare
115
+ it is **across the corpus you passed in** (that is what the IDF in TF-IDF means). The two numbers are
116
+ close but not identical - `similarity("The cat sat.", "The cat sat on the mat.")` is `0.65` alone and
117
+ `0.63` inside a three-note corpus - so use `similarity()` to get a feel for the scale, and calibrate the
118
+ threshold you ship on a sample of the real corpus.
119
+
120
+ ```python
121
+ import semantic_dedup
122
+ from sentence_transformers import SentenceTransformer # not a dependency of this package
123
+
124
+ model = SentenceTransformer("all-MiniLM-L6-v2")
125
+ result = semantic_dedup.dedupe(texts, method="embed", embed=lambda batch: model.encode(batch))
126
+ ```
127
+
128
+ `Deduper(threshold=0.82, method="auto", keep="longest", embed=None, word_ngram=(1, 2), char_ngram=(3, 4),
129
+ num_perm=128, random_state=0, column=None)` is the class underneath, with `.run(texts, drop=True)`.
130
+
131
+ `DedupeResult`
132
+
133
+ - `.kept` - `list[int]`, input positions that survived; `.removed` - what was dropped
134
+ - `.groups` - `list[list[int]]`, one ascending list per duplicate group (size 2 or more)
135
+ - `.pairs` - `list[(i, j, similarity)]` with `i < j`, the edges that formed the groups
136
+ - `.texts` - the surviving texts; `.all_texts` - the input as it was read
137
+ - `.n_removed`, `.n_kept`, `.n_groups`, `.n_duplicates`, `.n_texts`, `.reduction` (fraction removed)
138
+ - `.threshold`, `.method`, `.keep`, `.warnings`
139
+ - `.summary()` - the human-readable report shown above; `.to_dict()` - a JSON-safe dict
140
+
141
+ ```python
142
+ semantic_dedup.similarity("Our prices went up.", "Costs increased.") # 1.0
143
+ semantic_dedup.similarity("Please fix the login bug.", "Lunch is at noon.") # 0.0
144
+ semantic_dedup.find_duplicates(texts).groups # report only, nothing dropped
145
+ semantic_dedup.dedupe(texts, threshold=1.0) # exact duplicates only
146
+ ```
147
+
148
+ ## CLI
149
+
150
+ ```
151
+ semantic-dedup INPUT [--threshold 0.82] [--method auto|tfidf|minhash|embed] [--keep longest|first|last|most_complete]
152
+ [--column NAME] [--report] [--max-groups 5] [--json] [--output PATH]
153
+ ```
154
+
155
+ - `semantic-dedup notes.txt` prints the summary.
156
+ - `semantic-dedup faq.csv --column question --threshold 0.9` reads one column of a table.
157
+ - `--report` finds duplicates without choosing anything to remove.
158
+ - `--json` prints `to_dict()` as JSON (UTF-8, never escaped); `--output PATH` writes the surviving texts,
159
+ one per line.
160
+
161
+ `--json` is an **index report**: `kept`, `removed`, `groups` and `pairs` are positions in the input file,
162
+ not the texts, so join it back to that file (line *n* of the `.txt`, row *n* of the `.csv`) to read it.
163
+ `pairs` carries one entry per matching pair and can be long on a large, repetitive corpus; `groups` is the
164
+ compact view. Use `--output PATH` when what you want is the cleaned text itself.
165
+
166
+ Output is UTF-8 whatever the console is set to, so piping a summary full of non-ASCII text is safe.
167
+
168
+ ## License
169
+
170
+ MIT
@@ -0,0 +1,146 @@
1
+ # semantic-dedup
2
+
3
+ Remove passages that repeat the same meaning, not just the same words - "the meeting was postponed"
4
+ and "we moved the meeting to a later date" are one idea, and you only need to keep one of them.
5
+
6
+ ## Install
7
+
8
+ ```
9
+ pip install semantic-dedup
10
+ ```
11
+
12
+ The only dependency is numpy. Nothing is downloaded at runtime, and no model is ever fetched.
13
+
14
+ ## Quickstart
15
+
16
+ ```python
17
+ import semantic_dedup
18
+
19
+ notes = ["The meeting was postponed.", "the meeting was postponed", "We moved the meeting to a later date.", "Lunch is at noon."]
20
+ result = semantic_dedup.dedupe(notes)
21
+ print(result.summary())
22
+ print(result.texts)
23
+ ```
24
+
25
+ ```
26
+ semantic-dedup: 4 texts, 1 duplicate group, 2 removed, 50.0% smaller
27
+ method tfidf, threshold 0.82, keep longest
28
+ group 1 (3 texts, similarity 1.00 to 1.00)
29
+ removed [0] "The meeting was postponed."
30
+ removed [1] "the meeting was postponed"
31
+ kept [2] "We moved the meeting to a later date."
32
+ ['We moved the meeting to a later date.', 'Lunch is at noon.']
33
+ ```
34
+
35
+ `semantic_dedup.dedupe("notes.txt")` works the same way on a `.txt`, `.csv` or `.jsonl` file.
36
+
37
+ ## What it does
38
+
39
+ - **Canonicalizes the wording first.** Text is NFKC-normalized, casefolded, split into words, and then
40
+ run through a small built-in English lexicon: a few dozen paraphrase families collapse onto one token
41
+ each (`postponed`, `delayed`, `pushed back`, `at a later date` all become `postpone`), stopwords go, and
42
+ a light suffix stemmer finishes the job. Negations (`not`, `never`, `no`) are deliberately *kept*, so a
43
+ sentence and its opposite do not look alike.
44
+ - **Then scores the canonical text.** Word 1- and 2-grams plus character 3- and 4-grams of the canonical
45
+ string are turned into a sublinear TF-IDF vector; similarity is the cosine between two vectors, in
46
+ `[0, 1]`. Because the characters are taken from the *canonical* string, two differently worded
47
+ sentences that canonicalize alike also share characters, and typos still land close together.
48
+ - **Numbers and identifiers are evidence, not noise.** `1000`, `2022` and `SKU12` are never stemmed and
49
+ are weighted above ordinary words, because an amount or a ticket number is the most distinguishing
50
+ thing in a passage. "Refund issued for 1000 rupees." and "Refund issued for 100 rupees." are two
51
+ different passages and stay apart; "Ticket 1199 was escalated." and "Ticket 1199 has been escalated."
52
+ are one, and group.
53
+ - **Groups and keeps one.** Pairs at or above `threshold` become edges; a duplicate group is a connected
54
+ component; `keep` decides which member survives.
55
+ - **How honest is this about "meaning"?** TF-IDF does not understand language. What it has is a lexicon
56
+ of common paraphrases and a robust surface metric, which handles the everyday cases - reworded tickets,
57
+ restated notes, copies with edits - and *will* miss anything whose paraphrase is not in the lexicon
58
+ ("the sprint slipped" vs "we are behind schedule" scores near zero). When you need real semantics, pass
59
+ `embed=` and this package will use your vectors instead; that hook is the honest answer, and it is why
60
+ nothing here depends on a model.
61
+ - **Exact duplicates are free.** Texts that are identical after normalization are collapsed before any
62
+ scoring, so they group at *every* threshold and a file full of copies is fast.
63
+ - **Indices are the caller's.** `kept`, `removed`, `groups` and `pairs` always refer to positions in the
64
+ input list, whatever happens internally.
65
+ - **Deterministic.** The same input gives the same output, every run; the MinHash permutations come from
66
+ a fixed `random_state`.
67
+
68
+ ## API
69
+
70
+ ```python
71
+ semantic_dedup.dedupe(texts, *, threshold=0.82, method="auto", keep="longest", embed=None) -> DedupeResult
72
+ semantic_dedup.find_duplicates(texts, **kw) -> DedupeResult # same analysis, drops nothing
73
+ semantic_dedup.similarity(a, b, *, method="tfidf") -> float # 0.0 to 1.0
74
+ ```
75
+
76
+ - `texts` - a list of strings, or a path to a `.txt` (one text per line), `.csv`/`.tsv` (a text column is
77
+ picked automatically, or name it with `column=`) or `.jsonl` file.
78
+ - `threshold` - minimum similarity, in `(0, 1]`. `1.0` means exact matches only: the similarity search is
79
+ skipped and only texts identical after normalization group.
80
+ - `method` -
81
+ `"tfidf"` compares every pair exactly;
82
+ `"minhash"` uses MinHash + LSH banding to propose candidate pairs and then scores each candidate with
83
+ the exact same cosine, so the numbers you get back are never estimates;
84
+ `"embed"` uses the vectors from your `embed` callable;
85
+ `"auto"` (default) is `embed` when you pass one, else `minhash` above 5000 distinct texts, else `tfidf`.
86
+ - `keep` - `"longest"` (default), `"first"`, `"last"`, or `"most_complete"`: the member whose words cover
87
+ the rest of the group best, which is usually the one that says everything the others say.
88
+ - `embed` - `callable(list[str]) -> ndarray` with one row per text. Rows are L2-normalized for you.
89
+
90
+ `similarity(a, b)` scores one pair **on its own**, while `dedupe(texts)` weights every word by how rare
91
+ it is **across the corpus you passed in** (that is what the IDF in TF-IDF means). The two numbers are
92
+ close but not identical - `similarity("The cat sat.", "The cat sat on the mat.")` is `0.65` alone and
93
+ `0.63` inside a three-note corpus - so use `similarity()` to get a feel for the scale, and calibrate the
94
+ threshold you ship on a sample of the real corpus.
95
+
96
+ ```python
97
+ import semantic_dedup
98
+ from sentence_transformers import SentenceTransformer # not a dependency of this package
99
+
100
+ model = SentenceTransformer("all-MiniLM-L6-v2")
101
+ result = semantic_dedup.dedupe(texts, method="embed", embed=lambda batch: model.encode(batch))
102
+ ```
103
+
104
+ `Deduper(threshold=0.82, method="auto", keep="longest", embed=None, word_ngram=(1, 2), char_ngram=(3, 4),
105
+ num_perm=128, random_state=0, column=None)` is the class underneath, with `.run(texts, drop=True)`.
106
+
107
+ `DedupeResult`
108
+
109
+ - `.kept` - `list[int]`, input positions that survived; `.removed` - what was dropped
110
+ - `.groups` - `list[list[int]]`, one ascending list per duplicate group (size 2 or more)
111
+ - `.pairs` - `list[(i, j, similarity)]` with `i < j`, the edges that formed the groups
112
+ - `.texts` - the surviving texts; `.all_texts` - the input as it was read
113
+ - `.n_removed`, `.n_kept`, `.n_groups`, `.n_duplicates`, `.n_texts`, `.reduction` (fraction removed)
114
+ - `.threshold`, `.method`, `.keep`, `.warnings`
115
+ - `.summary()` - the human-readable report shown above; `.to_dict()` - a JSON-safe dict
116
+
117
+ ```python
118
+ semantic_dedup.similarity("Our prices went up.", "Costs increased.") # 1.0
119
+ semantic_dedup.similarity("Please fix the login bug.", "Lunch is at noon.") # 0.0
120
+ semantic_dedup.find_duplicates(texts).groups # report only, nothing dropped
121
+ semantic_dedup.dedupe(texts, threshold=1.0) # exact duplicates only
122
+ ```
123
+
124
+ ## CLI
125
+
126
+ ```
127
+ semantic-dedup INPUT [--threshold 0.82] [--method auto|tfidf|minhash|embed] [--keep longest|first|last|most_complete]
128
+ [--column NAME] [--report] [--max-groups 5] [--json] [--output PATH]
129
+ ```
130
+
131
+ - `semantic-dedup notes.txt` prints the summary.
132
+ - `semantic-dedup faq.csv --column question --threshold 0.9` reads one column of a table.
133
+ - `--report` finds duplicates without choosing anything to remove.
134
+ - `--json` prints `to_dict()` as JSON (UTF-8, never escaped); `--output PATH` writes the surviving texts,
135
+ one per line.
136
+
137
+ `--json` is an **index report**: `kept`, `removed`, `groups` and `pairs` are positions in the input file,
138
+ not the texts, so join it back to that file (line *n* of the `.txt`, row *n* of the `.csv`) to read it.
139
+ `pairs` carries one entry per matching pair and can be long on a large, repetitive corpus; `groups` is the
140
+ compact view. Use `--output PATH` when what you want is the cleaned text itself.
141
+
142
+ Output is UTF-8 whatever the console is set to, so piping a summary full of non-ASCII text is safe.
143
+
144
+ ## License
145
+
146
+ MIT
@@ -0,0 +1,49 @@
1
+ [build-system]
2
+ requires = ["hatchling>=1.27"]
3
+ build-backend = "hatchling.build"
4
+
5
+ [project]
6
+ name = "semantic-dedup"
7
+ version = "0.1.0"
8
+ description = "Remove passages that repeat the same meaning, not just the same words"
9
+ readme = "README.md"
10
+ requires-python = ">=3.9"
11
+ license = "MIT"
12
+ license-files = ["LICENSE"]
13
+ authors = [{ name = "Pranay Mahendrakar" }]
14
+ keywords = [
15
+ "deduplication",
16
+ "semantic-similarity",
17
+ "paraphrase",
18
+ "tfidf",
19
+ "minhash",
20
+ "embeddings",
21
+ "text-cleaning",
22
+ "nlp",
23
+ ]
24
+ classifiers = [
25
+ "Development Status :: 4 - Beta",
26
+ "Intended Audience :: Developers",
27
+ "Intended Audience :: Science/Research",
28
+ "Programming Language :: Python :: 3",
29
+ "Programming Language :: Python :: 3 :: Only",
30
+ "Operating System :: OS Independent",
31
+ "Topic :: Scientific/Engineering :: Artificial Intelligence",
32
+ "Topic :: Text Processing :: Linguistic",
33
+ ]
34
+ dependencies = [
35
+ "numpy>=1.23",
36
+ ]
37
+
38
+ [project.optional-dependencies]
39
+ dev = ["pytest>=7"]
40
+
41
+ [project.scripts]
42
+ semantic-dedup = "semantic_dedup.cli:main"
43
+
44
+ [project.urls]
45
+ Homepage = "https://pypi.org/project/semantic-dedup/"
46
+ Author = "https://pypi.org/user/pranaymahendrakar/"
47
+
48
+ [tool.hatch.build.targets.wheel]
49
+ packages = ["src/semantic_dedup"]
@@ -0,0 +1,36 @@
1
+ """semantic-dedup: remove passages that repeat the same meaning, not just the same words.
2
+
3
+ Quick use::
4
+
5
+ import semantic_dedup
6
+ result = semantic_dedup.dedupe(texts) # or a .txt / .csv / .jsonl path
7
+ print(result.summary())
8
+ clean = result.texts
9
+ """
10
+ from ._core import (
11
+ AUTO_MINHASH_ABOVE,
12
+ KEEP_MODES,
13
+ METHODS,
14
+ DedupeResult,
15
+ Deduper,
16
+ dedupe,
17
+ find_duplicates,
18
+ similarity,
19
+ )
20
+ from ._text import canonical_tokens, normalize
21
+
22
+ __version__ = "0.1.0"
23
+
24
+ __all__ = [
25
+ "DedupeResult",
26
+ "Deduper",
27
+ "dedupe",
28
+ "find_duplicates",
29
+ "similarity",
30
+ "canonical_tokens",
31
+ "normalize",
32
+ "METHODS",
33
+ "KEEP_MODES",
34
+ "AUTO_MINHASH_ABOVE",
35
+ "__version__",
36
+ ]
@@ -0,0 +1,7 @@
1
+ """Allow ``python -m semantic_dedup``."""
2
+ import sys
3
+
4
+ from .cli import main
5
+
6
+ if __name__ == "__main__":
7
+ sys.exit(main())