patent-ate 1.0.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (37) hide show
  1. patent_ate-1.0.0/LICENSE +21 -0
  2. patent_ate-1.0.0/PKG-INFO +237 -0
  3. patent_ate-1.0.0/README.md +200 -0
  4. patent_ate-1.0.0/pyproject.toml +148 -0
  5. patent_ate-1.0.0/setup.cfg +4 -0
  6. patent_ate-1.0.0/src/patent_ate/__init__.py +24 -0
  7. patent_ate-1.0.0/src/patent_ate/__main__.py +340 -0
  8. patent_ate-1.0.0/src/patent_ate/ate_stop_surfaces.json +24 -0
  9. patent_ate-1.0.0/src/patent_ate/corpus.py +81 -0
  10. patent_ate-1.0.0/src/patent_ate/cvalue/__init__.py +5 -0
  11. patent_ate-1.0.0/src/patent_ate/cvalue/algebra.py +346 -0
  12. patent_ate-1.0.0/src/patent_ate/cvalue/exec.py +846 -0
  13. patent_ate-1.0.0/src/patent_ate/cvalue/facade.py +276 -0
  14. patent_ate-1.0.0/src/patent_ate/cvalue/index.py +552 -0
  15. patent_ate-1.0.0/src/patent_ate/cvalue/plan.py +483 -0
  16. patent_ate-1.0.0/src/patent_ate/cvalue/store.py +620 -0
  17. patent_ate-1.0.0/src/patent_ate/cvalue/text.py +81 -0
  18. patent_ate-1.0.0/src/patent_ate/extract.py +383 -0
  19. patent_ate-1.0.0/src/patent_ate/nlp.py +313 -0
  20. patent_ate-1.0.0/src/patent_ate/plan.py +67 -0
  21. patent_ate-1.0.0/src/patent_ate/ray_local.py +42 -0
  22. patent_ate-1.0.0/src/patent_ate/run.py +113 -0
  23. patent_ate-1.0.0/src/patent_ate/spec.py +69 -0
  24. patent_ate-1.0.0/src/patent_ate/termhood.py +404 -0
  25. patent_ate-1.0.0/src/patent_ate.egg-info/PKG-INFO +237 -0
  26. patent_ate-1.0.0/src/patent_ate.egg-info/SOURCES.txt +35 -0
  27. patent_ate-1.0.0/src/patent_ate.egg-info/dependency_links.txt +1 -0
  28. patent_ate-1.0.0/src/patent_ate.egg-info/entry_points.txt +2 -0
  29. patent_ate-1.0.0/src/patent_ate.egg-info/requires.txt +17 -0
  30. patent_ate-1.0.0/src/patent_ate.egg-info/top_level.txt +1 -0
  31. patent_ate-1.0.0/tests/test_cli.py +203 -0
  32. patent_ate-1.0.0/tests/test_cli_help.py +29 -0
  33. patent_ate-1.0.0/tests/test_cvalue_bench.py +92 -0
  34. patent_ate-1.0.0/tests/test_cvalue_indexed.py +875 -0
  35. patent_ate-1.0.0/tests/test_cvalue_stages.py +1855 -0
  36. patent_ate-1.0.0/tests/test_extract.py +1130 -0
  37. patent_ate-1.0.0/tests/test_termhood_artifact.py +408 -0
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Qubut
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,237 @@
1
+ Metadata-Version: 2.4
2
+ Name: patent-ate
3
+ Version: 1.0.0
4
+ Summary: Extract ranked multiword terms from patent text into an immutable termhood table.
5
+ Author: Qubut
6
+ License: MIT
7
+ Project-URL: Homepage, https://github.com/Qubut/patent-ate
8
+ Project-URL: Issues, https://github.com/Qubut/patent-ate/issues
9
+ Keywords: automatic-term-extraction,term-extraction,patents,nlp,c-value,termhood,spacy,duckdb,ibis,uspto
10
+ Classifier: Development Status :: 3 - Alpha
11
+ Classifier: Intended Audience :: Science/Research
12
+ Classifier: License :: OSI Approved :: MIT License
13
+ Classifier: Programming Language :: Python :: 3.12
14
+ Classifier: Topic :: Scientific/Engineering :: Information Analysis
15
+ Classifier: Topic :: Text Processing :: Linguistic
16
+ Requires-Python: <3.14,>=3.12
17
+ Description-Content-Type: text/markdown
18
+ License-File: LICENSE
19
+ Requires-Dist: pydantic>=2.10.0
20
+ Requires-Dist: pyyaml>=6.0.0
21
+ Requires-Dist: typer>=0.15.0
22
+ Requires-Dist: structlog>=25.5.0
23
+ Requires-Dist: polars>=1.31.0
24
+ Requires-Dist: pyarrow>=18.0.0
25
+ Requires-Dist: duckdb>=1.5.5
26
+ Requires-Dist: ibis-framework[duckdb]>=12.0.0
27
+ Requires-Dist: ray[data]>=2.50
28
+ Requires-Dist: spacy>=3.8.0
29
+ Requires-Dist: jate>=3.3.0
30
+ Requires-Dist: ahocorasick_rs==1.0.3
31
+ Requires-Dist: numpy>=2.0.0
32
+ Requires-Dist: pydivsufsort>=0.0.20
33
+ Requires-Dist: returns>=0.25.0
34
+ Requires-Dist: en-core-web-lg
35
+ Requires-Dist: sqlglot!=26.32.0,<30,>=23.4
36
+ Dynamic: license-file
37
+
38
+ # patent-ate
39
+
40
+ [![CI](https://github.com/Qubut/patent-ate/actions/workflows/ci.yml/badge.svg)](https://github.com/Qubut/patent-ate/actions/workflows/ci.yml)
41
+ [![Python 3.12](https://img.shields.io/badge/python-3.12-3776AB.svg)](https://www.python.org/downloads/)
42
+ [![License: MIT](https://img.shields.io/badge/license-MIT-green.svg)](LICENSE)
43
+ [![Hugging Face](https://img.shields.io/badge/dataset-Qubut%2Fpatent--ate--hupd-yellow.svg)](https://huggingface.co/datasets/Qubut/patent-ate-hupd)
44
+
45
+ patent-ate finds the technical phrases that actually appear in patent text
46
+ and scores each one by how much it behaves like a real term, not a generic
47
+ word.
48
+
49
+ Patents repeat short generic words such as "panel" and "method" inside
50
+ longer phrases. "Vehicle interior panel" contains "interior panel" and
51
+ "panel", so counting mentions ranks boilerplate highest. A lawyer or
52
+ engineer looking for the invention then sees "panel" at the top of the list.
53
+
54
+ C-value is a published score that down-weights those fragments by rewarding
55
+ longer phrases and subtracting the times a short string showed up only
56
+ inside a longer one (Frantzi, Ananiadou, and Mima, [*Automatic recognition
57
+ of multi-word terms: the C-value/NC-value method*](https://link.springer.com/article/10.1007/s007999900023),
58
+ 2000). This tool computes that score on a collection of patents, following
59
+ the formula in [JATE](https://github.com/ziqizhang/jate) 3.3, an open-source
60
+ toolkit for ranking phrases, and writes a table of each phrase, its
61
+ C-value, and how many documents contain it.
62
+
63
+ C-value still keeps headings that stand on their own in many patents, such
64
+ as "another aspect". The table therefore also stores document frequency:
65
+ the number of patents that contain the phrase.
66
+
67
+ ## Install and run
68
+
69
+ ```text
70
+ uv add patent-ate
71
+ ```
72
+
73
+ Until the first PyPI release, depend on the git repository:
74
+
75
+ ```text
76
+ uv add 'patent-ate @ git+https://github.com/Qubut/patent-ate'
77
+ ```
78
+
79
+ patent-ate reads a directory of JSON files, one patent object per file,
80
+ and looks for text in the `claims`, `abstract`, and `summary` fields.
81
+ Missing fields are skipped. Multi-record Hugging Face files (JSONL or
82
+ Parquet) are not read directly; write one JSON object per file first.
83
+
84
+ Sample the directory, extract phrases, score them, and write the table in
85
+ one job:
86
+
87
+ ```text
88
+ patent-ate corpus \
89
+ --output ./termhood \
90
+ --input-dir ./patents \
91
+ --limit 1000 \
92
+ --extract-workers 0 \
93
+ --extract-block-rows 256
94
+ ```
95
+
96
+ `--extract-workers 0` leaves one CPU free for Ray, the library that
97
+ schedules the extraction workers. Pass a positive count to run more
98
+ workers. `--config` is optional YAML for the spaCy English model (the
99
+ parser that finds noun phrases), DuckDB memory, and phrase matching;
100
+ package defaults apply when it is omitted.
101
+
102
+ Extract phrases first, then score without re-parsing the JSON:
103
+
104
+ ```text
105
+ patent-ate extract --output ./termhood --input-dir ./patents --limit 1000
106
+ patent-ate score --extract ./termhood/extract --output ./termhood
107
+ ```
108
+
109
+ `inspect` prints how many phrases a saved table contains. The output
110
+ directory can keep more than one scored table from later runs;
111
+ `list-generations` prints those files. The Python package runs the same
112
+ steps: extract a collection, score Parquet files, and open the table the
113
+ output directory currently selects.
114
+
115
+ ## Evidence from the Harvard USPTO Patent Dataset
116
+
117
+ Suzgun, Melas-Kyriazi, Sarkar, Kominers, and Shieber (2022) released the
118
+ **Harvard USPTO Patent Dataset (HUPD)**: English-language US utility
119
+ applications whose JSON objects already carry `claims`, `abstract`, and
120
+ `summary`
121
+ ([dataset](https://huggingface.co/datasets/HUPD/hupd),
122
+ [paper](https://arxiv.org/abs/2207.04043),
123
+ [site](https://patentdataset.org)). Scoring that collection with this
124
+ package produced 88,607,764 phrases across 4,518,254 applications. The
125
+ saved columns are the phrase (`key`), C-value (`c_value`), and document
126
+ frequency (`df`).
127
+
128
+ Most of that C-value sits in short strings. Two-word keys hold about half
129
+ of the positive mass, three-word keys most of the rest; unigrams barely
130
+ register and phrases of six words or more are a thin tail. Nested scoring
131
+ matters because the fragments that inflate a raw count are exactly those
132
+ two- and three-word shells.
133
+
134
+ ![Bar chart of nested C-value mass by number of whitespace tokens](docs/figures/c-mass-by-length.png)
135
+
136
+ Hyphenation and plural endings stay as written. The table also stores a
137
+ lowercase lookup spelling so either form can be found, which is why
138
+ "lithium ion battery" and "lithium-ion battery" are separate rows.
139
+
140
+ On that table, C-value and raw document frequency still agree on legal
141
+ headings. "Least a portion" leads C-value as a high-frequency claim
142
+ fragment that also stands alone; "detailed description" leads `df` as a
143
+ section title. The last column below is a combined rank: a high number is
144
+ a rare technical phrase; zero means the phrase is too common to keep as a
145
+ single rank.
146
+
147
+ | Phrase | C-value | Documents (`df`) | Combined rank |
148
+ | --- | ---: | ---: | ---: |
149
+ | `least a portion` | 2,345,468 | 333,608 | 0 |
150
+ | `another aspect` | 2,080,434 | 694,564 | 0 |
151
+ | `present disclosure` | 1,568,066 | 390,064 | 0 |
152
+ | `computer program product` | 1,381,133 | 154,388 | 0 |
153
+ | `detailed description` | 1,061,982 | 843,582 | 0 |
154
+ | `lithium ion battery` | 22,744 | 3,018 | 0 |
155
+ | `lithium-ion battery` | 9,028 | 1,756 | 71.5 |
156
+ | `sina molecule` | 86,557 | 317 | 108.7 |
157
+
158
+ Side by side, the two rankings share those headings: the left list still
159
+ opens with claim fragments such as "least a portion", and the right list
160
+ is section titles such as "detailed description".
161
+
162
+ ![Side-by-side bars of C-value leaders and document-frequency leaders](docs/figures/top-c-vs-top-df.png)
163
+
164
+ To down-weight phrases that appear in most of the collection, the library
165
+ can multiply log(1 + C-value) by inverse document frequency, a weight that
166
+ shrinks as a phrase appears in more documents (Sparck Jones, 1972; Lucene
167
+ BM25-style). The product is zero when `df * df` exceeds the document count,
168
+ so a phrase that appears in more than about √N documents drops out of a
169
+ single ranked list. Almost every key in the histogram sits at small `df`;
170
+ the long tail past √N ≈ 2,126 is where those headings live, and about
171
+ 8.65 million phrases (9.8%) hit the cutoff.
172
+
173
+ ![Log-log histogram of document frequency with a line at the square root of the document count](docs/figures/df-tail.png)
174
+
175
+ `lithium ion battery` (df 3,018) sits past that line and scores zero;
176
+ the hyphenated sibling `lithium-ion battery` (df 1,756) keeps a combined
177
+ rank of 71.5. The published parquet stores phrase, C-value, and document
178
+ frequency; multiply them in this library when you need one rank.
179
+
180
+ That product is what ranks rare technical compounds instead of those
181
+ headings: the left ranking still lists "least a portion", while the right
182
+ ranking lists phrases such as "rf network node" and
183
+ "polymer-anticancer agent conjugate".
184
+
185
+ ![Side-by-side bars of C-value leaders and combined-rank leaders](docs/figures/c-vs-product-score.png)
186
+
187
+ ## Get the published table
188
+
189
+ A 3,393-row preview on
190
+ [Qubut/patent-ate-hupd](https://huggingface.co/datasets/Qubut/patent-ate-hupd)
191
+ mixes C-value leaders, high-`df` headings, combined-rank leaders, phrases
192
+ of different lengths, and a random draw. Load that slice before fetching
193
+ the full parquet.
194
+
195
+ ```python
196
+ from datasets import load_dataset
197
+
198
+ preview = load_dataset('Qubut/patent-ate-hupd', 'slice', split='preview')
199
+ ```
200
+
201
+ ```python
202
+ import polars as pl
203
+
204
+ pl.scan_parquet(
205
+ 'hf://datasets/Qubut/patent-ate-hupd/slice/preview.parquet'
206
+ ).head(20).collect()
207
+ ```
208
+
209
+ ```python
210
+ import duckdb
211
+
212
+ con = duckdb.connect('preview.duckdb') # download slice/preview.duckdb
213
+ con.sql("SELECT * FROM termhood_slice WHERE stratum = 'top_score' LIMIT 10")
214
+ ```
215
+
216
+ `load_dataset('Qubut/patent-ate-hupd', split='termhood')` opens the full
217
+ table. The derived phrase table follows HUPD's license (CC BY-NC-SA 4.0);
218
+ this software is MIT.
219
+
220
+ ## Development
221
+
222
+ In a local clone, Python 3.12 tools come from devenv:
223
+
224
+ ```text
225
+ devenv shell -- uv sync --group dev --group test
226
+ devenv shell -- ruff check src tests
227
+ devenv shell -- ruff format --check src tests
228
+ devenv shell -- mypy src
229
+ devenv shell -- pytest
230
+ ```
231
+
232
+ The CPU image `ghcr.io/qubut/patent-ate` has entrypoint `patent-ate`. Tags
233
+ and conventional-commit changelogs follow `v*` releases.
234
+
235
+ ## License
236
+
237
+ [MIT](./LICENSE)
@@ -0,0 +1,200 @@
1
+ # patent-ate
2
+
3
+ [![CI](https://github.com/Qubut/patent-ate/actions/workflows/ci.yml/badge.svg)](https://github.com/Qubut/patent-ate/actions/workflows/ci.yml)
4
+ [![Python 3.12](https://img.shields.io/badge/python-3.12-3776AB.svg)](https://www.python.org/downloads/)
5
+ [![License: MIT](https://img.shields.io/badge/license-MIT-green.svg)](LICENSE)
6
+ [![Hugging Face](https://img.shields.io/badge/dataset-Qubut%2Fpatent--ate--hupd-yellow.svg)](https://huggingface.co/datasets/Qubut/patent-ate-hupd)
7
+
8
+ patent-ate finds the technical phrases that actually appear in patent text
9
+ and scores each one by how much it behaves like a real term, not a generic
10
+ word.
11
+
12
+ Patents repeat short generic words such as "panel" and "method" inside
13
+ longer phrases. "Vehicle interior panel" contains "interior panel" and
14
+ "panel", so counting mentions ranks boilerplate highest. A lawyer or
15
+ engineer looking for the invention then sees "panel" at the top of the list.
16
+
17
+ C-value is a published score that down-weights those fragments by rewarding
18
+ longer phrases and subtracting the times a short string showed up only
19
+ inside a longer one (Frantzi, Ananiadou, and Mima, [*Automatic recognition
20
+ of multi-word terms: the C-value/NC-value method*](https://link.springer.com/article/10.1007/s007999900023),
21
+ 2000). This tool computes that score on a collection of patents, following
22
+ the formula in [JATE](https://github.com/ziqizhang/jate) 3.3, an open-source
23
+ toolkit for ranking phrases, and writes a table of each phrase, its
24
+ C-value, and how many documents contain it.
25
+
26
+ C-value still keeps headings that stand on their own in many patents, such
27
+ as "another aspect". The table therefore also stores document frequency:
28
+ the number of patents that contain the phrase.
29
+
30
+ ## Install and run
31
+
32
+ ```text
33
+ uv add patent-ate
34
+ ```
35
+
36
+ Until the first PyPI release, depend on the git repository:
37
+
38
+ ```text
39
+ uv add 'patent-ate @ git+https://github.com/Qubut/patent-ate'
40
+ ```
41
+
42
+ patent-ate reads a directory of JSON files, one patent object per file,
43
+ and looks for text in the `claims`, `abstract`, and `summary` fields.
44
+ Missing fields are skipped. Multi-record Hugging Face files (JSONL or
45
+ Parquet) are not read directly; write one JSON object per file first.
46
+
47
+ Sample the directory, extract phrases, score them, and write the table in
48
+ one job:
49
+
50
+ ```text
51
+ patent-ate corpus \
52
+ --output ./termhood \
53
+ --input-dir ./patents \
54
+ --limit 1000 \
55
+ --extract-workers 0 \
56
+ --extract-block-rows 256
57
+ ```
58
+
59
+ `--extract-workers 0` leaves one CPU free for Ray, the library that
60
+ schedules the extraction workers. Pass a positive count to run more
61
+ workers. `--config` is optional YAML for the spaCy English model (the
62
+ parser that finds noun phrases), DuckDB memory, and phrase matching;
63
+ package defaults apply when it is omitted.
64
+
65
+ Extract phrases first, then score without re-parsing the JSON:
66
+
67
+ ```text
68
+ patent-ate extract --output ./termhood --input-dir ./patents --limit 1000
69
+ patent-ate score --extract ./termhood/extract --output ./termhood
70
+ ```
71
+
72
+ `inspect` prints how many phrases a saved table contains. The output
73
+ directory can keep more than one scored table from later runs;
74
+ `list-generations` prints those files. The Python package runs the same
75
+ steps: extract a collection, score Parquet files, and open the table the
76
+ output directory currently selects.
77
+
78
+ ## Evidence from the Harvard USPTO Patent Dataset
79
+
80
+ Suzgun, Melas-Kyriazi, Sarkar, Kominers, and Shieber (2022) released the
81
+ **Harvard USPTO Patent Dataset (HUPD)**: English-language US utility
82
+ applications whose JSON objects already carry `claims`, `abstract`, and
83
+ `summary`
84
+ ([dataset](https://huggingface.co/datasets/HUPD/hupd),
85
+ [paper](https://arxiv.org/abs/2207.04043),
86
+ [site](https://patentdataset.org)). Scoring that collection with this
87
+ package produced 88,607,764 phrases across 4,518,254 applications. The
88
+ saved columns are the phrase (`key`), C-value (`c_value`), and document
89
+ frequency (`df`).
90
+
91
+ Most of that C-value sits in short strings. Two-word keys hold about half
92
+ of the positive mass, three-word keys most of the rest; unigrams barely
93
+ register and phrases of six words or more are a thin tail. Nested scoring
94
+ matters because the fragments that inflate a raw count are exactly those
95
+ two- and three-word shells.
96
+
97
+ ![Bar chart of nested C-value mass by number of whitespace tokens](docs/figures/c-mass-by-length.png)
98
+
99
+ Hyphenation and plural endings stay as written. The table also stores a
100
+ lowercase lookup spelling so either form can be found, which is why
101
+ "lithium ion battery" and "lithium-ion battery" are separate rows.
102
+
103
+ On that table, C-value and raw document frequency still agree on legal
104
+ headings. "Least a portion" leads C-value as a high-frequency claim
105
+ fragment that also stands alone; "detailed description" leads `df` as a
106
+ section title. The last column below is a combined rank: a high number is
107
+ a rare technical phrase; zero means the phrase is too common to keep as a
108
+ single rank.
109
+
110
+ | Phrase | C-value | Documents (`df`) | Combined rank |
111
+ | --- | ---: | ---: | ---: |
112
+ | `least a portion` | 2,345,468 | 333,608 | 0 |
113
+ | `another aspect` | 2,080,434 | 694,564 | 0 |
114
+ | `present disclosure` | 1,568,066 | 390,064 | 0 |
115
+ | `computer program product` | 1,381,133 | 154,388 | 0 |
116
+ | `detailed description` | 1,061,982 | 843,582 | 0 |
117
+ | `lithium ion battery` | 22,744 | 3,018 | 0 |
118
+ | `lithium-ion battery` | 9,028 | 1,756 | 71.5 |
119
+ | `sina molecule` | 86,557 | 317 | 108.7 |
120
+
121
+ Side by side, the two rankings share those headings: the left list still
122
+ opens with claim fragments such as "least a portion", and the right list
123
+ is section titles such as "detailed description".
124
+
125
+ ![Side-by-side bars of C-value leaders and document-frequency leaders](docs/figures/top-c-vs-top-df.png)
126
+
127
+ To down-weight phrases that appear in most of the collection, the library
128
+ can multiply log(1 + C-value) by inverse document frequency, a weight that
129
+ shrinks as a phrase appears in more documents (Sparck Jones, 1972; Lucene
130
+ BM25-style). The product is zero when `df * df` exceeds the document count,
131
+ so a phrase that appears in more than about √N documents drops out of a
132
+ single ranked list. Almost every key in the histogram sits at small `df`;
133
+ the long tail past √N ≈ 2,126 is where those headings live, and about
134
+ 8.65 million phrases (9.8%) hit the cutoff.
135
+
136
+ ![Log-log histogram of document frequency with a line at the square root of the document count](docs/figures/df-tail.png)
137
+
138
+ `lithium ion battery` (df 3,018) sits past that line and scores zero;
139
+ the hyphenated sibling `lithium-ion battery` (df 1,756) keeps a combined
140
+ rank of 71.5. The published parquet stores phrase, C-value, and document
141
+ frequency; multiply them in this library when you need one rank.
142
+
143
+ That product is what ranks rare technical compounds instead of those
144
+ headings: the left ranking still lists "least a portion", while the right
145
+ ranking lists phrases such as "rf network node" and
146
+ "polymer-anticancer agent conjugate".
147
+
148
+ ![Side-by-side bars of C-value leaders and combined-rank leaders](docs/figures/c-vs-product-score.png)
149
+
150
+ ## Get the published table
151
+
152
+ A 3,393-row preview on
153
+ [Qubut/patent-ate-hupd](https://huggingface.co/datasets/Qubut/patent-ate-hupd)
154
+ mixes C-value leaders, high-`df` headings, combined-rank leaders, phrases
155
+ of different lengths, and a random draw. Load that slice before fetching
156
+ the full parquet.
157
+
158
+ ```python
159
+ from datasets import load_dataset
160
+
161
+ preview = load_dataset('Qubut/patent-ate-hupd', 'slice', split='preview')
162
+ ```
163
+
164
+ ```python
165
+ import polars as pl
166
+
167
+ pl.scan_parquet(
168
+ 'hf://datasets/Qubut/patent-ate-hupd/slice/preview.parquet'
169
+ ).head(20).collect()
170
+ ```
171
+
172
+ ```python
173
+ import duckdb
174
+
175
+ con = duckdb.connect('preview.duckdb') # download slice/preview.duckdb
176
+ con.sql("SELECT * FROM termhood_slice WHERE stratum = 'top_score' LIMIT 10")
177
+ ```
178
+
179
+ `load_dataset('Qubut/patent-ate-hupd', split='termhood')` opens the full
180
+ table. The derived phrase table follows HUPD's license (CC BY-NC-SA 4.0);
181
+ this software is MIT.
182
+
183
+ ## Development
184
+
185
+ In a local clone, Python 3.12 tools come from devenv:
186
+
187
+ ```text
188
+ devenv shell -- uv sync --group dev --group test
189
+ devenv shell -- ruff check src tests
190
+ devenv shell -- ruff format --check src tests
191
+ devenv shell -- mypy src
192
+ devenv shell -- pytest
193
+ ```
194
+
195
+ The CPU image `ghcr.io/qubut/patent-ate` has entrypoint `patent-ate`. Tags
196
+ and conventional-commit changelogs follow `v*` releases.
197
+
198
+ ## License
199
+
200
+ [MIT](./LICENSE)
@@ -0,0 +1,148 @@
1
+ [project]
2
+ name = "patent-ate"
3
+ version = "1.0.0"
4
+ description = "Extract ranked multiword terms from patent text into an immutable termhood table."
5
+ readme = "README.md"
6
+ requires-python = ">=3.12,<3.14"
7
+ authors = [{ name = "Qubut" }]
8
+ license = { text = "MIT" }
9
+ keywords = [
10
+ "automatic-term-extraction",
11
+ "term-extraction",
12
+ "patents",
13
+ "nlp",
14
+ "c-value",
15
+ "termhood",
16
+ "spacy",
17
+ "duckdb",
18
+ "ibis",
19
+ "uspto",
20
+ ]
21
+ classifiers = [
22
+ "Development Status :: 3 - Alpha",
23
+ "Intended Audience :: Science/Research",
24
+ "License :: OSI Approved :: MIT License",
25
+ "Programming Language :: Python :: 3.12",
26
+ "Topic :: Scientific/Engineering :: Information Analysis",
27
+ "Topic :: Text Processing :: Linguistic",
28
+ ]
29
+ dependencies = [
30
+ "pydantic>=2.10.0",
31
+ "pyyaml>=6.0.0",
32
+ "typer>=0.15.0",
33
+ "structlog>=25.5.0",
34
+ "polars>=1.31.0",
35
+ "pyarrow>=18.0.0",
36
+ "duckdb>=1.5.5",
37
+ "ibis-framework[duckdb]>=12.0.0",
38
+ "ray[data]>=2.50",
39
+ "spacy>=3.8.0",
40
+ "jate>=3.3.0",
41
+ "ahocorasick_rs==1.0.3",
42
+ "numpy>=2.0.0",
43
+ "pydivsufsort>=0.0.20",
44
+ "returns>=0.25.0",
45
+ "en-core-web-lg",
46
+ "sqlglot>=23.4,!=26.32.0,<30",
47
+ ]
48
+
49
+ [dependency-groups]
50
+ dev = [
51
+ "mypy>=1.13.0",
52
+ "ruff>=0.11.0",
53
+ ]
54
+ test = [
55
+ "pytest>=8.3.0",
56
+ "hypothesis>=6.0",
57
+ "en-core-web-sm",
58
+ ]
59
+
60
+ [build-system]
61
+ requires = ["setuptools>=68.0", "wheel"]
62
+ build-backend = "setuptools.build_meta"
63
+
64
+ [tool.setuptools.packages.find]
65
+ where = ["src"]
66
+ include = ["patent_ate*"]
67
+
68
+ [tool.setuptools.package-data]
69
+ patent_ate = ["ate_stop_surfaces.json"]
70
+
71
+ [project.urls]
72
+ Homepage = "https://github.com/Qubut/patent-ate"
73
+ Issues = "https://github.com/Qubut/patent-ate/issues"
74
+
75
+ [project.scripts]
76
+ patent-ate = "patent_ate.__main__:main"
77
+
78
+ [tool.uv]
79
+ default-groups = ["dev", "test"]
80
+
81
+ [tool.uv.sources]
82
+ en-core-web-lg = { url = "https://github.com/explosion/spacy-models/releases/download/en_core_web_lg-3.8.0/en_core_web_lg-3.8.0-py3-none-any.whl" }
83
+ en-core-web-sm = { url = "https://github.com/explosion/spacy-models/releases/download/en_core_web_sm-3.8.0/en_core_web_sm-3.8.0-py3-none-any.whl" }
84
+
85
+ [tool.ruff]
86
+ src = ["src", "tests"]
87
+ preview = true
88
+ fix = true
89
+ target-version = "py312"
90
+ line-length = 100
91
+
92
+ [tool.ruff.format]
93
+ quote-style = "single"
94
+ docstring-code-format = true
95
+
96
+ [tool.ruff.lint]
97
+ select = [
98
+ "A","B","C4","C90","COM","D","DTZ","E","ERA","EXE","F","FBT","FLY","FURB",
99
+ "G","I","ICN","ISC","LOG","N","PERF","PIE","PL","PT","PTH","Q","RET","RSE",
100
+ "RUF","S","SIM","SLF","SLOT","T100","TRY","UP","W","YTT",
101
+ ]
102
+ ignore = [
103
+ "A005","COM812","D100","D101","D104","D106","D107","D203","D212","D401","D404","D405",
104
+ "FBT003","ISC001","ISC003","N812","PLR09","PLR2004","PLR6301","TRY003",
105
+ ]
106
+
107
+ [tool.ruff.lint.flake8-quotes]
108
+ inline-quotes = "single"
109
+
110
+ [tool.ruff.lint.pydocstyle]
111
+ convention = "google"
112
+
113
+ [tool.ruff.lint.mccabe]
114
+ max-complexity = 7
115
+
116
+ [tool.ruff.lint.per-file-ignores]
117
+ "tests/**/*.py" = ["S101","S105","S106","S404","S603","S607","D","PT011","PLC2701","SLF001","RUF012"]
118
+
119
+ [tool.pytest.ini_options]
120
+ testpaths = ["tests"]
121
+ python_files = "test_*.py"
122
+ xfail_strict = true
123
+ addopts = ["--strict-config","--strict-markers","-ra"]
124
+ markers = [
125
+ "benchmark: machine-local C-value timing gate",
126
+ ]
127
+ filterwarnings = [
128
+ "error",
129
+ "ignore:.*corpus-level ranking algorithm.*:UserWarning",
130
+ "ignore::DeprecationWarning:ray.data.context",
131
+ "ignore:.*use_push_based_shuffle.*:DeprecationWarning",
132
+ "ignore:.*fetch_arrow_table.*is deprecated.*:DeprecationWarning",
133
+ "ignore:.*CUDA initialization.*:UserWarning",
134
+ "ignore:In Polars 2.0, the default behavior for `empty_as_null`:DeprecationWarning",
135
+ ]
136
+
137
+ [tool.mypy]
138
+ files = ["src"]
139
+ python_version = "3.12"
140
+ strict = true
141
+ plugins = ["returns.contrib.mypy.returns_plugin"]
142
+ ignore_missing_imports = true
143
+ local_partial_types = true
144
+ warn_unreachable = true
145
+ enable_error_code = [
146
+ "truthy-bool","truthy-iterable","redundant-expr","unused-awaitable",
147
+ "possibly-undefined","redundant-self","unimported-reveal","deprecated",
148
+ ]
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+
@@ -0,0 +1,24 @@
1
+ """CPU automatic term extraction for patent claim, abstract, and summary text."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from patent_ate.cvalue import score_term_parquet
6
+ from patent_ate.extract import (
7
+ corpus_termhood,
8
+ extract_corpus,
9
+ extract_parquet_parts,
10
+ write_termhood,
11
+ )
12
+ from patent_ate.spec import AteSpec
13
+ from patent_ate.termhood import TermhoodIndex, TermhoodStore
14
+
15
+ __all__ = [
16
+ 'AteSpec',
17
+ 'TermhoodIndex',
18
+ 'TermhoodStore',
19
+ 'corpus_termhood',
20
+ 'extract_corpus',
21
+ 'extract_parquet_parts',
22
+ 'score_term_parquet',
23
+ 'write_termhood',
24
+ ]