patent-ate 1.0.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- patent_ate-1.0.0/LICENSE +21 -0
- patent_ate-1.0.0/PKG-INFO +237 -0
- patent_ate-1.0.0/README.md +200 -0
- patent_ate-1.0.0/pyproject.toml +148 -0
- patent_ate-1.0.0/setup.cfg +4 -0
- patent_ate-1.0.0/src/patent_ate/__init__.py +24 -0
- patent_ate-1.0.0/src/patent_ate/__main__.py +340 -0
- patent_ate-1.0.0/src/patent_ate/ate_stop_surfaces.json +24 -0
- patent_ate-1.0.0/src/patent_ate/corpus.py +81 -0
- patent_ate-1.0.0/src/patent_ate/cvalue/__init__.py +5 -0
- patent_ate-1.0.0/src/patent_ate/cvalue/algebra.py +346 -0
- patent_ate-1.0.0/src/patent_ate/cvalue/exec.py +846 -0
- patent_ate-1.0.0/src/patent_ate/cvalue/facade.py +276 -0
- patent_ate-1.0.0/src/patent_ate/cvalue/index.py +552 -0
- patent_ate-1.0.0/src/patent_ate/cvalue/plan.py +483 -0
- patent_ate-1.0.0/src/patent_ate/cvalue/store.py +620 -0
- patent_ate-1.0.0/src/patent_ate/cvalue/text.py +81 -0
- patent_ate-1.0.0/src/patent_ate/extract.py +383 -0
- patent_ate-1.0.0/src/patent_ate/nlp.py +313 -0
- patent_ate-1.0.0/src/patent_ate/plan.py +67 -0
- patent_ate-1.0.0/src/patent_ate/ray_local.py +42 -0
- patent_ate-1.0.0/src/patent_ate/run.py +113 -0
- patent_ate-1.0.0/src/patent_ate/spec.py +69 -0
- patent_ate-1.0.0/src/patent_ate/termhood.py +404 -0
- patent_ate-1.0.0/src/patent_ate.egg-info/PKG-INFO +237 -0
- patent_ate-1.0.0/src/patent_ate.egg-info/SOURCES.txt +35 -0
- patent_ate-1.0.0/src/patent_ate.egg-info/dependency_links.txt +1 -0
- patent_ate-1.0.0/src/patent_ate.egg-info/entry_points.txt +2 -0
- patent_ate-1.0.0/src/patent_ate.egg-info/requires.txt +17 -0
- patent_ate-1.0.0/src/patent_ate.egg-info/top_level.txt +1 -0
- patent_ate-1.0.0/tests/test_cli.py +203 -0
- patent_ate-1.0.0/tests/test_cli_help.py +29 -0
- patent_ate-1.0.0/tests/test_cvalue_bench.py +92 -0
- patent_ate-1.0.0/tests/test_cvalue_indexed.py +875 -0
- patent_ate-1.0.0/tests/test_cvalue_stages.py +1855 -0
- patent_ate-1.0.0/tests/test_extract.py +1130 -0
- patent_ate-1.0.0/tests/test_termhood_artifact.py +408 -0
patent_ate-1.0.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Qubut
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,237 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: patent-ate
|
|
3
|
+
Version: 1.0.0
|
|
4
|
+
Summary: Extract ranked multiword terms from patent text into an immutable termhood table.
|
|
5
|
+
Author: Qubut
|
|
6
|
+
License: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/Qubut/patent-ate
|
|
8
|
+
Project-URL: Issues, https://github.com/Qubut/patent-ate/issues
|
|
9
|
+
Keywords: automatic-term-extraction,term-extraction,patents,nlp,c-value,termhood,spacy,duckdb,ibis,uspto
|
|
10
|
+
Classifier: Development Status :: 3 - Alpha
|
|
11
|
+
Classifier: Intended Audience :: Science/Research
|
|
12
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
13
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
14
|
+
Classifier: Topic :: Scientific/Engineering :: Information Analysis
|
|
15
|
+
Classifier: Topic :: Text Processing :: Linguistic
|
|
16
|
+
Requires-Python: <3.14,>=3.12
|
|
17
|
+
Description-Content-Type: text/markdown
|
|
18
|
+
License-File: LICENSE
|
|
19
|
+
Requires-Dist: pydantic>=2.10.0
|
|
20
|
+
Requires-Dist: pyyaml>=6.0.0
|
|
21
|
+
Requires-Dist: typer>=0.15.0
|
|
22
|
+
Requires-Dist: structlog>=25.5.0
|
|
23
|
+
Requires-Dist: polars>=1.31.0
|
|
24
|
+
Requires-Dist: pyarrow>=18.0.0
|
|
25
|
+
Requires-Dist: duckdb>=1.5.5
|
|
26
|
+
Requires-Dist: ibis-framework[duckdb]>=12.0.0
|
|
27
|
+
Requires-Dist: ray[data]>=2.50
|
|
28
|
+
Requires-Dist: spacy>=3.8.0
|
|
29
|
+
Requires-Dist: jate>=3.3.0
|
|
30
|
+
Requires-Dist: ahocorasick_rs==1.0.3
|
|
31
|
+
Requires-Dist: numpy>=2.0.0
|
|
32
|
+
Requires-Dist: pydivsufsort>=0.0.20
|
|
33
|
+
Requires-Dist: returns>=0.25.0
|
|
34
|
+
Requires-Dist: en-core-web-lg
|
|
35
|
+
Requires-Dist: sqlglot!=26.32.0,<30,>=23.4
|
|
36
|
+
Dynamic: license-file
|
|
37
|
+
|
|
38
|
+
# patent-ate
|
|
39
|
+
|
|
40
|
+
[](https://github.com/Qubut/patent-ate/actions/workflows/ci.yml)
|
|
41
|
+
[](https://www.python.org/downloads/)
|
|
42
|
+
[](LICENSE)
|
|
43
|
+
[](https://huggingface.co/datasets/Qubut/patent-ate-hupd)
|
|
44
|
+
|
|
45
|
+
patent-ate finds the technical phrases that actually appear in patent text
|
|
46
|
+
and scores each one by how much it behaves like a real term, not a generic
|
|
47
|
+
word.
|
|
48
|
+
|
|
49
|
+
Patents repeat short generic words such as "panel" and "method" inside
|
|
50
|
+
longer phrases. "Vehicle interior panel" contains "interior panel" and
|
|
51
|
+
"panel", so counting mentions ranks boilerplate highest. A lawyer or
|
|
52
|
+
engineer looking for the invention then sees "panel" at the top of the list.
|
|
53
|
+
|
|
54
|
+
C-value is a published score that down-weights those fragments by rewarding
|
|
55
|
+
longer phrases and subtracting the times a short string showed up only
|
|
56
|
+
inside a longer one (Frantzi, Ananiadou, and Mima, [*Automatic recognition
|
|
57
|
+
of multi-word terms: the C-value/NC-value method*](https://link.springer.com/article/10.1007/s007999900023),
|
|
58
|
+
2000). This tool computes that score on a collection of patents, following
|
|
59
|
+
the formula in [JATE](https://github.com/ziqizhang/jate) 3.3, an open-source
|
|
60
|
+
toolkit for ranking phrases, and writes a table of each phrase, its
|
|
61
|
+
C-value, and how many documents contain it.
|
|
62
|
+
|
|
63
|
+
C-value still keeps headings that stand on their own in many patents, such
|
|
64
|
+
as "another aspect". The table therefore also stores document frequency:
|
|
65
|
+
the number of patents that contain the phrase.
|
|
66
|
+
|
|
67
|
+
## Install and run
|
|
68
|
+
|
|
69
|
+
```text
|
|
70
|
+
uv add patent-ate
|
|
71
|
+
```
|
|
72
|
+
|
|
73
|
+
Until the first PyPI release, depend on the git repository:
|
|
74
|
+
|
|
75
|
+
```text
|
|
76
|
+
uv add 'patent-ate @ git+https://github.com/Qubut/patent-ate'
|
|
77
|
+
```
|
|
78
|
+
|
|
79
|
+
patent-ate reads a directory of JSON files, one patent object per file,
|
|
80
|
+
and looks for text in the `claims`, `abstract`, and `summary` fields.
|
|
81
|
+
Missing fields are skipped. Multi-record Hugging Face files (JSONL or
|
|
82
|
+
Parquet) are not read directly; write one JSON object per file first.
|
|
83
|
+
|
|
84
|
+
Sample the directory, extract phrases, score them, and write the table in
|
|
85
|
+
one job:
|
|
86
|
+
|
|
87
|
+
```text
|
|
88
|
+
patent-ate corpus \
|
|
89
|
+
--output ./termhood \
|
|
90
|
+
--input-dir ./patents \
|
|
91
|
+
--limit 1000 \
|
|
92
|
+
--extract-workers 0 \
|
|
93
|
+
--extract-block-rows 256
|
|
94
|
+
```
|
|
95
|
+
|
|
96
|
+
`--extract-workers 0` leaves one CPU free for Ray, the library that
|
|
97
|
+
schedules the extraction workers. Pass a positive count to run more
|
|
98
|
+
workers. `--config` is optional YAML for the spaCy English model (the
|
|
99
|
+
parser that finds noun phrases), DuckDB memory, and phrase matching;
|
|
100
|
+
package defaults apply when it is omitted.
|
|
101
|
+
|
|
102
|
+
Extract phrases first, then score without re-parsing the JSON:
|
|
103
|
+
|
|
104
|
+
```text
|
|
105
|
+
patent-ate extract --output ./termhood --input-dir ./patents --limit 1000
|
|
106
|
+
patent-ate score --extract ./termhood/extract --output ./termhood
|
|
107
|
+
```
|
|
108
|
+
|
|
109
|
+
`inspect` prints how many phrases a saved table contains. The output
|
|
110
|
+
directory can keep more than one scored table from later runs;
|
|
111
|
+
`list-generations` prints those files. The Python package runs the same
|
|
112
|
+
steps: extract a collection, score Parquet files, and open the table the
|
|
113
|
+
output directory currently selects.
|
|
114
|
+
|
|
115
|
+
## Evidence from the Harvard USPTO Patent Dataset
|
|
116
|
+
|
|
117
|
+
Suzgun, Melas-Kyriazi, Sarkar, Kominers, and Shieber (2022) released the
|
|
118
|
+
**Harvard USPTO Patent Dataset (HUPD)**: English-language US utility
|
|
119
|
+
applications whose JSON objects already carry `claims`, `abstract`, and
|
|
120
|
+
`summary`
|
|
121
|
+
([dataset](https://huggingface.co/datasets/HUPD/hupd),
|
|
122
|
+
[paper](https://arxiv.org/abs/2207.04043),
|
|
123
|
+
[site](https://patentdataset.org)). Scoring that collection with this
|
|
124
|
+
package produced 88,607,764 phrases across 4,518,254 applications. The
|
|
125
|
+
saved columns are the phrase (`key`), C-value (`c_value`), and document
|
|
126
|
+
frequency (`df`).
|
|
127
|
+
|
|
128
|
+
Most of that C-value sits in short strings. Two-word keys hold about half
|
|
129
|
+
of the positive mass, three-word keys most of the rest; unigrams barely
|
|
130
|
+
register and phrases of six words or more are a thin tail. Nested scoring
|
|
131
|
+
matters because the fragments that inflate a raw count are exactly those
|
|
132
|
+
two- and three-word shells.
|
|
133
|
+
|
|
134
|
+

|
|
135
|
+
|
|
136
|
+
Hyphenation and plural endings stay as written. The table also stores a
|
|
137
|
+
lowercase lookup spelling so either form can be found, which is why
|
|
138
|
+
"lithium ion battery" and "lithium-ion battery" are separate rows.
|
|
139
|
+
|
|
140
|
+
On that table, C-value and raw document frequency still agree on legal
|
|
141
|
+
headings. "Least a portion" leads C-value as a high-frequency claim
|
|
142
|
+
fragment that also stands alone; "detailed description" leads `df` as a
|
|
143
|
+
section title. The last column below is a combined rank: a high number is
|
|
144
|
+
a rare technical phrase; zero means the phrase is too common to keep as a
|
|
145
|
+
single rank.
|
|
146
|
+
|
|
147
|
+
| Phrase | C-value | Documents (`df`) | Combined rank |
|
|
148
|
+
| --- | ---: | ---: | ---: |
|
|
149
|
+
| `least a portion` | 2,345,468 | 333,608 | 0 |
|
|
150
|
+
| `another aspect` | 2,080,434 | 694,564 | 0 |
|
|
151
|
+
| `present disclosure` | 1,568,066 | 390,064 | 0 |
|
|
152
|
+
| `computer program product` | 1,381,133 | 154,388 | 0 |
|
|
153
|
+
| `detailed description` | 1,061,982 | 843,582 | 0 |
|
|
154
|
+
| `lithium ion battery` | 22,744 | 3,018 | 0 |
|
|
155
|
+
| `lithium-ion battery` | 9,028 | 1,756 | 71.5 |
|
|
156
|
+
| `sina molecule` | 86,557 | 317 | 108.7 |
|
|
157
|
+
|
|
158
|
+
Side by side, the two rankings share those headings: the left list still
|
|
159
|
+
opens with claim fragments such as "least a portion", and the right list
|
|
160
|
+
is section titles such as "detailed description".
|
|
161
|
+
|
|
162
|
+

|
|
163
|
+
|
|
164
|
+
To down-weight phrases that appear in most of the collection, the library
|
|
165
|
+
can multiply log(1 + C-value) by inverse document frequency, a weight that
|
|
166
|
+
shrinks as a phrase appears in more documents (Sparck Jones, 1972; Lucene
|
|
167
|
+
BM25-style). The product is zero when `df * df` exceeds the document count,
|
|
168
|
+
so a phrase that appears in more than about √N documents drops out of a
|
|
169
|
+
single ranked list. Almost every key in the histogram sits at small `df`;
|
|
170
|
+
the long tail past √N ≈ 2,126 is where those headings live, and about
|
|
171
|
+
8.65 million phrases (9.8%) hit the cutoff.
|
|
172
|
+
|
|
173
|
+

|
|
174
|
+
|
|
175
|
+
`lithium ion battery` (df 3,018) sits past that line and scores zero;
|
|
176
|
+
the hyphenated sibling `lithium-ion battery` (df 1,756) keeps a combined
|
|
177
|
+
rank of 71.5. The published parquet stores phrase, C-value, and document
|
|
178
|
+
frequency; multiply them in this library when you need one rank.
|
|
179
|
+
|
|
180
|
+
That product is what ranks rare technical compounds instead of those
|
|
181
|
+
headings: the left ranking still lists "least a portion", while the right
|
|
182
|
+
ranking lists phrases such as "rf network node" and
|
|
183
|
+
"polymer-anticancer agent conjugate".
|
|
184
|
+
|
|
185
|
+

|
|
186
|
+
|
|
187
|
+
## Get the published table
|
|
188
|
+
|
|
189
|
+
A 3,393-row preview on
|
|
190
|
+
[Qubut/patent-ate-hupd](https://huggingface.co/datasets/Qubut/patent-ate-hupd)
|
|
191
|
+
mixes C-value leaders, high-`df` headings, combined-rank leaders, phrases
|
|
192
|
+
of different lengths, and a random draw. Load that slice before fetching
|
|
193
|
+
the full parquet.
|
|
194
|
+
|
|
195
|
+
```python
|
|
196
|
+
from datasets import load_dataset
|
|
197
|
+
|
|
198
|
+
preview = load_dataset('Qubut/patent-ate-hupd', 'slice', split='preview')
|
|
199
|
+
```
|
|
200
|
+
|
|
201
|
+
```python
|
|
202
|
+
import polars as pl
|
|
203
|
+
|
|
204
|
+
pl.scan_parquet(
|
|
205
|
+
'hf://datasets/Qubut/patent-ate-hupd/slice/preview.parquet'
|
|
206
|
+
).head(20).collect()
|
|
207
|
+
```
|
|
208
|
+
|
|
209
|
+
```python
|
|
210
|
+
import duckdb
|
|
211
|
+
|
|
212
|
+
con = duckdb.connect('preview.duckdb') # download slice/preview.duckdb
|
|
213
|
+
con.sql("SELECT * FROM termhood_slice WHERE stratum = 'top_score' LIMIT 10")
|
|
214
|
+
```
|
|
215
|
+
|
|
216
|
+
`load_dataset('Qubut/patent-ate-hupd', split='termhood')` opens the full
|
|
217
|
+
table. The derived phrase table follows HUPD's license (CC BY-NC-SA 4.0);
|
|
218
|
+
this software is MIT.
|
|
219
|
+
|
|
220
|
+
## Development
|
|
221
|
+
|
|
222
|
+
In a local clone, Python 3.12 tools come from devenv:
|
|
223
|
+
|
|
224
|
+
```text
|
|
225
|
+
devenv shell -- uv sync --group dev --group test
|
|
226
|
+
devenv shell -- ruff check src tests
|
|
227
|
+
devenv shell -- ruff format --check src tests
|
|
228
|
+
devenv shell -- mypy src
|
|
229
|
+
devenv shell -- pytest
|
|
230
|
+
```
|
|
231
|
+
|
|
232
|
+
The CPU image `ghcr.io/qubut/patent-ate` has entrypoint `patent-ate`. Tags
|
|
233
|
+
and conventional-commit changelogs follow `v*` releases.
|
|
234
|
+
|
|
235
|
+
## License
|
|
236
|
+
|
|
237
|
+
[MIT](./LICENSE)
|
|
@@ -0,0 +1,200 @@
|
|
|
1
|
+
# patent-ate
|
|
2
|
+
|
|
3
|
+
[](https://github.com/Qubut/patent-ate/actions/workflows/ci.yml)
|
|
4
|
+
[](https://www.python.org/downloads/)
|
|
5
|
+
[](LICENSE)
|
|
6
|
+
[](https://huggingface.co/datasets/Qubut/patent-ate-hupd)
|
|
7
|
+
|
|
8
|
+
patent-ate finds the technical phrases that actually appear in patent text
|
|
9
|
+
and scores each one by how much it behaves like a real term, not a generic
|
|
10
|
+
word.
|
|
11
|
+
|
|
12
|
+
Patents repeat short generic words such as "panel" and "method" inside
|
|
13
|
+
longer phrases. "Vehicle interior panel" contains "interior panel" and
|
|
14
|
+
"panel", so counting mentions ranks boilerplate highest. A lawyer or
|
|
15
|
+
engineer looking for the invention then sees "panel" at the top of the list.
|
|
16
|
+
|
|
17
|
+
C-value is a published score that down-weights those fragments by rewarding
|
|
18
|
+
longer phrases and subtracting the times a short string showed up only
|
|
19
|
+
inside a longer one (Frantzi, Ananiadou, and Mima, [*Automatic recognition
|
|
20
|
+
of multi-word terms: the C-value/NC-value method*](https://link.springer.com/article/10.1007/s007999900023),
|
|
21
|
+
2000). This tool computes that score on a collection of patents, following
|
|
22
|
+
the formula in [JATE](https://github.com/ziqizhang/jate) 3.3, an open-source
|
|
23
|
+
toolkit for ranking phrases, and writes a table of each phrase, its
|
|
24
|
+
C-value, and how many documents contain it.
|
|
25
|
+
|
|
26
|
+
C-value still keeps headings that stand on their own in many patents, such
|
|
27
|
+
as "another aspect". The table therefore also stores document frequency:
|
|
28
|
+
the number of patents that contain the phrase.
|
|
29
|
+
|
|
30
|
+
## Install and run
|
|
31
|
+
|
|
32
|
+
```text
|
|
33
|
+
uv add patent-ate
|
|
34
|
+
```
|
|
35
|
+
|
|
36
|
+
Until the first PyPI release, depend on the git repository:
|
|
37
|
+
|
|
38
|
+
```text
|
|
39
|
+
uv add 'patent-ate @ git+https://github.com/Qubut/patent-ate'
|
|
40
|
+
```
|
|
41
|
+
|
|
42
|
+
patent-ate reads a directory of JSON files, one patent object per file,
|
|
43
|
+
and looks for text in the `claims`, `abstract`, and `summary` fields.
|
|
44
|
+
Missing fields are skipped. Multi-record Hugging Face files (JSONL or
|
|
45
|
+
Parquet) are not read directly; write one JSON object per file first.
|
|
46
|
+
|
|
47
|
+
Sample the directory, extract phrases, score them, and write the table in
|
|
48
|
+
one job:
|
|
49
|
+
|
|
50
|
+
```text
|
|
51
|
+
patent-ate corpus \
|
|
52
|
+
--output ./termhood \
|
|
53
|
+
--input-dir ./patents \
|
|
54
|
+
--limit 1000 \
|
|
55
|
+
--extract-workers 0 \
|
|
56
|
+
--extract-block-rows 256
|
|
57
|
+
```
|
|
58
|
+
|
|
59
|
+
`--extract-workers 0` leaves one CPU free for Ray, the library that
|
|
60
|
+
schedules the extraction workers. Pass a positive count to run more
|
|
61
|
+
workers. `--config` is optional YAML for the spaCy English model (the
|
|
62
|
+
parser that finds noun phrases), DuckDB memory, and phrase matching;
|
|
63
|
+
package defaults apply when it is omitted.
|
|
64
|
+
|
|
65
|
+
Extract phrases first, then score without re-parsing the JSON:
|
|
66
|
+
|
|
67
|
+
```text
|
|
68
|
+
patent-ate extract --output ./termhood --input-dir ./patents --limit 1000
|
|
69
|
+
patent-ate score --extract ./termhood/extract --output ./termhood
|
|
70
|
+
```
|
|
71
|
+
|
|
72
|
+
`inspect` prints how many phrases a saved table contains. The output
|
|
73
|
+
directory can keep more than one scored table from later runs;
|
|
74
|
+
`list-generations` prints those files. The Python package runs the same
|
|
75
|
+
steps: extract a collection, score Parquet files, and open the table the
|
|
76
|
+
output directory currently selects.
|
|
77
|
+
|
|
78
|
+
## Evidence from the Harvard USPTO Patent Dataset
|
|
79
|
+
|
|
80
|
+
Suzgun, Melas-Kyriazi, Sarkar, Kominers, and Shieber (2022) released the
|
|
81
|
+
**Harvard USPTO Patent Dataset (HUPD)**: English-language US utility
|
|
82
|
+
applications whose JSON objects already carry `claims`, `abstract`, and
|
|
83
|
+
`summary`
|
|
84
|
+
([dataset](https://huggingface.co/datasets/HUPD/hupd),
|
|
85
|
+
[paper](https://arxiv.org/abs/2207.04043),
|
|
86
|
+
[site](https://patentdataset.org)). Scoring that collection with this
|
|
87
|
+
package produced 88,607,764 phrases across 4,518,254 applications. The
|
|
88
|
+
saved columns are the phrase (`key`), C-value (`c_value`), and document
|
|
89
|
+
frequency (`df`).
|
|
90
|
+
|
|
91
|
+
Most of that C-value sits in short strings. Two-word keys hold about half
|
|
92
|
+
of the positive mass, three-word keys most of the rest; unigrams barely
|
|
93
|
+
register and phrases of six words or more are a thin tail. Nested scoring
|
|
94
|
+
matters because the fragments that inflate a raw count are exactly those
|
|
95
|
+
two- and three-word shells.
|
|
96
|
+
|
|
97
|
+

|
|
98
|
+
|
|
99
|
+
Hyphenation and plural endings stay as written. The table also stores a
|
|
100
|
+
lowercase lookup spelling so either form can be found, which is why
|
|
101
|
+
"lithium ion battery" and "lithium-ion battery" are separate rows.
|
|
102
|
+
|
|
103
|
+
On that table, C-value and raw document frequency still agree on legal
|
|
104
|
+
headings. "Least a portion" leads C-value as a high-frequency claim
|
|
105
|
+
fragment that also stands alone; "detailed description" leads `df` as a
|
|
106
|
+
section title. The last column below is a combined rank: a high number is
|
|
107
|
+
a rare technical phrase; zero means the phrase is too common to keep as a
|
|
108
|
+
single rank.
|
|
109
|
+
|
|
110
|
+
| Phrase | C-value | Documents (`df`) | Combined rank |
|
|
111
|
+
| --- | ---: | ---: | ---: |
|
|
112
|
+
| `least a portion` | 2,345,468 | 333,608 | 0 |
|
|
113
|
+
| `another aspect` | 2,080,434 | 694,564 | 0 |
|
|
114
|
+
| `present disclosure` | 1,568,066 | 390,064 | 0 |
|
|
115
|
+
| `computer program product` | 1,381,133 | 154,388 | 0 |
|
|
116
|
+
| `detailed description` | 1,061,982 | 843,582 | 0 |
|
|
117
|
+
| `lithium ion battery` | 22,744 | 3,018 | 0 |
|
|
118
|
+
| `lithium-ion battery` | 9,028 | 1,756 | 71.5 |
|
|
119
|
+
| `sina molecule` | 86,557 | 317 | 108.7 |
|
|
120
|
+
|
|
121
|
+
Side by side, the two rankings share those headings: the left list still
|
|
122
|
+
opens with claim fragments such as "least a portion", and the right list
|
|
123
|
+
is section titles such as "detailed description".
|
|
124
|
+
|
|
125
|
+

|
|
126
|
+
|
|
127
|
+
To down-weight phrases that appear in most of the collection, the library
|
|
128
|
+
can multiply log(1 + C-value) by inverse document frequency, a weight that
|
|
129
|
+
shrinks as a phrase appears in more documents (Sparck Jones, 1972; Lucene
|
|
130
|
+
BM25-style). The product is zero when `df * df` exceeds the document count,
|
|
131
|
+
so a phrase that appears in more than about √N documents drops out of a
|
|
132
|
+
single ranked list. Almost every key in the histogram sits at small `df`;
|
|
133
|
+
the long tail past √N ≈ 2,126 is where those headings live, and about
|
|
134
|
+
8.65 million phrases (9.8%) hit the cutoff.
|
|
135
|
+
|
|
136
|
+

|
|
137
|
+
|
|
138
|
+
`lithium ion battery` (df 3,018) sits past that line and scores zero;
|
|
139
|
+
the hyphenated sibling `lithium-ion battery` (df 1,756) keeps a combined
|
|
140
|
+
rank of 71.5. The published parquet stores phrase, C-value, and document
|
|
141
|
+
frequency; multiply them in this library when you need one rank.
|
|
142
|
+
|
|
143
|
+
That product is what ranks rare technical compounds instead of those
|
|
144
|
+
headings: the left ranking still lists "least a portion", while the right
|
|
145
|
+
ranking lists phrases such as "rf network node" and
|
|
146
|
+
"polymer-anticancer agent conjugate".
|
|
147
|
+
|
|
148
|
+

|
|
149
|
+
|
|
150
|
+
## Get the published table
|
|
151
|
+
|
|
152
|
+
A 3,393-row preview on
|
|
153
|
+
[Qubut/patent-ate-hupd](https://huggingface.co/datasets/Qubut/patent-ate-hupd)
|
|
154
|
+
mixes C-value leaders, high-`df` headings, combined-rank leaders, phrases
|
|
155
|
+
of different lengths, and a random draw. Load that slice before fetching
|
|
156
|
+
the full parquet.
|
|
157
|
+
|
|
158
|
+
```python
|
|
159
|
+
from datasets import load_dataset
|
|
160
|
+
|
|
161
|
+
preview = load_dataset('Qubut/patent-ate-hupd', 'slice', split='preview')
|
|
162
|
+
```
|
|
163
|
+
|
|
164
|
+
```python
|
|
165
|
+
import polars as pl
|
|
166
|
+
|
|
167
|
+
pl.scan_parquet(
|
|
168
|
+
'hf://datasets/Qubut/patent-ate-hupd/slice/preview.parquet'
|
|
169
|
+
).head(20).collect()
|
|
170
|
+
```
|
|
171
|
+
|
|
172
|
+
```python
|
|
173
|
+
import duckdb
|
|
174
|
+
|
|
175
|
+
con = duckdb.connect('preview.duckdb') # download slice/preview.duckdb
|
|
176
|
+
con.sql("SELECT * FROM termhood_slice WHERE stratum = 'top_score' LIMIT 10")
|
|
177
|
+
```
|
|
178
|
+
|
|
179
|
+
`load_dataset('Qubut/patent-ate-hupd', split='termhood')` opens the full
|
|
180
|
+
table. The derived phrase table follows HUPD's license (CC BY-NC-SA 4.0);
|
|
181
|
+
this software is MIT.
|
|
182
|
+
|
|
183
|
+
## Development
|
|
184
|
+
|
|
185
|
+
In a local clone, Python 3.12 tools come from devenv:
|
|
186
|
+
|
|
187
|
+
```text
|
|
188
|
+
devenv shell -- uv sync --group dev --group test
|
|
189
|
+
devenv shell -- ruff check src tests
|
|
190
|
+
devenv shell -- ruff format --check src tests
|
|
191
|
+
devenv shell -- mypy src
|
|
192
|
+
devenv shell -- pytest
|
|
193
|
+
```
|
|
194
|
+
|
|
195
|
+
The CPU image `ghcr.io/qubut/patent-ate` has entrypoint `patent-ate`. Tags
|
|
196
|
+
and conventional-commit changelogs follow `v*` releases.
|
|
197
|
+
|
|
198
|
+
## License
|
|
199
|
+
|
|
200
|
+
[MIT](./LICENSE)
|
|
@@ -0,0 +1,148 @@
|
|
|
1
|
+
[project]
|
|
2
|
+
name = "patent-ate"
|
|
3
|
+
version = "1.0.0"
|
|
4
|
+
description = "Extract ranked multiword terms from patent text into an immutable termhood table."
|
|
5
|
+
readme = "README.md"
|
|
6
|
+
requires-python = ">=3.12,<3.14"
|
|
7
|
+
authors = [{ name = "Qubut" }]
|
|
8
|
+
license = { text = "MIT" }
|
|
9
|
+
keywords = [
|
|
10
|
+
"automatic-term-extraction",
|
|
11
|
+
"term-extraction",
|
|
12
|
+
"patents",
|
|
13
|
+
"nlp",
|
|
14
|
+
"c-value",
|
|
15
|
+
"termhood",
|
|
16
|
+
"spacy",
|
|
17
|
+
"duckdb",
|
|
18
|
+
"ibis",
|
|
19
|
+
"uspto",
|
|
20
|
+
]
|
|
21
|
+
classifiers = [
|
|
22
|
+
"Development Status :: 3 - Alpha",
|
|
23
|
+
"Intended Audience :: Science/Research",
|
|
24
|
+
"License :: OSI Approved :: MIT License",
|
|
25
|
+
"Programming Language :: Python :: 3.12",
|
|
26
|
+
"Topic :: Scientific/Engineering :: Information Analysis",
|
|
27
|
+
"Topic :: Text Processing :: Linguistic",
|
|
28
|
+
]
|
|
29
|
+
dependencies = [
|
|
30
|
+
"pydantic>=2.10.0",
|
|
31
|
+
"pyyaml>=6.0.0",
|
|
32
|
+
"typer>=0.15.0",
|
|
33
|
+
"structlog>=25.5.0",
|
|
34
|
+
"polars>=1.31.0",
|
|
35
|
+
"pyarrow>=18.0.0",
|
|
36
|
+
"duckdb>=1.5.5",
|
|
37
|
+
"ibis-framework[duckdb]>=12.0.0",
|
|
38
|
+
"ray[data]>=2.50",
|
|
39
|
+
"spacy>=3.8.0",
|
|
40
|
+
"jate>=3.3.0",
|
|
41
|
+
"ahocorasick_rs==1.0.3",
|
|
42
|
+
"numpy>=2.0.0",
|
|
43
|
+
"pydivsufsort>=0.0.20",
|
|
44
|
+
"returns>=0.25.0",
|
|
45
|
+
"en-core-web-lg",
|
|
46
|
+
"sqlglot>=23.4,!=26.32.0,<30",
|
|
47
|
+
]
|
|
48
|
+
|
|
49
|
+
[dependency-groups]
|
|
50
|
+
dev = [
|
|
51
|
+
"mypy>=1.13.0",
|
|
52
|
+
"ruff>=0.11.0",
|
|
53
|
+
]
|
|
54
|
+
test = [
|
|
55
|
+
"pytest>=8.3.0",
|
|
56
|
+
"hypothesis>=6.0",
|
|
57
|
+
"en-core-web-sm",
|
|
58
|
+
]
|
|
59
|
+
|
|
60
|
+
[build-system]
|
|
61
|
+
requires = ["setuptools>=68.0", "wheel"]
|
|
62
|
+
build-backend = "setuptools.build_meta"
|
|
63
|
+
|
|
64
|
+
[tool.setuptools.packages.find]
|
|
65
|
+
where = ["src"]
|
|
66
|
+
include = ["patent_ate*"]
|
|
67
|
+
|
|
68
|
+
[tool.setuptools.package-data]
|
|
69
|
+
patent_ate = ["ate_stop_surfaces.json"]
|
|
70
|
+
|
|
71
|
+
[project.urls]
|
|
72
|
+
Homepage = "https://github.com/Qubut/patent-ate"
|
|
73
|
+
Issues = "https://github.com/Qubut/patent-ate/issues"
|
|
74
|
+
|
|
75
|
+
[project.scripts]
|
|
76
|
+
patent-ate = "patent_ate.__main__:main"
|
|
77
|
+
|
|
78
|
+
[tool.uv]
|
|
79
|
+
default-groups = ["dev", "test"]
|
|
80
|
+
|
|
81
|
+
[tool.uv.sources]
|
|
82
|
+
en-core-web-lg = { url = "https://github.com/explosion/spacy-models/releases/download/en_core_web_lg-3.8.0/en_core_web_lg-3.8.0-py3-none-any.whl" }
|
|
83
|
+
en-core-web-sm = { url = "https://github.com/explosion/spacy-models/releases/download/en_core_web_sm-3.8.0/en_core_web_sm-3.8.0-py3-none-any.whl" }
|
|
84
|
+
|
|
85
|
+
[tool.ruff]
|
|
86
|
+
src = ["src", "tests"]
|
|
87
|
+
preview = true
|
|
88
|
+
fix = true
|
|
89
|
+
target-version = "py312"
|
|
90
|
+
line-length = 100
|
|
91
|
+
|
|
92
|
+
[tool.ruff.format]
|
|
93
|
+
quote-style = "single"
|
|
94
|
+
docstring-code-format = true
|
|
95
|
+
|
|
96
|
+
[tool.ruff.lint]
|
|
97
|
+
select = [
|
|
98
|
+
"A","B","C4","C90","COM","D","DTZ","E","ERA","EXE","F","FBT","FLY","FURB",
|
|
99
|
+
"G","I","ICN","ISC","LOG","N","PERF","PIE","PL","PT","PTH","Q","RET","RSE",
|
|
100
|
+
"RUF","S","SIM","SLF","SLOT","T100","TRY","UP","W","YTT",
|
|
101
|
+
]
|
|
102
|
+
ignore = [
|
|
103
|
+
"A005","COM812","D100","D101","D104","D106","D107","D203","D212","D401","D404","D405",
|
|
104
|
+
"FBT003","ISC001","ISC003","N812","PLR09","PLR2004","PLR6301","TRY003",
|
|
105
|
+
]
|
|
106
|
+
|
|
107
|
+
[tool.ruff.lint.flake8-quotes]
|
|
108
|
+
inline-quotes = "single"
|
|
109
|
+
|
|
110
|
+
[tool.ruff.lint.pydocstyle]
|
|
111
|
+
convention = "google"
|
|
112
|
+
|
|
113
|
+
[tool.ruff.lint.mccabe]
|
|
114
|
+
max-complexity = 7
|
|
115
|
+
|
|
116
|
+
[tool.ruff.lint.per-file-ignores]
|
|
117
|
+
"tests/**/*.py" = ["S101","S105","S106","S404","S603","S607","D","PT011","PLC2701","SLF001","RUF012"]
|
|
118
|
+
|
|
119
|
+
[tool.pytest.ini_options]
|
|
120
|
+
testpaths = ["tests"]
|
|
121
|
+
python_files = "test_*.py"
|
|
122
|
+
xfail_strict = true
|
|
123
|
+
addopts = ["--strict-config","--strict-markers","-ra"]
|
|
124
|
+
markers = [
|
|
125
|
+
"benchmark: machine-local C-value timing gate",
|
|
126
|
+
]
|
|
127
|
+
filterwarnings = [
|
|
128
|
+
"error",
|
|
129
|
+
"ignore:.*corpus-level ranking algorithm.*:UserWarning",
|
|
130
|
+
"ignore::DeprecationWarning:ray.data.context",
|
|
131
|
+
"ignore:.*use_push_based_shuffle.*:DeprecationWarning",
|
|
132
|
+
"ignore:.*fetch_arrow_table.*is deprecated.*:DeprecationWarning",
|
|
133
|
+
"ignore:.*CUDA initialization.*:UserWarning",
|
|
134
|
+
"ignore:In Polars 2.0, the default behavior for `empty_as_null`:DeprecationWarning",
|
|
135
|
+
]
|
|
136
|
+
|
|
137
|
+
[tool.mypy]
|
|
138
|
+
files = ["src"]
|
|
139
|
+
python_version = "3.12"
|
|
140
|
+
strict = true
|
|
141
|
+
plugins = ["returns.contrib.mypy.returns_plugin"]
|
|
142
|
+
ignore_missing_imports = true
|
|
143
|
+
local_partial_types = true
|
|
144
|
+
warn_unreachable = true
|
|
145
|
+
enable_error_code = [
|
|
146
|
+
"truthy-bool","truthy-iterable","redundant-expr","unused-awaitable",
|
|
147
|
+
"possibly-undefined","redundant-self","unimported-reveal","deprecated",
|
|
148
|
+
]
|
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
"""CPU automatic term extraction for patent claim, abstract, and summary text."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from patent_ate.cvalue import score_term_parquet
|
|
6
|
+
from patent_ate.extract import (
|
|
7
|
+
corpus_termhood,
|
|
8
|
+
extract_corpus,
|
|
9
|
+
extract_parquet_parts,
|
|
10
|
+
write_termhood,
|
|
11
|
+
)
|
|
12
|
+
from patent_ate.spec import AteSpec
|
|
13
|
+
from patent_ate.termhood import TermhoodIndex, TermhoodStore
|
|
14
|
+
|
|
15
|
+
__all__ = [
|
|
16
|
+
'AteSpec',
|
|
17
|
+
'TermhoodIndex',
|
|
18
|
+
'TermhoodStore',
|
|
19
|
+
'corpus_termhood',
|
|
20
|
+
'extract_corpus',
|
|
21
|
+
'extract_parquet_parts',
|
|
22
|
+
'score_term_parquet',
|
|
23
|
+
'write_termhood',
|
|
24
|
+
]
|