structflo-ner 0.6.0__tar.gz → 0.7.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- structflo_ner-0.6.0/README.md → structflo_ner-0.7.0/PKG-INFO +25 -3
- structflo_ner-0.6.0/PKG-INFO → structflo_ner-0.7.0/README.md +12 -16
- {structflo_ner-0.6.0 → structflo_ner-0.7.0}/pyproject.toml +1 -1
- {structflo_ner-0.6.0 → structflo_ner-0.7.0}/structflo/ner/__init__.py +1 -1
- {structflo_ner-0.6.0 → structflo_ner-0.7.0}/structflo/ner/_examples.py +96 -0
- {structflo_ner-0.6.0 → structflo_ner-0.7.0}/structflo/ner/_mapping.py +4 -0
- {structflo_ner-0.6.0 → structflo_ner-0.7.0}/structflo/ner/_prompts.py +45 -8
- {structflo_ner-0.6.0 → structflo_ner-0.7.0}/structflo/ner/extractor.py +69 -2
- {structflo_ner-0.6.0 → structflo_ner-0.7.0}/tests/test_extractor.py +96 -1
- {structflo_ner-0.6.0 → structflo_ner-0.7.0}/uv.lock +1 -1
- {structflo_ner-0.6.0 → structflo_ner-0.7.0}/.github/workflows/ci.yml +0 -0
- {structflo_ner-0.6.0 → structflo_ner-0.7.0}/.github/workflows/publish.yml +0 -0
- {structflo_ner-0.6.0 → structflo_ner-0.7.0}/.gitignore +0 -0
- {structflo_ner-0.6.0 → structflo_ner-0.7.0}/Makefile +0 -0
- {structflo_ner-0.6.0 → structflo_ner-0.7.0}/coverage.xml +0 -0
- {structflo_ner-0.6.0 → structflo_ner-0.7.0}/images/fast-viz.png +0 -0
- {structflo_ner-0.6.0 → structflo_ner-0.7.0}/images/local-gen-pandas.png +0 -0
- {structflo_ner-0.6.0 → structflo_ner-0.7.0}/images/local-gen-viz.png +0 -0
- {structflo_ner-0.6.0 → structflo_ner-0.7.0}/images/local-tb-viz.png +0 -0
- {structflo_ner-0.6.0 → structflo_ner-0.7.0}/images/ner_visualization.gif +0 -0
- {structflo_ner-0.6.0 → structflo_ner-0.7.0}/images/struct-flo-ner.png +0 -0
- {structflo_ner-0.6.0 → structflo_ner-0.7.0}/notebooks/01_quickstart.ipynb +0 -0
- {structflo_ner-0.6.0 → structflo_ner-0.7.0}/notebooks/02_fast_ner.ipynb +0 -0
- {structflo_ner-0.6.0 → structflo_ner-0.7.0}/structflo/ner/_display.py +0 -0
- {structflo_ner-0.6.0 → structflo_ner-0.7.0}/structflo/ner/_entities.py +0 -0
- {structflo_ner-0.6.0 → structflo_ner-0.7.0}/structflo/ner/fast/README.md +0 -0
- {structflo_ner-0.6.0 → structflo_ner-0.7.0}/structflo/ner/fast/__init__.py +0 -0
- {structflo_ner-0.6.0 → structflo_ner-0.7.0}/structflo/ner/fast/_loader.py +0 -0
- {structflo_ner-0.6.0 → structflo_ner-0.7.0}/structflo/ner/fast/_matcher.py +0 -0
- {structflo_ner-0.6.0 → structflo_ner-0.7.0}/structflo/ner/fast/_normalize.py +0 -0
- {structflo_ner-0.6.0 → structflo_ner-0.7.0}/structflo/ner/fast/extractor.py +0 -0
- {structflo_ner-0.6.0 → structflo_ner-0.7.0}/structflo/ner/fast/gazetteers/accession_number.yml +0 -0
- {structflo_ner-0.6.0 → structflo_ner-0.7.0}/structflo/ner/fast/gazetteers/compound_name.yml +0 -0
- {structflo_ner-0.6.0 → structflo_ner-0.7.0}/structflo/ner/fast/gazetteers/disease.yml +0 -0
- {structflo_ner-0.6.0 → structflo_ner-0.7.0}/structflo/ner/fast/gazetteers/functional_category.yml +0 -0
- {structflo_ner-0.6.0 → structflo_ner-0.7.0}/structflo/ner/fast/gazetteers/gene_name.yml +0 -0
- {structflo_ner-0.6.0 → structflo_ner-0.7.0}/structflo/ner/fast/gazetteers/product.yml +0 -0
- {structflo_ner-0.6.0 → structflo_ner-0.7.0}/structflo/ner/fast/gazetteers/screening_method.yml +0 -0
- {structflo_ner-0.6.0 → structflo_ner-0.7.0}/structflo/ner/fast/gazetteers/strain.yml +0 -0
- {structflo_ner-0.6.0 → structflo_ner-0.7.0}/structflo/ner/fast/gazetteers/target.yml +0 -0
- {structflo_ner-0.6.0 → structflo_ner-0.7.0}/structflo/ner/profiles.py +0 -0
- {structflo_ner-0.6.0 → structflo_ner-0.7.0}/tests/__init__.py +0 -0
- {structflo_ner-0.6.0 → structflo_ner-0.7.0}/tests/test_entities.py +0 -0
- {structflo_ner-0.6.0 → structflo_ner-0.7.0}/tests/test_fast.py +0 -0
|
@@ -1,3 +1,16 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: structflo-ner
|
|
3
|
+
Version: 0.7.0
|
|
4
|
+
Summary: Drug discovery NER wrapper around LangExtract — zero-config entity extraction for chemistry and biology.
|
|
5
|
+
License: Apache-2.0
|
|
6
|
+
Requires-Python: >=3.10
|
|
7
|
+
Requires-Dist: langextract>=1.7.0
|
|
8
|
+
Requires-Dist: pyyaml>=6.0
|
|
9
|
+
Requires-Dist: rapidfuzz>=3.0
|
|
10
|
+
Provides-Extra: dataframe
|
|
11
|
+
Requires-Dist: pandas>=1.5; extra == 'dataframe'
|
|
12
|
+
Description-Content-Type: text/markdown
|
|
13
|
+
|
|
1
14
|
|
|
2
15
|
<h1 align="center">structflo.ner</h1>
|
|
3
16
|
<p align="center">
|
|
@@ -343,10 +356,19 @@ Each bioactivity carries its measurement in `attributes`:
|
|
|
343
356
|
| `assay_type` | `CC50` | the endpoint (IC50, MIC90, GI50, ED90, ...) |
|
|
344
357
|
| `assay` | `HepG2 MTT` | assay, cell line or read-out the value was measured in |
|
|
345
358
|
| `compound_name` | `8t` | compound the value belongs to |
|
|
346
|
-
| `
|
|
359
|
+
| `target` | `InhA` | protein measured against (`TB` profile only) |
|
|
360
|
+
| `strain` | `H37Rv` | organism, strain or virus measured against (`TB` only) |
|
|
361
|
+
| `combination` | `meropenem` | second compound dosed alongside (`TB` profile only) |
|
|
362
|
+
|
|
363
|
+
An attribute the text does not state is absent from `attributes`.
|
|
364
|
+
|
|
365
|
+
A table on one page often names its target only on another. Pass text from
|
|
366
|
+
elsewhere in the document (its title or summary) as `context`; the model uses it
|
|
367
|
+
to fill attributes the page leaves unstated, and extracts nothing from it:
|
|
347
368
|
|
|
348
|
-
|
|
349
|
-
|
|
369
|
+
```python
|
|
370
|
+
result = extractor.extract(page_text, context=document_summary)
|
|
371
|
+
```
|
|
350
372
|
|
|
351
373
|
|
|
352
374
|
## Notebooks
|
|
@@ -1,16 +1,3 @@
|
|
|
1
|
-
Metadata-Version: 2.5
|
|
2
|
-
Name: structflo-ner
|
|
3
|
-
Version: 0.6.0
|
|
4
|
-
Summary: Drug discovery NER wrapper around LangExtract — zero-config entity extraction for chemistry and biology.
|
|
5
|
-
License: Apache-2.0
|
|
6
|
-
Requires-Python: >=3.10
|
|
7
|
-
Requires-Dist: langextract>=1.6.0
|
|
8
|
-
Requires-Dist: pyyaml>=6.0
|
|
9
|
-
Requires-Dist: rapidfuzz>=3.0
|
|
10
|
-
Provides-Extra: dataframe
|
|
11
|
-
Requires-Dist: pandas>=1.5; extra == 'dataframe'
|
|
12
|
-
Description-Content-Type: text/markdown
|
|
13
|
-
|
|
14
1
|
|
|
15
2
|
<h1 align="center">structflo.ner</h1>
|
|
16
3
|
<p align="center">
|
|
@@ -356,10 +343,19 @@ Each bioactivity carries its measurement in `attributes`:
|
|
|
356
343
|
| `assay_type` | `CC50` | the endpoint (IC50, MIC90, GI50, ED90, ...) |
|
|
357
344
|
| `assay` | `HepG2 MTT` | assay, cell line or read-out the value was measured in |
|
|
358
345
|
| `compound_name` | `8t` | compound the value belongs to |
|
|
359
|
-
| `
|
|
346
|
+
| `target` | `InhA` | protein measured against (`TB` profile only) |
|
|
347
|
+
| `strain` | `H37Rv` | organism, strain or virus measured against (`TB` only) |
|
|
348
|
+
| `combination` | `meropenem` | second compound dosed alongside (`TB` profile only) |
|
|
349
|
+
|
|
350
|
+
An attribute the text does not state is absent from `attributes`.
|
|
351
|
+
|
|
352
|
+
A table on one page often names its target only on another. Pass text from
|
|
353
|
+
elsewhere in the document (its title or summary) as `context`; the model uses it
|
|
354
|
+
to fill attributes the page leaves unstated, and extracts nothing from it:
|
|
360
355
|
|
|
361
|
-
|
|
362
|
-
|
|
356
|
+
```python
|
|
357
|
+
result = extractor.extract(page_text, context=document_summary)
|
|
358
|
+
```
|
|
363
359
|
|
|
364
360
|
|
|
365
361
|
## Notebooks
|
|
@@ -625,6 +625,7 @@ _TB_EXAMPLE_3 = lx.data.ExampleData(
|
|
|
625
625
|
"assay_type": "IC50",
|
|
626
626
|
"compound_name": "Compound 14a",
|
|
627
627
|
"assay": "biochemical assay",
|
|
628
|
+
"target": "InhA",
|
|
628
629
|
},
|
|
629
630
|
),
|
|
630
631
|
lx.data.Extraction(
|
|
@@ -853,6 +854,7 @@ _TB_EXAMPLE_5 = lx.data.ExampleData(
|
|
|
853
854
|
"unit": "nM",
|
|
854
855
|
"assay_type": "IC50",
|
|
855
856
|
"compound_name": "SACC-3060",
|
|
857
|
+
"target": "DprE1",
|
|
856
858
|
},
|
|
857
859
|
),
|
|
858
860
|
lx.data.Extraction(
|
|
@@ -980,6 +982,99 @@ _TB_EXAMPLE_6 = lx.data.ExampleData(
|
|
|
980
982
|
],
|
|
981
983
|
)
|
|
982
984
|
|
|
985
|
+
# Strain panel: organism column headers are the strain, a protein column the
|
|
986
|
+
# target, a cytotoxicity column the assay; a column dosed with a second drug
|
|
987
|
+
# names the combination partner. The disease stays a disease entity.
|
|
988
|
+
_TB_EXAMPLE_7 = lx.data.ExampleData(
|
|
989
|
+
text=(
|
|
990
|
+
"Table 3. Antibacterial activity against hospital-acquired pneumonia isolates\n\n"
|
|
991
|
+
"| Cpd | ID | E. coli ATCC 25922 | K. pneumoniae BAA-1705 | A. baumannii ATCC 19606 "
|
|
992
|
+
"| ATCC 19606 + polymyxin B | HepG2 | hERG |\n"
|
|
993
|
+
"| --- | --- | --- | --- | --- | --- | --- | --- |\n"
|
|
994
|
+
"| 9c | CHEMBL5311027 | 0.25 | 1 | >64 | 2 | >100 | 12.5 |\n\n"
|
|
995
|
+
"MIC (µg/mL); polymyxin B at 0.5 µg/mL. HepG2: CC50 (µM). hERG: IC50 (µM)."
|
|
996
|
+
),
|
|
997
|
+
extractions=[
|
|
998
|
+
lx.data.Extraction(
|
|
999
|
+
extraction_class="disease",
|
|
1000
|
+
extraction_text="hospital-acquired pneumonia",
|
|
1001
|
+
attributes={"therapeutic_area": "infectious disease"},
|
|
1002
|
+
),
|
|
1003
|
+
lx.data.Extraction(
|
|
1004
|
+
extraction_class="compound_name",
|
|
1005
|
+
extraction_text="CHEMBL5311027",
|
|
1006
|
+
attributes={"synonyms": "9c"},
|
|
1007
|
+
),
|
|
1008
|
+
lx.data.Extraction(
|
|
1009
|
+
extraction_class="bioactivity",
|
|
1010
|
+
extraction_text="0.25",
|
|
1011
|
+
attributes={
|
|
1012
|
+
"value": "0.25",
|
|
1013
|
+
"unit": "µg/mL",
|
|
1014
|
+
"assay_type": "MIC",
|
|
1015
|
+
"strain": "E. coli ATCC 25922",
|
|
1016
|
+
"compound_name": "CHEMBL5311027",
|
|
1017
|
+
},
|
|
1018
|
+
),
|
|
1019
|
+
lx.data.Extraction(
|
|
1020
|
+
extraction_class="bioactivity",
|
|
1021
|
+
extraction_text="1",
|
|
1022
|
+
attributes={
|
|
1023
|
+
"value": "1",
|
|
1024
|
+
"unit": "µg/mL",
|
|
1025
|
+
"assay_type": "MIC",
|
|
1026
|
+
"strain": "K. pneumoniae BAA-1705",
|
|
1027
|
+
"compound_name": "CHEMBL5311027",
|
|
1028
|
+
},
|
|
1029
|
+
),
|
|
1030
|
+
lx.data.Extraction(
|
|
1031
|
+
extraction_class="bioactivity",
|
|
1032
|
+
extraction_text=">64",
|
|
1033
|
+
attributes={
|
|
1034
|
+
"value": ">64",
|
|
1035
|
+
"unit": "µg/mL",
|
|
1036
|
+
"assay_type": "MIC",
|
|
1037
|
+
"strain": "A. baumannii ATCC 19606",
|
|
1038
|
+
"compound_name": "CHEMBL5311027",
|
|
1039
|
+
},
|
|
1040
|
+
),
|
|
1041
|
+
lx.data.Extraction(
|
|
1042
|
+
extraction_class="bioactivity",
|
|
1043
|
+
extraction_text="2",
|
|
1044
|
+
attributes={
|
|
1045
|
+
"value": "2",
|
|
1046
|
+
"unit": "µg/mL",
|
|
1047
|
+
"assay_type": "MIC",
|
|
1048
|
+
"strain": "ATCC 19606",
|
|
1049
|
+
"combination": "polymyxin B",
|
|
1050
|
+
"compound_name": "CHEMBL5311027",
|
|
1051
|
+
},
|
|
1052
|
+
),
|
|
1053
|
+
lx.data.Extraction(
|
|
1054
|
+
extraction_class="bioactivity",
|
|
1055
|
+
extraction_text=">100",
|
|
1056
|
+
attributes={
|
|
1057
|
+
"value": ">100",
|
|
1058
|
+
"unit": "µM",
|
|
1059
|
+
"assay_type": "CC50",
|
|
1060
|
+
"compound_name": "CHEMBL5311027",
|
|
1061
|
+
"assay": "HepG2",
|
|
1062
|
+
},
|
|
1063
|
+
),
|
|
1064
|
+
lx.data.Extraction(
|
|
1065
|
+
extraction_class="bioactivity",
|
|
1066
|
+
extraction_text="12.5",
|
|
1067
|
+
attributes={
|
|
1068
|
+
"value": "12.5",
|
|
1069
|
+
"unit": "µM",
|
|
1070
|
+
"assay_type": "IC50",
|
|
1071
|
+
"compound_name": "CHEMBL5311027",
|
|
1072
|
+
"target": "hERG",
|
|
1073
|
+
},
|
|
1074
|
+
),
|
|
1075
|
+
],
|
|
1076
|
+
)
|
|
1077
|
+
|
|
983
1078
|
TB_EXAMPLES: list[lx.data.ExampleData] = [
|
|
984
1079
|
_TB_EXAMPLE_1,
|
|
985
1080
|
_TB_EXAMPLE_2,
|
|
@@ -987,6 +1082,7 @@ TB_EXAMPLES: list[lx.data.ExampleData] = [
|
|
|
987
1082
|
_TB_EXAMPLE_4,
|
|
988
1083
|
_TB_EXAMPLE_5,
|
|
989
1084
|
_TB_EXAMPLE_6,
|
|
1085
|
+
_TB_EXAMPLE_7,
|
|
990
1086
|
]
|
|
991
1087
|
|
|
992
1088
|
TB_CHEMISTRY_EXAMPLES: list[lx.data.ExampleData] = [
|
|
@@ -29,6 +29,10 @@ def extraction_to_entity(extraction: lx.data.Extraction) -> NEREntity:
|
|
|
29
29
|
attributes: dict[str, str] = {}
|
|
30
30
|
if extraction.attributes:
|
|
31
31
|
for key, value in extraction.attributes.items():
|
|
32
|
+
# Strict-schema providers (OpenAI) must emit every attribute key, so an
|
|
33
|
+
# unstated one arrives as null; leave it out rather than store "None".
|
|
34
|
+
if value is None:
|
|
35
|
+
continue
|
|
32
36
|
if isinstance(value, list):
|
|
33
37
|
attributes[key] = ", ".join(value)
|
|
34
38
|
else:
|
|
@@ -13,20 +13,41 @@ _BIOACTIVITY_ATTRIBUTES = (
|
|
|
13
13
|
"MIC, MIC50, MIC90, MIC99, MBC, log10 CFU reduction; CC50, TC50, SI (selectivity index); "
|
|
14
14
|
"GI50, TGI, LC50; DC50, Dmax (degraders); ED50, ED90, %TGI, T/C (in vivo); "
|
|
15
15
|
"PRR (parasite reduction ratio), parasite clearance half-life.\n"
|
|
16
|
+
"- compound_name: the compound the value belongs to, by its name or identifier "
|
|
17
|
+
"(e.g. 'bedaquiline', 'BTZ043', 'CHEMBL1234', 'SACC-3000'); a document-local label "
|
|
18
|
+
"('7a', 'Compound 9b') only when nothing else names the compound.\n"
|
|
19
|
+
)
|
|
20
|
+
_ASSAY_ATTRIBUTE = (
|
|
16
21
|
"- assay: the assay, cell line or read-out the value was measured in, e.g. 'MABA', "
|
|
17
22
|
"'LORA', 'HepG2 MTT', 'THP-1 macrophage infection assay', 'asexual blood-stage', "
|
|
18
23
|
"'liver-stage', 'gametocyte', 'NCI-60', 'hERG', 'P. berghei 4-day suppressive test'. "
|
|
19
24
|
"null when no assay is stated.\n"
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
25
|
+
)
|
|
26
|
+
# TB gives each kind of context its own slot: what the value was measured in
|
|
27
|
+
# (assay), against (target, strain), and with (combination).
|
|
28
|
+
_TB_SLOT_ATTRIBUTES = (
|
|
29
|
+
"- assay: the assay system the value was measured in: a cell line, format or read-out "
|
|
30
|
+
"('MABA', 'LORA', 'HepG2 MTT', 'THP-1 macrophage infection assay', 'asexual blood-stage', "
|
|
31
|
+
"'AlphaScreen', 'FP', 'NCI-60', 'P. berghei 4-day suppressive test'). null when none is "
|
|
32
|
+
"stated.\n"
|
|
33
|
+
"- target: the protein the value was measured against: an enzyme, receptor, channel or "
|
|
34
|
+
"other molecular target, as written ('InhA', 'DprE1', 'hERG', 'KasA', 'PDE4B', "
|
|
35
|
+
"'BACE1'). null for whole-cell, organism and cytotoxicity values unless one is named.\n"
|
|
36
|
+
"- strain: the organism the value was measured against, as written: species, strain, "
|
|
37
|
+
"isolate or virus (H37Rv, M. tuberculosis, E. coli ATCC 25922, A. baumannii ATCC 19606, "
|
|
38
|
+
"Pf3D7, Dd2, S. mansoni, HIV-1, norovirus). An organism always goes here, never in assay or target, even "
|
|
39
|
+
"when it heads a table column or is all that names the assay. null when not stated.\n"
|
|
40
|
+
"- combination: a second compound dosed together with the one measured (a combination "
|
|
41
|
+
"or potentiation partner), as written ('avibactam', 'polymyxin B 0.5 µg/mL'). Not a "
|
|
42
|
+
"reference compound the value is relative to ('% of isoproterenol') or an agonist, substrate "
|
|
43
|
+
"or tracer the assay uses. null for a single agent.\n"
|
|
23
44
|
)
|
|
24
45
|
_BIOACTIVITY_TABLES = (
|
|
25
46
|
"In a data table, extract every cell value as its own bioactivity. An ID or registry "
|
|
26
47
|
"column, when present, supplies the compound_name and the row label is only its synonym; "
|
|
27
|
-
"otherwise the row label is the compound_name. A column header
|
|
28
|
-
"
|
|
29
|
-
"('CC50 in µM')."
|
|
48
|
+
"otherwise the row label is the compound_name. A column header fills the attribute for "
|
|
49
|
+
"what it names (an assay or cell line, a protein target, an organism or strain), and the "
|
|
50
|
+
"endpoint and unit often come from the caption or a footnote ('CC50 in µM')."
|
|
30
51
|
)
|
|
31
52
|
|
|
32
53
|
CHEMISTRY_PROMPT = (
|
|
@@ -49,6 +70,7 @@ BIOACTIVITY_PROMPT = (
|
|
|
49
70
|
"Extract bioactivity measurements and assay data from the text: potency, selectivity, "
|
|
50
71
|
"cytotoxicity and efficacy values.\n"
|
|
51
72
|
+ _BIOACTIVITY_ATTRIBUTES
|
|
73
|
+
+ _ASSAY_ATTRIBUTE
|
|
52
74
|
+ _BIOACTIVITY_TABLES
|
|
53
75
|
+ "\nAlso extract assay descriptions: cell lines used (e.g. 'HeLa', 'A549'), assay formats "
|
|
54
76
|
"(e.g. 'cell viability', 'binding assay', 'enzymatic assay'), and organisms."
|
|
@@ -75,6 +97,7 @@ FULL_PROMPT = (
|
|
|
75
97
|
"- Diseases and indications: cancer types, disease names, therapeutic areas.\n"
|
|
76
98
|
"- Mechanisms of action: binding modes, inhibition types, selectivity descriptions.\n"
|
|
77
99
|
+ _BIOACTIVITY_ATTRIBUTES
|
|
100
|
+
+ _ASSAY_ATTRIBUTE
|
|
78
101
|
+ _BIOACTIVITY_TABLES
|
|
79
102
|
+ "\nExtract only what is explicitly stated; do not infer or generate values."
|
|
80
103
|
)
|
|
@@ -84,8 +107,9 @@ FULL_PROMPT = (
|
|
|
84
107
|
TB_PROMPT = (
|
|
85
108
|
"Extract drug discovery entities from this tuberculosis research text.\n\n"
|
|
86
109
|
+ _BIOACTIVITY_ATTRIBUTES
|
|
87
|
-
+
|
|
88
|
-
|
|
110
|
+
+ _TB_SLOT_ATTRIBUTES
|
|
111
|
+
+ _BIOACTIVITY_TABLES
|
|
112
|
+
+ "\n\n"
|
|
89
113
|
"DISAMBIGUATION RULES:\n"
|
|
90
114
|
"- Mycobacterial proteins (e.g. ClpC1, DprE1, InhA, AtpE, MmpL3, QcrB) "
|
|
91
115
|
"are biological targets, NOT compounds.\n"
|
|
@@ -102,6 +126,9 @@ TB_PROMPT = (
|
|
|
102
126
|
"- Enzyme descriptions like 'enoyl-ACP reductase' are product, not target.\n"
|
|
103
127
|
"- 'cell wall', 'lipid metabolism' are functional_category, not mechanism_of_action.\n"
|
|
104
128
|
"- 'fragment screening', 'biochemical assay' are screening_method, not assay.\n"
|
|
129
|
+
"- A disease or infection is a disease entity wherever it is named, in prose, a heading "
|
|
130
|
+
"or a table cell (tuberculosis, HIV infection, malaria, schistosomiasis, glioblastoma in 'U87 (glioblastoma)'), "
|
|
131
|
+
"even when its organism also fills a bioactivity's strain.\n"
|
|
105
132
|
"- Use target for proteins in a drug-targeting context, gene_name for loci, "
|
|
106
133
|
"protein_name for non-drug-target proteins.\n\n"
|
|
107
134
|
"Extract only what is explicitly stated; do not infer or generate values."
|
|
@@ -125,3 +152,13 @@ TB_BIOLOGY_PROMPT = (
|
|
|
125
152
|
"are functional_category. "
|
|
126
153
|
"Extract only what is explicitly stated; do not infer or generate values."
|
|
127
154
|
)
|
|
155
|
+
|
|
156
|
+
# Framing for NERExtractor.extract(context=...), rendered by langextract ahead of
|
|
157
|
+
# the examples.
|
|
158
|
+
DOCUMENT_CONTEXT = (
|
|
159
|
+
"Document context, from elsewhere in the same document. It is not the text to extract "
|
|
160
|
+
"from: take no entity and no value from it, and do not pair a value in the text with a "
|
|
161
|
+
"number or claim made here. Use it only to fill a bioactivity attribute the text leaves "
|
|
162
|
+
"unstated, such as the target or organism of a table, when the context names one for "
|
|
163
|
+
"the whole document.\n"
|
|
164
|
+
)
|
|
@@ -3,15 +3,43 @@
|
|
|
3
3
|
from __future__ import annotations
|
|
4
4
|
|
|
5
5
|
import logging
|
|
6
|
+
import re
|
|
6
7
|
|
|
7
8
|
import langextract as lx
|
|
8
9
|
|
|
10
|
+
from structflo.ner import _prompts
|
|
9
11
|
from structflo.ner._entities import NERResult
|
|
10
12
|
from structflo.ner._mapping import annotated_doc_to_result
|
|
11
13
|
from structflo.ner.profiles import FULL, EntityProfile
|
|
12
14
|
|
|
13
15
|
logger = logging.getLogger(__name__)
|
|
14
16
|
|
|
17
|
+
# A table cell holding only a value: '0.94', '>32', '2.4*', '0.62µM'.
|
|
18
|
+
_VALUE_CELL = re.compile(r"^[<>≤≥~=]*\s*\d[\d,]*\.?\d*\s*\*?\s*(%|[nµu]M|µg/mL|mg/kg)?$")
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def _bioactivity_count(doc: lx.data.AnnotatedDocument) -> int:
|
|
22
|
+
return sum(1 for ext in doc.extractions if ext.extraction_class == "bioactivity")
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def _table_values_dropped(text: str, doc: lx.data.AnnotatedDocument) -> bool:
|
|
26
|
+
"""True when fewer than half of a page's table values came back as bioactivities.
|
|
27
|
+
|
|
28
|
+
Models now and then return a complete, valid answer that lists a table's
|
|
29
|
+
compounds but leaves out its values. On the ChEMBL-gold decks this fires on
|
|
30
|
+
~2% of pages and catches ~60% of such dropouts.
|
|
31
|
+
"""
|
|
32
|
+
# ponytail: markdown tables only; values in prose and qualitative results
|
|
33
|
+
# are not checked, so those dropouts still pass silently.
|
|
34
|
+
cells = sum(
|
|
35
|
+
1
|
|
36
|
+
for line in text.splitlines()
|
|
37
|
+
if line.lstrip().startswith("|")
|
|
38
|
+
for cell in line.strip().strip("|").split("|")
|
|
39
|
+
if _VALUE_CELL.match(cell.strip())
|
|
40
|
+
)
|
|
41
|
+
return cells >= 5 and _bioactivity_count(doc) < 0.5 * cells
|
|
42
|
+
|
|
15
43
|
|
|
16
44
|
class NERExtractor:
|
|
17
45
|
"""Extract drug discovery entities from text with zero configuration.
|
|
@@ -90,12 +118,17 @@ class NERExtractor:
|
|
|
90
118
|
self,
|
|
91
119
|
text: str | list[str],
|
|
92
120
|
profile: EntityProfile | None = None,
|
|
121
|
+
context: str | None = None,
|
|
93
122
|
) -> NERResult | list[NERResult]:
|
|
94
123
|
"""Extract drug discovery entities from text.
|
|
95
124
|
|
|
96
125
|
Args:
|
|
97
126
|
text: Input text (or list of texts) to process.
|
|
98
127
|
profile: Override the default profile for this call only.
|
|
128
|
+
context: Text from elsewhere in the same document (its title or
|
|
129
|
+
summary). The model reads it to fill attributes the text leaves
|
|
130
|
+
unstated, such as the target of a table whose target is named
|
|
131
|
+
only on another page; no entities are extracted from it.
|
|
99
132
|
|
|
100
133
|
Returns:
|
|
101
134
|
A :class:`NERResult` for a single string input, or a list of
|
|
@@ -107,7 +140,13 @@ class NERExtractor:
|
|
|
107
140
|
|
|
108
141
|
results = []
|
|
109
142
|
for single_text in texts:
|
|
110
|
-
doc = self._run_extraction(single_text, active_profile)
|
|
143
|
+
doc = self._run_extraction(single_text, active_profile, context)
|
|
144
|
+
if "bioactivity" in active_profile.entity_classes and _table_values_dropped(
|
|
145
|
+
single_text, doc
|
|
146
|
+
):
|
|
147
|
+
logger.warning("Table values missing from extraction; retrying once.")
|
|
148
|
+
retry = self._run_extraction(single_text, active_profile, context)
|
|
149
|
+
doc = max(doc, retry, key=_bioactivity_count)
|
|
111
150
|
results.append(annotated_doc_to_result(doc, single_text))
|
|
112
151
|
|
|
113
152
|
return results if is_batch else results[0]
|
|
@@ -198,6 +237,7 @@ class NERExtractor:
|
|
|
198
237
|
self,
|
|
199
238
|
text: str,
|
|
200
239
|
profile: EntityProfile,
|
|
240
|
+
context: str | None = None,
|
|
201
241
|
) -> lx.data.AnnotatedDocument:
|
|
202
242
|
"""Call lx.extract and return the AnnotatedDocument."""
|
|
203
243
|
examples = self._build_examples(profile)
|
|
@@ -206,6 +246,8 @@ class NERExtractor:
|
|
|
206
246
|
kwargs: dict = dict(self._langextract_kwargs)
|
|
207
247
|
kwargs.setdefault("use_schema_constraints", True)
|
|
208
248
|
kwargs.setdefault("show_progress", False)
|
|
249
|
+
if context:
|
|
250
|
+
kwargs["additional_context"] = _prompts.DOCUMENT_CONTEXT + context
|
|
209
251
|
|
|
210
252
|
if self._provider is not None:
|
|
211
253
|
# Explicit provider → deterministic routing via ModelConfig.
|
|
@@ -236,4 +278,29 @@ class NERExtractor:
|
|
|
236
278
|
|
|
237
279
|
# Post-process: drop extractions with hallucinated entity classes
|
|
238
280
|
allowed = set(profile.entity_classes)
|
|
239
|
-
|
|
281
|
+
doc = self._filter_extractions(doc, allowed)
|
|
282
|
+
if context:
|
|
283
|
+
doc = self._drop_context_copies(doc, context)
|
|
284
|
+
return doc
|
|
285
|
+
|
|
286
|
+
@staticmethod
|
|
287
|
+
def _drop_context_copies(
|
|
288
|
+
doc: lx.data.AnnotatedDocument,
|
|
289
|
+
context: str,
|
|
290
|
+
) -> lx.data.AnnotatedDocument:
|
|
291
|
+
"""Drop extractions the model took from the context instead of the text.
|
|
292
|
+
|
|
293
|
+
The prompt forbids it, but models still extract from the context now and
|
|
294
|
+
then. Such an extraction does not align to the text, and its text occurs
|
|
295
|
+
in the context but not in the text (an unaligned one that is in the text
|
|
296
|
+
is a missed alignment, not a copy).
|
|
297
|
+
"""
|
|
298
|
+
ctx, text = context.casefold(), doc.text.casefold()
|
|
299
|
+
kept = [
|
|
300
|
+
ext
|
|
301
|
+
for ext in doc.extractions
|
|
302
|
+
if ext.char_interval is not None
|
|
303
|
+
or ext.extraction_text.casefold() not in ctx
|
|
304
|
+
or ext.extraction_text.casefold() in text
|
|
305
|
+
]
|
|
306
|
+
return lx.data.AnnotatedDocument(text=doc.text, extractions=kept)
|
|
@@ -122,6 +122,50 @@ class TestNERExtractorExtract:
|
|
|
122
122
|
result = extractor.extract("My source text")
|
|
123
123
|
assert result.source_text == "My source text"
|
|
124
124
|
|
|
125
|
+
_TABLE = "| cpd | MIC | CC50 |\n| --- | --- | --- |\n" + "".join(
|
|
126
|
+
f"| C{i} | {i}.5 | >50 |\n" for i in range(5)
|
|
127
|
+
)
|
|
128
|
+
|
|
129
|
+
def _bio_doc(self, n: int) -> lx.data.AnnotatedDocument:
|
|
130
|
+
return _make_annotated_doc(
|
|
131
|
+
[
|
|
132
|
+
lx.data.Extraction(extraction_class="bioactivity", extraction_text=f"{i}.5")
|
|
133
|
+
for i in range(n)
|
|
134
|
+
]
|
|
135
|
+
)
|
|
136
|
+
|
|
137
|
+
def test_retries_once_when_table_values_are_missing(self):
|
|
138
|
+
extractor = NERExtractor(profile=TB)
|
|
139
|
+
extractor._run_extraction = MagicMock(side_effect=[self._bio_doc(0), self._bio_doc(10)])
|
|
140
|
+
result = extractor.extract(self._TABLE)
|
|
141
|
+
assert extractor._run_extraction.call_count == 2
|
|
142
|
+
assert len(result.bioactivities) == 10
|
|
143
|
+
|
|
144
|
+
def test_no_retry_when_table_values_came_back(self):
|
|
145
|
+
extractor = NERExtractor(profile=TB)
|
|
146
|
+
extractor._run_extraction = MagicMock(return_value=self._bio_doc(10))
|
|
147
|
+
extractor.extract(self._TABLE)
|
|
148
|
+
assert extractor._run_extraction.call_count == 1
|
|
149
|
+
|
|
150
|
+
def test_no_retry_for_profiles_without_bioactivity(self):
|
|
151
|
+
extractor = NERExtractor(profile=CHEMISTRY)
|
|
152
|
+
extractor._run_extraction = MagicMock(return_value=self._bio_doc(0))
|
|
153
|
+
extractor.extract(self._TABLE)
|
|
154
|
+
assert extractor._run_extraction.call_count == 1
|
|
155
|
+
|
|
156
|
+
def test_null_attributes_are_absent_not_none_strings(self):
|
|
157
|
+
# OpenAI strict mode returns every schema key, null when unstated.
|
|
158
|
+
extractor = self._extractor_with_mock(
|
|
159
|
+
[
|
|
160
|
+
lx.data.Extraction(
|
|
161
|
+
extraction_class="bioactivity",
|
|
162
|
+
extraction_text="IC50 = 3 nM",
|
|
163
|
+
attributes={"value": "3", "strain": None, "assay": None},
|
|
164
|
+
)
|
|
165
|
+
]
|
|
166
|
+
)
|
|
167
|
+
assert extractor.extract("text").bioactivities[0].attributes == {"value": "3"}
|
|
168
|
+
|
|
125
169
|
|
|
126
170
|
class TestProviderRouting:
|
|
127
171
|
"""Explicit provider selection routes deterministically via ModelConfig,
|
|
@@ -181,6 +225,24 @@ class TestProviderRouting:
|
|
|
181
225
|
assert pk["base_url"] == "http://ollama:11434"
|
|
182
226
|
assert pk["num_ctx"] == 8192
|
|
183
227
|
|
|
228
|
+
def test_context_reaches_langextract_framed(self, monkeypatch):
|
|
229
|
+
from structflo.ner import extractor as extractor_mod
|
|
230
|
+
|
|
231
|
+
captured = {}
|
|
232
|
+
|
|
233
|
+
def fake_extract(**kwargs):
|
|
234
|
+
captured.update(kwargs)
|
|
235
|
+
return _make_annotated_doc([])
|
|
236
|
+
|
|
237
|
+
monkeypatch.setattr(extractor_mod.lx, "extract", fake_extract)
|
|
238
|
+
NERExtractor().extract("IC50s table", context="Deck on InhA inhibitors.")
|
|
239
|
+
ctx = captured["additional_context"]
|
|
240
|
+
assert ctx.startswith("Document context") and ctx.endswith("Deck on InhA inhibitors.")
|
|
241
|
+
|
|
242
|
+
def test_no_context_passes_no_additional_context(self, monkeypatch):
|
|
243
|
+
captured = self._run_and_capture(monkeypatch)
|
|
244
|
+
assert "additional_context" not in captured
|
|
245
|
+
|
|
184
246
|
def test_cloud_provider_kwargs_carry_api_key(self, monkeypatch):
|
|
185
247
|
captured = self._run_and_capture(
|
|
186
248
|
monkeypatch, provider="openai", model_id="gpt-4o", api_key="sk-test"
|
|
@@ -235,6 +297,27 @@ class TestFilterExtractions:
|
|
|
235
297
|
assert len(filtered.extractions) == 0
|
|
236
298
|
|
|
237
299
|
|
|
300
|
+
class TestDropContextCopies:
|
|
301
|
+
def test_drops_only_unaligned_extractions_found_in_context(self):
|
|
302
|
+
aligned = lx.data.Extraction(
|
|
303
|
+
extraction_class="target",
|
|
304
|
+
extraction_text="InhA",
|
|
305
|
+
char_interval=lx.data.CharInterval(start_pos=0, end_pos=4),
|
|
306
|
+
)
|
|
307
|
+
copied = lx.data.Extraction(extraction_class="compound_name", extraction_text="Isoniazid")
|
|
308
|
+
unaligned = lx.data.Extraction(extraction_class="compound_name", extraction_text="7a")
|
|
309
|
+
# In the context too, but also on the page: a missed alignment, not a copy.
|
|
310
|
+
on_page = lx.data.Extraction(extraction_class="disease", extraction_text="malaria")
|
|
311
|
+
doc = lx.data.AnnotatedDocument(
|
|
312
|
+
text="InhA IC50s for 7a; malaria panel",
|
|
313
|
+
extractions=[aligned, copied, unaligned, on_page],
|
|
314
|
+
)
|
|
315
|
+
kept = NERExtractor._drop_context_copies(
|
|
316
|
+
doc, "InhA inhibitors that match isoniazid, and a malaria screen"
|
|
317
|
+
)
|
|
318
|
+
assert kept.extractions == [aligned, unaligned, on_page]
|
|
319
|
+
|
|
320
|
+
|
|
238
321
|
class TestBuildExamples:
|
|
239
322
|
def test_extra_examples_appended(self):
|
|
240
323
|
extra = lx.data.ExampleData(text="extra", extractions=[])
|
|
@@ -298,8 +381,20 @@ class TestTBProfile:
|
|
|
298
381
|
}
|
|
299
382
|
assert {"value", "unit", "assay_type", "compound_name", "assay"} <= keys, profile.name
|
|
300
383
|
|
|
384
|
+
def test_tb_bioactivity_schema_has_context_slots(self):
|
|
385
|
+
"""TB gives what a value was measured against (target, strain) and with
|
|
386
|
+
(combination) their own slots, apart from assay."""
|
|
387
|
+
keys = {
|
|
388
|
+
k
|
|
389
|
+
for ex in TB.examples
|
|
390
|
+
for e in ex.extractions
|
|
391
|
+
if e.extraction_class == "bioactivity"
|
|
392
|
+
for k in (e.attributes or {})
|
|
393
|
+
}
|
|
394
|
+
assert {"target", "strain", "combination"} <= keys
|
|
395
|
+
|
|
301
396
|
def test_tb_examples_count(self):
|
|
302
|
-
assert len(TB.examples) ==
|
|
397
|
+
assert len(TB.examples) == 7
|
|
303
398
|
assert len(TB_CHEMISTRY.examples) == 2
|
|
304
399
|
assert len(TB_BIOLOGY.examples) == 2
|
|
305
400
|
|
|
@@ -2170,7 +2170,7 @@ dev = [
|
|
|
2170
2170
|
|
|
2171
2171
|
[package.metadata]
|
|
2172
2172
|
requires-dist = [
|
|
2173
|
-
{ name = "langextract", specifier = ">=1.
|
|
2173
|
+
{ name = "langextract", specifier = ">=1.7.0" },
|
|
2174
2174
|
{ name = "pandas", marker = "extra == 'dataframe'", specifier = ">=1.5" },
|
|
2175
2175
|
{ name = "pyyaml", specifier = ">=6.0" },
|
|
2176
2176
|
{ name = "rapidfuzz", specifier = ">=3.0" },
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{structflo_ner-0.6.0 → structflo_ner-0.7.0}/structflo/ner/fast/gazetteers/accession_number.yml
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{structflo_ner-0.6.0 → structflo_ner-0.7.0}/structflo/ner/fast/gazetteers/functional_category.yml
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{structflo_ner-0.6.0 → structflo_ner-0.7.0}/structflo/ner/fast/gazetteers/screening_method.yml
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|