structflo-ner 0.5.0__tar.gz → 0.7.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- structflo_ner-0.5.0/README.md → structflo_ner-0.7.0/PKG-INFO +36 -0
- structflo_ner-0.5.0/PKG-INFO → structflo_ner-0.7.0/README.md +23 -13
- {structflo_ner-0.5.0 → structflo_ner-0.7.0}/pyproject.toml +1 -1
- {structflo_ner-0.5.0 → structflo_ner-0.7.0}/structflo/ner/__init__.py +1 -1
- {structflo_ner-0.5.0 → structflo_ner-0.7.0}/structflo/ner/_examples.py +222 -5
- {structflo_ner-0.5.0 → structflo_ner-0.7.0}/structflo/ner/_mapping.py +4 -0
- structflo_ner-0.7.0/structflo/ner/_prompts.py +164 -0
- {structflo_ner-0.5.0 → structflo_ner-0.7.0}/structflo/ner/extractor.py +69 -2
- {structflo_ner-0.5.0 → structflo_ner-0.7.0}/structflo/ner/fast/README.md +1 -1
- {structflo_ner-0.5.0 → structflo_ner-0.7.0}/structflo/ner/fast/_loader.py +7 -2
- {structflo_ner-0.5.0 → structflo_ner-0.7.0}/structflo/ner/fast/_matcher.py +3 -2
- {structflo_ner-0.5.0 → structflo_ner-0.7.0}/tests/test_extractor.py +110 -1
- {structflo_ner-0.5.0 → structflo_ner-0.7.0}/tests/test_fast.py +14 -0
- {structflo_ner-0.5.0 → structflo_ner-0.7.0}/uv.lock +4 -4
- structflo_ner-0.5.0/structflo/ner/_prompts.py +0 -95
- {structflo_ner-0.5.0 → structflo_ner-0.7.0}/.github/workflows/ci.yml +0 -0
- {structflo_ner-0.5.0 → structflo_ner-0.7.0}/.github/workflows/publish.yml +0 -0
- {structflo_ner-0.5.0 → structflo_ner-0.7.0}/.gitignore +0 -0
- {structflo_ner-0.5.0 → structflo_ner-0.7.0}/Makefile +0 -0
- {structflo_ner-0.5.0 → structflo_ner-0.7.0}/coverage.xml +0 -0
- {structflo_ner-0.5.0 → structflo_ner-0.7.0}/images/fast-viz.png +0 -0
- {structflo_ner-0.5.0 → structflo_ner-0.7.0}/images/local-gen-pandas.png +0 -0
- {structflo_ner-0.5.0 → structflo_ner-0.7.0}/images/local-gen-viz.png +0 -0
- {structflo_ner-0.5.0 → structflo_ner-0.7.0}/images/local-tb-viz.png +0 -0
- {structflo_ner-0.5.0 → structflo_ner-0.7.0}/images/ner_visualization.gif +0 -0
- {structflo_ner-0.5.0 → structflo_ner-0.7.0}/images/struct-flo-ner.png +0 -0
- {structflo_ner-0.5.0 → structflo_ner-0.7.0}/notebooks/01_quickstart.ipynb +0 -0
- {structflo_ner-0.5.0 → structflo_ner-0.7.0}/notebooks/02_fast_ner.ipynb +0 -0
- {structflo_ner-0.5.0 → structflo_ner-0.7.0}/structflo/ner/_display.py +0 -0
- {structflo_ner-0.5.0 → structflo_ner-0.7.0}/structflo/ner/_entities.py +0 -0
- {structflo_ner-0.5.0 → structflo_ner-0.7.0}/structflo/ner/fast/__init__.py +0 -0
- {structflo_ner-0.5.0 → structflo_ner-0.7.0}/structflo/ner/fast/_normalize.py +0 -0
- {structflo_ner-0.5.0 → structflo_ner-0.7.0}/structflo/ner/fast/extractor.py +0 -0
- {structflo_ner-0.5.0 → structflo_ner-0.7.0}/structflo/ner/fast/gazetteers/accession_number.yml +0 -0
- {structflo_ner-0.5.0 → structflo_ner-0.7.0}/structflo/ner/fast/gazetteers/compound_name.yml +0 -0
- {structflo_ner-0.5.0 → structflo_ner-0.7.0}/structflo/ner/fast/gazetteers/disease.yml +0 -0
- {structflo_ner-0.5.0 → structflo_ner-0.7.0}/structflo/ner/fast/gazetteers/functional_category.yml +0 -0
- {structflo_ner-0.5.0 → structflo_ner-0.7.0}/structflo/ner/fast/gazetteers/gene_name.yml +0 -0
- {structflo_ner-0.5.0 → structflo_ner-0.7.0}/structflo/ner/fast/gazetteers/product.yml +0 -0
- {structflo_ner-0.5.0 → structflo_ner-0.7.0}/structflo/ner/fast/gazetteers/screening_method.yml +0 -0
- {structflo_ner-0.5.0 → structflo_ner-0.7.0}/structflo/ner/fast/gazetteers/strain.yml +0 -0
- {structflo_ner-0.5.0 → structflo_ner-0.7.0}/structflo/ner/fast/gazetteers/target.yml +0 -0
- {structflo_ner-0.5.0 → structflo_ner-0.7.0}/structflo/ner/profiles.py +0 -0
- {structflo_ner-0.5.0 → structflo_ner-0.7.0}/tests/__init__.py +0 -0
- {structflo_ner-0.5.0 → structflo_ner-0.7.0}/tests/test_entities.py +0 -0
|
@@ -1,3 +1,16 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: structflo-ner
|
|
3
|
+
Version: 0.7.0
|
|
4
|
+
Summary: Drug discovery NER wrapper around LangExtract — zero-config entity extraction for chemistry and biology.
|
|
5
|
+
License: Apache-2.0
|
|
6
|
+
Requires-Python: >=3.10
|
|
7
|
+
Requires-Dist: langextract>=1.7.0
|
|
8
|
+
Requires-Dist: pyyaml>=6.0
|
|
9
|
+
Requires-Dist: rapidfuzz>=3.0
|
|
10
|
+
Provides-Extra: dataframe
|
|
11
|
+
Requires-Dist: pandas>=1.5; extra == 'dataframe'
|
|
12
|
+
Description-Content-Type: text/markdown
|
|
13
|
+
|
|
1
14
|
|
|
2
15
|
<h1 align="center">structflo.ner</h1>
|
|
3
16
|
<p align="center">
|
|
@@ -334,6 +347,29 @@ df = result.to_dataframe()
|
|
|
334
347
|
result.to_dict()
|
|
335
348
|
```
|
|
336
349
|
|
|
350
|
+
Each bioactivity carries its measurement in `attributes`:
|
|
351
|
+
|
|
352
|
+
| Attribute | Example | Meaning |
|
|
353
|
+
| --------------- | --------------------------- | ---------------------------------------------------------- |
|
|
354
|
+
| `value` | `>20.0` | number as reported, qualifier kept |
|
|
355
|
+
| `unit` | `µM` | unit as written |
|
|
356
|
+
| `assay_type` | `CC50` | the endpoint (IC50, MIC90, GI50, ED90, ...) |
|
|
357
|
+
| `assay` | `HepG2 MTT` | assay, cell line or read-out the value was measured in |
|
|
358
|
+
| `compound_name` | `8t` | compound the value belongs to |
|
|
359
|
+
| `target` | `InhA` | protein measured against (`TB` profile only) |
|
|
360
|
+
| `strain` | `H37Rv` | organism, strain or virus measured against (`TB` only) |
|
|
361
|
+
| `combination` | `meropenem` | second compound dosed alongside (`TB` profile only) |
|
|
362
|
+
|
|
363
|
+
An attribute the text does not state is absent from `attributes`.
|
|
364
|
+
|
|
365
|
+
A table on one page often names its target only on another. Pass text from
|
|
366
|
+
elsewhere in the document (its title or summary) as `context`; the model uses it
|
|
367
|
+
to fill attributes the page leaves unstated, and extracts nothing from it:
|
|
368
|
+
|
|
369
|
+
```python
|
|
370
|
+
result = extractor.extract(page_text, context=document_summary)
|
|
371
|
+
```
|
|
372
|
+
|
|
337
373
|
|
|
338
374
|
## Notebooks
|
|
339
375
|
|
|
@@ -1,16 +1,3 @@
|
|
|
1
|
-
Metadata-Version: 2.5
|
|
2
|
-
Name: structflo-ner
|
|
3
|
-
Version: 0.5.0
|
|
4
|
-
Summary: Drug discovery NER wrapper around LangExtract — zero-config entity extraction for chemistry and biology.
|
|
5
|
-
License: Apache-2.0
|
|
6
|
-
Requires-Python: >=3.10
|
|
7
|
-
Requires-Dist: langextract>=1.6.0
|
|
8
|
-
Requires-Dist: pyyaml>=6.0
|
|
9
|
-
Requires-Dist: rapidfuzz>=3.0
|
|
10
|
-
Provides-Extra: dataframe
|
|
11
|
-
Requires-Dist: pandas>=1.5; extra == 'dataframe'
|
|
12
|
-
Description-Content-Type: text/markdown
|
|
13
|
-
|
|
14
1
|
|
|
15
2
|
<h1 align="center">structflo.ner</h1>
|
|
16
3
|
<p align="center">
|
|
@@ -347,6 +334,29 @@ df = result.to_dataframe()
|
|
|
347
334
|
result.to_dict()
|
|
348
335
|
```
|
|
349
336
|
|
|
337
|
+
Each bioactivity carries its measurement in `attributes`:
|
|
338
|
+
|
|
339
|
+
| Attribute | Example | Meaning |
|
|
340
|
+
| --------------- | --------------------------- | ---------------------------------------------------------- |
|
|
341
|
+
| `value` | `>20.0` | number as reported, qualifier kept |
|
|
342
|
+
| `unit` | `µM` | unit as written |
|
|
343
|
+
| `assay_type` | `CC50` | the endpoint (IC50, MIC90, GI50, ED90, ...) |
|
|
344
|
+
| `assay` | `HepG2 MTT` | assay, cell line or read-out the value was measured in |
|
|
345
|
+
| `compound_name` | `8t` | compound the value belongs to |
|
|
346
|
+
| `target` | `InhA` | protein measured against (`TB` profile only) |
|
|
347
|
+
| `strain` | `H37Rv` | organism, strain or virus measured against (`TB` only) |
|
|
348
|
+
| `combination` | `meropenem` | second compound dosed alongside (`TB` profile only) |
|
|
349
|
+
|
|
350
|
+
An attribute the text does not state is absent from `attributes`.
|
|
351
|
+
|
|
352
|
+
A table on one page often names its target only on another. Pass text from
|
|
353
|
+
elsewhere in the document (its title or summary) as `context`; the model uses it
|
|
354
|
+
to fill attributes the page leaves unstated, and extracts nothing from it:
|
|
355
|
+
|
|
356
|
+
```python
|
|
357
|
+
result = extractor.extract(page_text, context=document_summary)
|
|
358
|
+
```
|
|
359
|
+
|
|
350
360
|
|
|
351
361
|
## Notebooks
|
|
352
362
|
|
|
@@ -158,8 +158,8 @@ BIOLOGY_EXAMPLES: list[lx.data.ExampleData] = [
|
|
|
158
158
|
|
|
159
159
|
_BIOACTIVITY_EXAMPLE_1 = lx.data.ExampleData(
|
|
160
160
|
text=(
|
|
161
|
-
"Compound 7 inhibited EGFR with an IC50 of 2.3 nM in a cell-free
|
|
162
|
-
"and showed an EC50 of 45 nM in A549 (human lung adenocarcinoma) cell proliferation assay. "
|
|
161
|
+
"Compound 7 (CHEMBL5171042) inhibited EGFR with an IC50 of 2.3 nM in a cell-free "
|
|
162
|
+
"enzymatic assay and showed an EC50 of 45 nM in A549 (human lung adenocarcinoma) cell proliferation assay. "
|
|
163
163
|
"Selectivity over ERBB2 was >100-fold (Ki = 0.8 nM vs 95 nM)."
|
|
164
164
|
),
|
|
165
165
|
extractions=[
|
|
@@ -170,7 +170,8 @@ _BIOACTIVITY_EXAMPLE_1 = lx.data.ExampleData(
|
|
|
170
170
|
"value": "2.3",
|
|
171
171
|
"unit": "nM",
|
|
172
172
|
"assay_type": "IC50",
|
|
173
|
-
"compound_name": "
|
|
173
|
+
"compound_name": "CHEMBL5171042",
|
|
174
|
+
"assay": "cell-free enzymatic assay",
|
|
174
175
|
},
|
|
175
176
|
),
|
|
176
177
|
lx.data.Extraction(
|
|
@@ -185,7 +186,8 @@ _BIOACTIVITY_EXAMPLE_1 = lx.data.ExampleData(
|
|
|
185
186
|
"value": "45",
|
|
186
187
|
"unit": "nM",
|
|
187
188
|
"assay_type": "EC50",
|
|
188
|
-
"compound_name": "
|
|
189
|
+
"compound_name": "CHEMBL5171042",
|
|
190
|
+
"assay": "A549 cell proliferation assay",
|
|
189
191
|
},
|
|
190
192
|
),
|
|
191
193
|
lx.data.Extraction(
|
|
@@ -204,7 +206,7 @@ _BIOACTIVITY_EXAMPLE_1 = lx.data.ExampleData(
|
|
|
204
206
|
"value": "0.8",
|
|
205
207
|
"unit": "nM",
|
|
206
208
|
"assay_type": "Ki",
|
|
207
|
-
"compound_name": "
|
|
209
|
+
"compound_name": "CHEMBL5171042",
|
|
208
210
|
},
|
|
209
211
|
),
|
|
210
212
|
],
|
|
@@ -337,6 +339,7 @@ _FULL_EXAMPLE_1 = lx.data.ExampleData(
|
|
|
337
339
|
"unit": "µM",
|
|
338
340
|
"assay_type": "IC50",
|
|
339
341
|
"compound_name": "Gefitinib",
|
|
342
|
+
"assay": "cell-free biochemical assay",
|
|
340
343
|
},
|
|
341
344
|
),
|
|
342
345
|
lx.data.Extraction(
|
|
@@ -357,6 +360,7 @@ _FULL_EXAMPLE_1 = lx.data.ExampleData(
|
|
|
357
360
|
"unit": "µM",
|
|
358
361
|
"assay_type": "IC50",
|
|
359
362
|
"compound_name": "Gefitinib",
|
|
363
|
+
"assay": "A431 cells",
|
|
360
364
|
},
|
|
361
365
|
),
|
|
362
366
|
lx.data.Extraction(
|
|
@@ -449,6 +453,7 @@ _TB_EXAMPLE_1 = lx.data.ExampleData(
|
|
|
449
453
|
"assay_type": "MIC",
|
|
450
454
|
"strain": "H37Rv",
|
|
451
455
|
"compound_name": "Bedaquiline",
|
|
456
|
+
"assay": "MABA",
|
|
452
457
|
},
|
|
453
458
|
),
|
|
454
459
|
lx.data.Extraction(
|
|
@@ -619,6 +624,8 @@ _TB_EXAMPLE_3 = lx.data.ExampleData(
|
|
|
619
624
|
"unit": "nM",
|
|
620
625
|
"assay_type": "IC50",
|
|
621
626
|
"compound_name": "Compound 14a",
|
|
627
|
+
"assay": "biochemical assay",
|
|
628
|
+
"target": "InhA",
|
|
622
629
|
},
|
|
623
630
|
),
|
|
624
631
|
lx.data.Extraction(
|
|
@@ -643,6 +650,7 @@ _TB_EXAMPLE_3 = lx.data.ExampleData(
|
|
|
643
650
|
"assay_type": "MIC90",
|
|
644
651
|
"strain": "H37Rv",
|
|
645
652
|
"compound_name": "Compound 14a",
|
|
653
|
+
"assay": "REMA",
|
|
646
654
|
},
|
|
647
655
|
),
|
|
648
656
|
lx.data.Extraction(
|
|
@@ -658,6 +666,7 @@ _TB_EXAMPLE_3 = lx.data.ExampleData(
|
|
|
658
666
|
"unit": "uM",
|
|
659
667
|
"assay_type": "EC50",
|
|
660
668
|
"compound_name": "Compound 14a",
|
|
669
|
+
"assay": "THP-1 macrophage infection assay",
|
|
661
670
|
},
|
|
662
671
|
),
|
|
663
672
|
lx.data.Extraction(
|
|
@@ -673,6 +682,7 @@ _TB_EXAMPLE_3 = lx.data.ExampleData(
|
|
|
673
682
|
"unit": "uM",
|
|
674
683
|
"assay_type": "MIC",
|
|
675
684
|
"compound_name": "Compound 14a",
|
|
685
|
+
"assay": "LORA",
|
|
676
686
|
},
|
|
677
687
|
),
|
|
678
688
|
lx.data.Extraction(
|
|
@@ -688,6 +698,7 @@ _TB_EXAMPLE_3 = lx.data.ExampleData(
|
|
|
688
698
|
"unit": "uM",
|
|
689
699
|
"assay_type": "CC50",
|
|
690
700
|
"compound_name": "Compound 14a",
|
|
701
|
+
"assay": "HepG2 cells",
|
|
691
702
|
},
|
|
692
703
|
),
|
|
693
704
|
],
|
|
@@ -843,6 +854,7 @@ _TB_EXAMPLE_5 = lx.data.ExampleData(
|
|
|
843
854
|
"unit": "nM",
|
|
844
855
|
"assay_type": "IC50",
|
|
845
856
|
"compound_name": "SACC-3060",
|
|
857
|
+
"target": "DprE1",
|
|
846
858
|
},
|
|
847
859
|
),
|
|
848
860
|
lx.data.Extraction(
|
|
@@ -860,12 +872,217 @@ _TB_EXAMPLE_5 = lx.data.ExampleData(
|
|
|
860
872
|
],
|
|
861
873
|
)
|
|
862
874
|
|
|
875
|
+
# Table as Docling emits it: the column header names the assay, the endpoint
|
|
876
|
+
# and unit come from the footnote, and every cell is its own bioactivity. The
|
|
877
|
+
# ID column, not the deck-local row label, names the compound.
|
|
878
|
+
_TB_EXAMPLE_6 = lx.data.ExampleData(
|
|
879
|
+
text=(
|
|
880
|
+
"Table 2. Whole-cell activity\n\n"
|
|
881
|
+
"| Cpd | ID | MABA | THP-1 | Vero |\n"
|
|
882
|
+
"| --- | --- | --- | --- | --- |\n"
|
|
883
|
+
"| 21a | SACC-4412 | 0.12 | 0.9 | >50 |\n"
|
|
884
|
+
"| 21b | CHEMBL5290347 | <0.03 | 2.4* | 18.5 |\n\n"
|
|
885
|
+
"MABA: MIC90 (µM) against M. tuberculosis H37Rv. THP-1: intracellular EC90 (µM). "
|
|
886
|
+
"Vero: CC50 (µM). *Single determination."
|
|
887
|
+
),
|
|
888
|
+
extractions=[
|
|
889
|
+
lx.data.Extraction(
|
|
890
|
+
extraction_class="assay",
|
|
891
|
+
extraction_text="MABA",
|
|
892
|
+
attributes={"assay_format": "whole-cell", "strain": "H37Rv"},
|
|
893
|
+
),
|
|
894
|
+
lx.data.Extraction(
|
|
895
|
+
extraction_class="assay",
|
|
896
|
+
extraction_text="THP-1",
|
|
897
|
+
attributes={"cell_line": "THP-1", "assay_format": "intracellular"},
|
|
898
|
+
),
|
|
899
|
+
lx.data.Extraction(
|
|
900
|
+
extraction_class="assay",
|
|
901
|
+
extraction_text="Vero",
|
|
902
|
+
attributes={"cell_line": "Vero", "assay_format": "cytotoxicity"},
|
|
903
|
+
),
|
|
904
|
+
lx.data.Extraction(
|
|
905
|
+
extraction_class="compound_name",
|
|
906
|
+
extraction_text="SACC-4412",
|
|
907
|
+
attributes={"synonyms": "21a"},
|
|
908
|
+
),
|
|
909
|
+
lx.data.Extraction(
|
|
910
|
+
extraction_class="bioactivity",
|
|
911
|
+
extraction_text="0.12",
|
|
912
|
+
attributes={
|
|
913
|
+
"value": "0.12",
|
|
914
|
+
"unit": "µM",
|
|
915
|
+
"assay_type": "MIC90",
|
|
916
|
+
"strain": "H37Rv",
|
|
917
|
+
"compound_name": "SACC-4412",
|
|
918
|
+
"assay": "MABA",
|
|
919
|
+
},
|
|
920
|
+
),
|
|
921
|
+
lx.data.Extraction(
|
|
922
|
+
extraction_class="bioactivity",
|
|
923
|
+
extraction_text="0.9",
|
|
924
|
+
attributes={
|
|
925
|
+
"value": "0.9",
|
|
926
|
+
"unit": "µM",
|
|
927
|
+
"assay_type": "EC90",
|
|
928
|
+
"compound_name": "SACC-4412",
|
|
929
|
+
"assay": "THP-1",
|
|
930
|
+
},
|
|
931
|
+
),
|
|
932
|
+
lx.data.Extraction(
|
|
933
|
+
extraction_class="bioactivity",
|
|
934
|
+
extraction_text=">50",
|
|
935
|
+
attributes={
|
|
936
|
+
"value": ">50",
|
|
937
|
+
"unit": "µM",
|
|
938
|
+
"assay_type": "CC50",
|
|
939
|
+
"compound_name": "SACC-4412",
|
|
940
|
+
"assay": "Vero",
|
|
941
|
+
},
|
|
942
|
+
),
|
|
943
|
+
lx.data.Extraction(
|
|
944
|
+
extraction_class="compound_name",
|
|
945
|
+
extraction_text="CHEMBL5290347",
|
|
946
|
+
attributes={"synonyms": "21b"},
|
|
947
|
+
),
|
|
948
|
+
lx.data.Extraction(
|
|
949
|
+
extraction_class="bioactivity",
|
|
950
|
+
extraction_text="<0.03",
|
|
951
|
+
attributes={
|
|
952
|
+
"value": "<0.03",
|
|
953
|
+
"unit": "µM",
|
|
954
|
+
"assay_type": "MIC90",
|
|
955
|
+
"strain": "H37Rv",
|
|
956
|
+
"compound_name": "CHEMBL5290347",
|
|
957
|
+
"assay": "MABA",
|
|
958
|
+
},
|
|
959
|
+
),
|
|
960
|
+
lx.data.Extraction(
|
|
961
|
+
extraction_class="bioactivity",
|
|
962
|
+
extraction_text="2.4",
|
|
963
|
+
attributes={
|
|
964
|
+
"value": "2.4",
|
|
965
|
+
"unit": "µM",
|
|
966
|
+
"assay_type": "EC90",
|
|
967
|
+
"compound_name": "CHEMBL5290347",
|
|
968
|
+
"assay": "THP-1",
|
|
969
|
+
},
|
|
970
|
+
),
|
|
971
|
+
lx.data.Extraction(
|
|
972
|
+
extraction_class="bioactivity",
|
|
973
|
+
extraction_text="18.5",
|
|
974
|
+
attributes={
|
|
975
|
+
"value": "18.5",
|
|
976
|
+
"unit": "µM",
|
|
977
|
+
"assay_type": "CC50",
|
|
978
|
+
"compound_name": "CHEMBL5290347",
|
|
979
|
+
"assay": "Vero",
|
|
980
|
+
},
|
|
981
|
+
),
|
|
982
|
+
],
|
|
983
|
+
)
|
|
984
|
+
|
|
985
|
+
# Strain panel: organism column headers are the strain, a protein column the
|
|
986
|
+
# target, a cytotoxicity column the assay; a column dosed with a second drug
|
|
987
|
+
# names the combination partner. The disease stays a disease entity.
|
|
988
|
+
_TB_EXAMPLE_7 = lx.data.ExampleData(
|
|
989
|
+
text=(
|
|
990
|
+
"Table 3. Antibacterial activity against hospital-acquired pneumonia isolates\n\n"
|
|
991
|
+
"| Cpd | ID | E. coli ATCC 25922 | K. pneumoniae BAA-1705 | A. baumannii ATCC 19606 "
|
|
992
|
+
"| ATCC 19606 + polymyxin B | HepG2 | hERG |\n"
|
|
993
|
+
"| --- | --- | --- | --- | --- | --- | --- | --- |\n"
|
|
994
|
+
"| 9c | CHEMBL5311027 | 0.25 | 1 | >64 | 2 | >100 | 12.5 |\n\n"
|
|
995
|
+
"MIC (µg/mL); polymyxin B at 0.5 µg/mL. HepG2: CC50 (µM). hERG: IC50 (µM)."
|
|
996
|
+
),
|
|
997
|
+
extractions=[
|
|
998
|
+
lx.data.Extraction(
|
|
999
|
+
extraction_class="disease",
|
|
1000
|
+
extraction_text="hospital-acquired pneumonia",
|
|
1001
|
+
attributes={"therapeutic_area": "infectious disease"},
|
|
1002
|
+
),
|
|
1003
|
+
lx.data.Extraction(
|
|
1004
|
+
extraction_class="compound_name",
|
|
1005
|
+
extraction_text="CHEMBL5311027",
|
|
1006
|
+
attributes={"synonyms": "9c"},
|
|
1007
|
+
),
|
|
1008
|
+
lx.data.Extraction(
|
|
1009
|
+
extraction_class="bioactivity",
|
|
1010
|
+
extraction_text="0.25",
|
|
1011
|
+
attributes={
|
|
1012
|
+
"value": "0.25",
|
|
1013
|
+
"unit": "µg/mL",
|
|
1014
|
+
"assay_type": "MIC",
|
|
1015
|
+
"strain": "E. coli ATCC 25922",
|
|
1016
|
+
"compound_name": "CHEMBL5311027",
|
|
1017
|
+
},
|
|
1018
|
+
),
|
|
1019
|
+
lx.data.Extraction(
|
|
1020
|
+
extraction_class="bioactivity",
|
|
1021
|
+
extraction_text="1",
|
|
1022
|
+
attributes={
|
|
1023
|
+
"value": "1",
|
|
1024
|
+
"unit": "µg/mL",
|
|
1025
|
+
"assay_type": "MIC",
|
|
1026
|
+
"strain": "K. pneumoniae BAA-1705",
|
|
1027
|
+
"compound_name": "CHEMBL5311027",
|
|
1028
|
+
},
|
|
1029
|
+
),
|
|
1030
|
+
lx.data.Extraction(
|
|
1031
|
+
extraction_class="bioactivity",
|
|
1032
|
+
extraction_text=">64",
|
|
1033
|
+
attributes={
|
|
1034
|
+
"value": ">64",
|
|
1035
|
+
"unit": "µg/mL",
|
|
1036
|
+
"assay_type": "MIC",
|
|
1037
|
+
"strain": "A. baumannii ATCC 19606",
|
|
1038
|
+
"compound_name": "CHEMBL5311027",
|
|
1039
|
+
},
|
|
1040
|
+
),
|
|
1041
|
+
lx.data.Extraction(
|
|
1042
|
+
extraction_class="bioactivity",
|
|
1043
|
+
extraction_text="2",
|
|
1044
|
+
attributes={
|
|
1045
|
+
"value": "2",
|
|
1046
|
+
"unit": "µg/mL",
|
|
1047
|
+
"assay_type": "MIC",
|
|
1048
|
+
"strain": "ATCC 19606",
|
|
1049
|
+
"combination": "polymyxin B",
|
|
1050
|
+
"compound_name": "CHEMBL5311027",
|
|
1051
|
+
},
|
|
1052
|
+
),
|
|
1053
|
+
lx.data.Extraction(
|
|
1054
|
+
extraction_class="bioactivity",
|
|
1055
|
+
extraction_text=">100",
|
|
1056
|
+
attributes={
|
|
1057
|
+
"value": ">100",
|
|
1058
|
+
"unit": "µM",
|
|
1059
|
+
"assay_type": "CC50",
|
|
1060
|
+
"compound_name": "CHEMBL5311027",
|
|
1061
|
+
"assay": "HepG2",
|
|
1062
|
+
},
|
|
1063
|
+
),
|
|
1064
|
+
lx.data.Extraction(
|
|
1065
|
+
extraction_class="bioactivity",
|
|
1066
|
+
extraction_text="12.5",
|
|
1067
|
+
attributes={
|
|
1068
|
+
"value": "12.5",
|
|
1069
|
+
"unit": "µM",
|
|
1070
|
+
"assay_type": "IC50",
|
|
1071
|
+
"compound_name": "CHEMBL5311027",
|
|
1072
|
+
"target": "hERG",
|
|
1073
|
+
},
|
|
1074
|
+
),
|
|
1075
|
+
],
|
|
1076
|
+
)
|
|
1077
|
+
|
|
863
1078
|
TB_EXAMPLES: list[lx.data.ExampleData] = [
|
|
864
1079
|
_TB_EXAMPLE_1,
|
|
865
1080
|
_TB_EXAMPLE_2,
|
|
866
1081
|
_TB_EXAMPLE_3,
|
|
867
1082
|
_TB_EXAMPLE_4,
|
|
868
1083
|
_TB_EXAMPLE_5,
|
|
1084
|
+
_TB_EXAMPLE_6,
|
|
1085
|
+
_TB_EXAMPLE_7,
|
|
869
1086
|
]
|
|
870
1087
|
|
|
871
1088
|
TB_CHEMISTRY_EXAMPLES: list[lx.data.ExampleData] = [
|
|
@@ -29,6 +29,10 @@ def extraction_to_entity(extraction: lx.data.Extraction) -> NEREntity:
|
|
|
29
29
|
attributes: dict[str, str] = {}
|
|
30
30
|
if extraction.attributes:
|
|
31
31
|
for key, value in extraction.attributes.items():
|
|
32
|
+
# Strict-schema providers (OpenAI) must emit every attribute key, so an
|
|
33
|
+
# unstated one arrives as null; leave it out rather than store "None".
|
|
34
|
+
if value is None:
|
|
35
|
+
continue
|
|
32
36
|
if isinstance(value, list):
|
|
33
37
|
attributes[key] = ", ".join(value)
|
|
34
38
|
else:
|
|
@@ -0,0 +1,164 @@
|
|
|
1
|
+
"""Prompt strings for each built-in EntityProfile."""
|
|
2
|
+
|
|
3
|
+
# Shared by every prompt that extracts bioactivity values. The attribute names
|
|
4
|
+
# must match the keys used in the few-shot examples: langextract derives the
|
|
5
|
+
# output schema from those keys, not from this text.
|
|
6
|
+
_BIOACTIVITY_ATTRIBUTES = (
|
|
7
|
+
"For each bioactivity value capture:\n"
|
|
8
|
+
"- value: the number as reported, keeping qualifiers ('>20.0', '<1.0') and dropping "
|
|
9
|
+
"footnote markers ('0.3*' → '0.3').\n"
|
|
10
|
+
"- unit: as written (nM, µM, µg/mL, mg/kg, %, h).\n"
|
|
11
|
+
"- assay_type: the endpoint only, in plain characters (CC₅₀ → CC50). Common endpoints: "
|
|
12
|
+
"IC50, IC90, EC50, EC90, pIC50, Ki, Kd, '% inhibition at <concentration>'; "
|
|
13
|
+
"MIC, MIC50, MIC90, MIC99, MBC, log10 CFU reduction; CC50, TC50, SI (selectivity index); "
|
|
14
|
+
"GI50, TGI, LC50; DC50, Dmax (degraders); ED50, ED90, %TGI, T/C (in vivo); "
|
|
15
|
+
"PRR (parasite reduction ratio), parasite clearance half-life.\n"
|
|
16
|
+
"- compound_name: the compound the value belongs to, by its name or identifier "
|
|
17
|
+
"(e.g. 'bedaquiline', 'BTZ043', 'CHEMBL1234', 'SACC-3000'); a document-local label "
|
|
18
|
+
"('7a', 'Compound 9b') only when nothing else names the compound.\n"
|
|
19
|
+
)
|
|
20
|
+
_ASSAY_ATTRIBUTE = (
|
|
21
|
+
"- assay: the assay, cell line or read-out the value was measured in, e.g. 'MABA', "
|
|
22
|
+
"'LORA', 'HepG2 MTT', 'THP-1 macrophage infection assay', 'asexual blood-stage', "
|
|
23
|
+
"'liver-stage', 'gametocyte', 'NCI-60', 'hERG', 'P. berghei 4-day suppressive test'. "
|
|
24
|
+
"null when no assay is stated.\n"
|
|
25
|
+
)
|
|
26
|
+
# TB gives each kind of context its own slot: what the value was measured in
|
|
27
|
+
# (assay), against (target, strain), and with (combination).
|
|
28
|
+
_TB_SLOT_ATTRIBUTES = (
|
|
29
|
+
"- assay: the assay system the value was measured in: a cell line, format or read-out "
|
|
30
|
+
"('MABA', 'LORA', 'HepG2 MTT', 'THP-1 macrophage infection assay', 'asexual blood-stage', "
|
|
31
|
+
"'AlphaScreen', 'FP', 'NCI-60', 'P. berghei 4-day suppressive test'). null when none is "
|
|
32
|
+
"stated.\n"
|
|
33
|
+
"- target: the protein the value was measured against: an enzyme, receptor, channel or "
|
|
34
|
+
"other molecular target, as written ('InhA', 'DprE1', 'hERG', 'KasA', 'PDE4B', "
|
|
35
|
+
"'BACE1'). null for whole-cell, organism and cytotoxicity values unless one is named.\n"
|
|
36
|
+
"- strain: the organism the value was measured against, as written: species, strain, "
|
|
37
|
+
"isolate or virus (H37Rv, M. tuberculosis, E. coli ATCC 25922, A. baumannii ATCC 19606, "
|
|
38
|
+
"Pf3D7, Dd2, S. mansoni, HIV-1, norovirus). An organism always goes here, never in assay or target, even "
|
|
39
|
+
"when it heads a table column or is all that names the assay. null when not stated.\n"
|
|
40
|
+
"- combination: a second compound dosed together with the one measured (a combination "
|
|
41
|
+
"or potentiation partner), as written ('avibactam', 'polymyxin B 0.5 µg/mL'). Not a "
|
|
42
|
+
"reference compound the value is relative to ('% of isoproterenol') or an agonist, substrate "
|
|
43
|
+
"or tracer the assay uses. null for a single agent.\n"
|
|
44
|
+
)
|
|
45
|
+
_BIOACTIVITY_TABLES = (
|
|
46
|
+
"In a data table, extract every cell value as its own bioactivity. An ID or registry "
|
|
47
|
+
"column, when present, supplies the compound_name and the row label is only its synonym; "
|
|
48
|
+
"otherwise the row label is the compound_name. A column header fills the attribute for "
|
|
49
|
+
"what it names (an assay or cell line, a protein target, an organism or strain), and the "
|
|
50
|
+
"endpoint and unit often come from the caption or a footnote ('CC50 in µM')."
|
|
51
|
+
)
|
|
52
|
+
|
|
53
|
+
CHEMISTRY_PROMPT = (
|
|
54
|
+
"Extract chemical entities from the text. "
|
|
55
|
+
"Include: compound names (generic names, IUPAC names, code names like 'Compound 5b', "
|
|
56
|
+
"brand names, and abbreviations), SMILES strings (the exact SMILES notation as written), "
|
|
57
|
+
"CAS registry numbers (e.g. '50-78-2'), and molecular formulas (e.g. 'C9H8O4'). "
|
|
58
|
+
"Do not infer or generate SMILES — only extract them if explicitly present in the text."
|
|
59
|
+
)
|
|
60
|
+
|
|
61
|
+
BIOLOGY_PROMPT = (
|
|
62
|
+
"Extract biological target entities from the text. "
|
|
63
|
+
"Include: protein targets (e.g. 'EGFR', 'CDK4/6', 'PD-L1'), gene names (e.g. 'KRAS', 'TP53', "
|
|
64
|
+
"'BRCA1'), receptor names, enzyme names, and pathway names. "
|
|
65
|
+
"For each target, capture the gene symbol if mentioned alongside a protein name. "
|
|
66
|
+
"Capture the organism context if specified (e.g. 'human', 'mouse')."
|
|
67
|
+
)
|
|
68
|
+
|
|
69
|
+
BIOACTIVITY_PROMPT = (
|
|
70
|
+
"Extract bioactivity measurements and assay data from the text: potency, selectivity, "
|
|
71
|
+
"cytotoxicity and efficacy values.\n"
|
|
72
|
+
+ _BIOACTIVITY_ATTRIBUTES
|
|
73
|
+
+ _ASSAY_ATTRIBUTE
|
|
74
|
+
+ _BIOACTIVITY_TABLES
|
|
75
|
+
+ "\nAlso extract assay descriptions: cell lines used (e.g. 'HeLa', 'A549'), assay formats "
|
|
76
|
+
"(e.g. 'cell viability', 'binding assay', 'enzymatic assay'), and organisms."
|
|
77
|
+
)
|
|
78
|
+
|
|
79
|
+
DISEASE_PROMPT = (
|
|
80
|
+
"Extract disease names and clinical indications from the text. "
|
|
81
|
+
"Include: cancer types (e.g. 'non-small cell lung cancer', 'AML', 'NSCLC'), "
|
|
82
|
+
"non-oncology diseases (e.g. 'type 2 diabetes', 'rheumatoid arthritis'), "
|
|
83
|
+
"and therapeutic areas (e.g. 'oncology', 'CNS'). "
|
|
84
|
+
"Capture both full names and abbreviations. "
|
|
85
|
+
"For each disease, note the therapeutic area if discernible from context."
|
|
86
|
+
)
|
|
87
|
+
|
|
88
|
+
FULL_PROMPT = (
|
|
89
|
+
"Extract all drug discovery entities from the text. "
|
|
90
|
+
"This includes:\n"
|
|
91
|
+
"- Chemical entities: compound names (generic, IUPAC, code names, brand names), "
|
|
92
|
+
"SMILES strings (only if explicitly written), CAS numbers, molecular formulas.\n"
|
|
93
|
+
"- Biological targets: protein names, gene names, receptor names, enzyme names, pathways.\n"
|
|
94
|
+
"- Bioactivity data: potency, selectivity, cytotoxicity and efficacy values "
|
|
95
|
+
"(attributes below).\n"
|
|
96
|
+
"- Assay information: cell lines, assay formats, experimental organisms.\n"
|
|
97
|
+
"- Diseases and indications: cancer types, disease names, therapeutic areas.\n"
|
|
98
|
+
"- Mechanisms of action: binding modes, inhibition types, selectivity descriptions.\n"
|
|
99
|
+
+ _BIOACTIVITY_ATTRIBUTES
|
|
100
|
+
+ _ASSAY_ATTRIBUTE
|
|
101
|
+
+ _BIOACTIVITY_TABLES
|
|
102
|
+
+ "\nExtract only what is explicitly stated; do not infer or generate values."
|
|
103
|
+
)
|
|
104
|
+
|
|
105
|
+
# ── Tuberculosis early drug discovery prompts ──────────────────────────
|
|
106
|
+
|
|
107
|
+
TB_PROMPT = (
|
|
108
|
+
"Extract drug discovery entities from this tuberculosis research text.\n\n"
|
|
109
|
+
+ _BIOACTIVITY_ATTRIBUTES
|
|
110
|
+
+ _TB_SLOT_ATTRIBUTES
|
|
111
|
+
+ _BIOACTIVITY_TABLES
|
|
112
|
+
+ "\n\n"
|
|
113
|
+
"DISAMBIGUATION RULES:\n"
|
|
114
|
+
"- Mycobacterial proteins (e.g. ClpC1, DprE1, InhA, AtpE, MmpL3, QcrB) "
|
|
115
|
+
"are biological targets, NOT compounds.\n"
|
|
116
|
+
"- Rv locus tags (Rv3790, Rv1484), UniProt IDs (P9WPS1), and PDB codes "
|
|
117
|
+
"are accession_number, not target or gene_name.\n"
|
|
118
|
+
"- Compound registry and programme identifiers (CHEMBL4521987, ZINC000012345678, "
|
|
119
|
+
"DB00945, SACC-3060, TBDA-01187, GSK3036656) are compound_name; only Rv locus "
|
|
120
|
+
"tags, UniProt IDs, PDB codes and RefSeq IDs are accession_number.\n"
|
|
121
|
+
"- A compound with both a document-local label (a number with an optional letter: 7a, "
|
|
122
|
+
"12, 'Compound 9b') and a longer identifier (registry ID; programme, partner or vendor "
|
|
123
|
+
"code: CHEMBL…, SACC-…, a ChemBridge number) is extracted once, as the identifier, with "
|
|
124
|
+
"the label in synonyms; its bioactivities use the identifier as compound_name. A label "
|
|
125
|
+
"is the compound_name only when the document gives nothing else.\n"
|
|
126
|
+
"- Enzyme descriptions like 'enoyl-ACP reductase' are product, not target.\n"
|
|
127
|
+
"- 'cell wall', 'lipid metabolism' are functional_category, not mechanism_of_action.\n"
|
|
128
|
+
"- 'fragment screening', 'biochemical assay' are screening_method, not assay.\n"
|
|
129
|
+
"- A disease or infection is a disease entity wherever it is named, in prose, a heading "
|
|
130
|
+
"or a table cell (tuberculosis, HIV infection, malaria, schistosomiasis, glioblastoma in 'U87 (glioblastoma)'), "
|
|
131
|
+
"even when its organism also fills a bioactivity's strain.\n"
|
|
132
|
+
"- Use target for proteins in a drug-targeting context, gene_name for loci, "
|
|
133
|
+
"protein_name for non-drug-target proteins.\n\n"
|
|
134
|
+
"Extract only what is explicitly stated; do not infer or generate values."
|
|
135
|
+
)
|
|
136
|
+
|
|
137
|
+
TB_CHEMISTRY_PROMPT = (
|
|
138
|
+
"Extract chemical entities from this tuberculosis drug discovery text. "
|
|
139
|
+
"Include compound names, SMILES (only if explicitly written), CAS numbers, "
|
|
140
|
+
"and molecular formulas. "
|
|
141
|
+
"Mycobacterial proteins (ClpC1, DprE1, InhA, AtpE, etc.) are NOT compounds. "
|
|
142
|
+
"Extract only what is explicitly stated; do not infer or generate values."
|
|
143
|
+
)
|
|
144
|
+
|
|
145
|
+
TB_BIOLOGY_PROMPT = (
|
|
146
|
+
"Extract biological entities from this tuberculosis research text. "
|
|
147
|
+
"Use target for proteins in a drug-targeting context, gene_name for loci, "
|
|
148
|
+
"protein_name for non-drug-target proteins. "
|
|
149
|
+
"Rv locus tags and UniProt IDs are accession_number. "
|
|
150
|
+
"Enzyme descriptions (e.g. 'enoyl-ACP reductase') are product. "
|
|
151
|
+
"Protein functional categories (e.g. 'cell wall', 'lipid metabolism') "
|
|
152
|
+
"are functional_category. "
|
|
153
|
+
"Extract only what is explicitly stated; do not infer or generate values."
|
|
154
|
+
)
|
|
155
|
+
|
|
156
|
+
# Framing for NERExtractor.extract(context=...), rendered by langextract ahead of
|
|
157
|
+
# the examples.
|
|
158
|
+
DOCUMENT_CONTEXT = (
|
|
159
|
+
"Document context, from elsewhere in the same document. It is not the text to extract "
|
|
160
|
+
"from: take no entity and no value from it, and do not pair a value in the text with a "
|
|
161
|
+
"number or claim made here. Use it only to fill a bioactivity attribute the text leaves "
|
|
162
|
+
"unstated, such as the target or organism of a table, when the context names one for "
|
|
163
|
+
"the whole document.\n"
|
|
164
|
+
)
|
|
@@ -3,15 +3,43 @@
|
|
|
3
3
|
from __future__ import annotations
|
|
4
4
|
|
|
5
5
|
import logging
|
|
6
|
+
import re
|
|
6
7
|
|
|
7
8
|
import langextract as lx
|
|
8
9
|
|
|
10
|
+
from structflo.ner import _prompts
|
|
9
11
|
from structflo.ner._entities import NERResult
|
|
10
12
|
from structflo.ner._mapping import annotated_doc_to_result
|
|
11
13
|
from structflo.ner.profiles import FULL, EntityProfile
|
|
12
14
|
|
|
13
15
|
logger = logging.getLogger(__name__)
|
|
14
16
|
|
|
17
|
+
# A table cell holding only a value: '0.94', '>32', '2.4*', '0.62µM'.
|
|
18
|
+
_VALUE_CELL = re.compile(r"^[<>≤≥~=]*\s*\d[\d,]*\.?\d*\s*\*?\s*(%|[nµu]M|µg/mL|mg/kg)?$")
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def _bioactivity_count(doc: lx.data.AnnotatedDocument) -> int:
|
|
22
|
+
return sum(1 for ext in doc.extractions if ext.extraction_class == "bioactivity")
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def _table_values_dropped(text: str, doc: lx.data.AnnotatedDocument) -> bool:
|
|
26
|
+
"""True when fewer than half of a page's table values came back as bioactivities.
|
|
27
|
+
|
|
28
|
+
Models now and then return a complete, valid answer that lists a table's
|
|
29
|
+
compounds but leaves out its values. On the ChEMBL-gold decks this fires on
|
|
30
|
+
~2% of pages and catches ~60% of such dropouts.
|
|
31
|
+
"""
|
|
32
|
+
# ponytail: markdown tables only; values in prose and qualitative results
|
|
33
|
+
# are not checked, so those dropouts still pass silently.
|
|
34
|
+
cells = sum(
|
|
35
|
+
1
|
|
36
|
+
for line in text.splitlines()
|
|
37
|
+
if line.lstrip().startswith("|")
|
|
38
|
+
for cell in line.strip().strip("|").split("|")
|
|
39
|
+
if _VALUE_CELL.match(cell.strip())
|
|
40
|
+
)
|
|
41
|
+
return cells >= 5 and _bioactivity_count(doc) < 0.5 * cells
|
|
42
|
+
|
|
15
43
|
|
|
16
44
|
class NERExtractor:
|
|
17
45
|
"""Extract drug discovery entities from text with zero configuration.
|
|
@@ -90,12 +118,17 @@ class NERExtractor:
|
|
|
90
118
|
self,
|
|
91
119
|
text: str | list[str],
|
|
92
120
|
profile: EntityProfile | None = None,
|
|
121
|
+
context: str | None = None,
|
|
93
122
|
) -> NERResult | list[NERResult]:
|
|
94
123
|
"""Extract drug discovery entities from text.
|
|
95
124
|
|
|
96
125
|
Args:
|
|
97
126
|
text: Input text (or list of texts) to process.
|
|
98
127
|
profile: Override the default profile for this call only.
|
|
128
|
+
context: Text from elsewhere in the same document (its title or
|
|
129
|
+
summary). The model reads it to fill attributes the text leaves
|
|
130
|
+
unstated, such as the target of a table whose target is named
|
|
131
|
+
only on another page; no entities are extracted from it.
|
|
99
132
|
|
|
100
133
|
Returns:
|
|
101
134
|
A :class:`NERResult` for a single string input, or a list of
|
|
@@ -107,7 +140,13 @@ class NERExtractor:
|
|
|
107
140
|
|
|
108
141
|
results = []
|
|
109
142
|
for single_text in texts:
|
|
110
|
-
doc = self._run_extraction(single_text, active_profile)
|
|
143
|
+
doc = self._run_extraction(single_text, active_profile, context)
|
|
144
|
+
if "bioactivity" in active_profile.entity_classes and _table_values_dropped(
|
|
145
|
+
single_text, doc
|
|
146
|
+
):
|
|
147
|
+
logger.warning("Table values missing from extraction; retrying once.")
|
|
148
|
+
retry = self._run_extraction(single_text, active_profile, context)
|
|
149
|
+
doc = max(doc, retry, key=_bioactivity_count)
|
|
111
150
|
results.append(annotated_doc_to_result(doc, single_text))
|
|
112
151
|
|
|
113
152
|
return results if is_batch else results[0]
|
|
@@ -198,6 +237,7 @@ class NERExtractor:
|
|
|
198
237
|
self,
|
|
199
238
|
text: str,
|
|
200
239
|
profile: EntityProfile,
|
|
240
|
+
context: str | None = None,
|
|
201
241
|
) -> lx.data.AnnotatedDocument:
|
|
202
242
|
"""Call lx.extract and return the AnnotatedDocument."""
|
|
203
243
|
examples = self._build_examples(profile)
|
|
@@ -206,6 +246,8 @@ class NERExtractor:
|
|
|
206
246
|
kwargs: dict = dict(self._langextract_kwargs)
|
|
207
247
|
kwargs.setdefault("use_schema_constraints", True)
|
|
208
248
|
kwargs.setdefault("show_progress", False)
|
|
249
|
+
if context:
|
|
250
|
+
kwargs["additional_context"] = _prompts.DOCUMENT_CONTEXT + context
|
|
209
251
|
|
|
210
252
|
if self._provider is not None:
|
|
211
253
|
# Explicit provider → deterministic routing via ModelConfig.
|
|
@@ -236,4 +278,29 @@ class NERExtractor:
|
|
|
236
278
|
|
|
237
279
|
# Post-process: drop extractions with hallucinated entity classes
|
|
238
280
|
allowed = set(profile.entity_classes)
|
|
239
|
-
|
|
281
|
+
doc = self._filter_extractions(doc, allowed)
|
|
282
|
+
if context:
|
|
283
|
+
doc = self._drop_context_copies(doc, context)
|
|
284
|
+
return doc
|
|
285
|
+
|
|
286
|
+
@staticmethod
|
|
287
|
+
def _drop_context_copies(
|
|
288
|
+
doc: lx.data.AnnotatedDocument,
|
|
289
|
+
context: str,
|
|
290
|
+
) -> lx.data.AnnotatedDocument:
|
|
291
|
+
"""Drop extractions the model took from the context instead of the text.
|
|
292
|
+
|
|
293
|
+
The prompt forbids it, but models still extract from the context now and
|
|
294
|
+
then. Such an extraction does not align to the text, and its text occurs
|
|
295
|
+
in the context but not in the text (an unaligned one that is in the text
|
|
296
|
+
is a missed alignment, not a copy).
|
|
297
|
+
"""
|
|
298
|
+
ctx, text = context.casefold(), doc.text.casefold()
|
|
299
|
+
kept = [
|
|
300
|
+
ext
|
|
301
|
+
for ext in doc.extractions
|
|
302
|
+
if ext.char_interval is not None
|
|
303
|
+
or ext.extraction_text.casefold() not in ctx
|
|
304
|
+
or ext.extraction_text.casefold() in text
|
|
305
|
+
]
|
|
306
|
+
return lx.data.AnnotatedDocument(text=doc.text, extractions=kept)
|
|
@@ -66,7 +66,7 @@ so a deployment's own gazetteer decides which ones apply:
|
|
|
66
66
|
| Seed | Auto-derived Pattern | Matches |
|
|
67
67
|
|---|---|---|
|
|
68
68
|
| `Rv0005` | `Rv\d{4}[c]?` | All Rv locus tags |
|
|
69
|
-
| `MT0005` | `MT\w
|
|
69
|
+
| `MT0005` | `MT[A-Z]{0,2}\d{2,}\w*` | Mycobrowser IDs (not MTT, MTX) |
|
|
70
70
|
| `P9WGR1` | `[OPQ][0-9][A-Z0-9]{3}[0-9]` | UniProt accessions |
|
|
71
71
|
| `4TZK` | `(?=[A-Z0-9]{0,3}[A-Z])[1-9][A-Z0-9]{3}` | PDB codes |
|
|
72
72
|
| `WP_003407354` | `WP_\d+` | NCBI RefSeq proteins |
|
|
@@ -89,8 +89,13 @@ _COMPOUND_ID_PATTERNS: list[IdPattern] = [
|
|
|
89
89
|
_ACCESSION_PATTERNS: list[tuple[re.Pattern[str], re.Pattern[str], str]] = [
|
|
90
90
|
# Rv locus tags: Rv0005, Rv3854c
|
|
91
91
|
(re.compile(r"^Rv\d{4}[c]?$"), re.compile(r"\bRv\d{4}[c]?\b"), "Rv locus tag"),
|
|
92
|
-
# Mycobrowser MT IDs: MT0005,
|
|
93
|
-
(
|
|
92
|
+
# Mycobrowser MT IDs: MT0005, MT18B_0001, MTB000001. The digit run keeps
|
|
93
|
+
# MTT (viability assay), MTX and MTD from matching.
|
|
94
|
+
(
|
|
95
|
+
re.compile(r"^MT[A-Z]{0,2}\d{2,}\w*$"),
|
|
96
|
+
re.compile(r"\bMT[A-Z]{0,2}\d{2,}\w*\b"),
|
|
97
|
+
"Mycobrowser ID",
|
|
98
|
+
),
|
|
94
99
|
# UniProt accessions: P9WGR1, O53617
|
|
95
100
|
(
|
|
96
101
|
re.compile(r"^[OPQ][0-9][A-Z0-9]{3}[0-9]$"),
|
|
@@ -275,8 +275,9 @@ def _build_position_map(original: str, normalized: str) -> list[int]:
|
|
|
275
275
|
|
|
276
276
|
# Advance original pointer to find the matching character
|
|
277
277
|
while orig_idx < len(original):
|
|
278
|
-
|
|
279
|
-
|
|
278
|
+
# normalize() the char too: a newline became " " and an en dash "-",
|
|
279
|
+
# and comparing raw chars would desync every position after them.
|
|
280
|
+
if (normalize(original[orig_idx]) or " ") == norm_char:
|
|
280
281
|
position_map.append(orig_idx)
|
|
281
282
|
orig_idx += 1
|
|
282
283
|
break
|
|
@@ -7,6 +7,7 @@ from unittest.mock import MagicMock
|
|
|
7
7
|
import langextract as lx
|
|
8
8
|
|
|
9
9
|
from structflo.ner import (
|
|
10
|
+
BIOACTIVITY,
|
|
10
11
|
BIOLOGY,
|
|
11
12
|
CHEMISTRY,
|
|
12
13
|
FULL,
|
|
@@ -121,6 +122,50 @@ class TestNERExtractorExtract:
|
|
|
121
122
|
result = extractor.extract("My source text")
|
|
122
123
|
assert result.source_text == "My source text"
|
|
123
124
|
|
|
125
|
+
_TABLE = "| cpd | MIC | CC50 |\n| --- | --- | --- |\n" + "".join(
|
|
126
|
+
f"| C{i} | {i}.5 | >50 |\n" for i in range(5)
|
|
127
|
+
)
|
|
128
|
+
|
|
129
|
+
def _bio_doc(self, n: int) -> lx.data.AnnotatedDocument:
|
|
130
|
+
return _make_annotated_doc(
|
|
131
|
+
[
|
|
132
|
+
lx.data.Extraction(extraction_class="bioactivity", extraction_text=f"{i}.5")
|
|
133
|
+
for i in range(n)
|
|
134
|
+
]
|
|
135
|
+
)
|
|
136
|
+
|
|
137
|
+
def test_retries_once_when_table_values_are_missing(self):
|
|
138
|
+
extractor = NERExtractor(profile=TB)
|
|
139
|
+
extractor._run_extraction = MagicMock(side_effect=[self._bio_doc(0), self._bio_doc(10)])
|
|
140
|
+
result = extractor.extract(self._TABLE)
|
|
141
|
+
assert extractor._run_extraction.call_count == 2
|
|
142
|
+
assert len(result.bioactivities) == 10
|
|
143
|
+
|
|
144
|
+
def test_no_retry_when_table_values_came_back(self):
|
|
145
|
+
extractor = NERExtractor(profile=TB)
|
|
146
|
+
extractor._run_extraction = MagicMock(return_value=self._bio_doc(10))
|
|
147
|
+
extractor.extract(self._TABLE)
|
|
148
|
+
assert extractor._run_extraction.call_count == 1
|
|
149
|
+
|
|
150
|
+
def test_no_retry_for_profiles_without_bioactivity(self):
|
|
151
|
+
extractor = NERExtractor(profile=CHEMISTRY)
|
|
152
|
+
extractor._run_extraction = MagicMock(return_value=self._bio_doc(0))
|
|
153
|
+
extractor.extract(self._TABLE)
|
|
154
|
+
assert extractor._run_extraction.call_count == 1
|
|
155
|
+
|
|
156
|
+
def test_null_attributes_are_absent_not_none_strings(self):
|
|
157
|
+
# OpenAI strict mode returns every schema key, null when unstated.
|
|
158
|
+
extractor = self._extractor_with_mock(
|
|
159
|
+
[
|
|
160
|
+
lx.data.Extraction(
|
|
161
|
+
extraction_class="bioactivity",
|
|
162
|
+
extraction_text="IC50 = 3 nM",
|
|
163
|
+
attributes={"value": "3", "strain": None, "assay": None},
|
|
164
|
+
)
|
|
165
|
+
]
|
|
166
|
+
)
|
|
167
|
+
assert extractor.extract("text").bioactivities[0].attributes == {"value": "3"}
|
|
168
|
+
|
|
124
169
|
|
|
125
170
|
class TestProviderRouting:
|
|
126
171
|
"""Explicit provider selection routes deterministically via ModelConfig,
|
|
@@ -180,6 +225,24 @@ class TestProviderRouting:
|
|
|
180
225
|
assert pk["base_url"] == "http://ollama:11434"
|
|
181
226
|
assert pk["num_ctx"] == 8192
|
|
182
227
|
|
|
228
|
+
def test_context_reaches_langextract_framed(self, monkeypatch):
|
|
229
|
+
from structflo.ner import extractor as extractor_mod
|
|
230
|
+
|
|
231
|
+
captured = {}
|
|
232
|
+
|
|
233
|
+
def fake_extract(**kwargs):
|
|
234
|
+
captured.update(kwargs)
|
|
235
|
+
return _make_annotated_doc([])
|
|
236
|
+
|
|
237
|
+
monkeypatch.setattr(extractor_mod.lx, "extract", fake_extract)
|
|
238
|
+
NERExtractor().extract("IC50s table", context="Deck on InhA inhibitors.")
|
|
239
|
+
ctx = captured["additional_context"]
|
|
240
|
+
assert ctx.startswith("Document context") and ctx.endswith("Deck on InhA inhibitors.")
|
|
241
|
+
|
|
242
|
+
def test_no_context_passes_no_additional_context(self, monkeypatch):
|
|
243
|
+
captured = self._run_and_capture(monkeypatch)
|
|
244
|
+
assert "additional_context" not in captured
|
|
245
|
+
|
|
183
246
|
def test_cloud_provider_kwargs_carry_api_key(self, monkeypatch):
|
|
184
247
|
captured = self._run_and_capture(
|
|
185
248
|
monkeypatch, provider="openai", model_id="gpt-4o", api_key="sk-test"
|
|
@@ -234,6 +297,27 @@ class TestFilterExtractions:
|
|
|
234
297
|
assert len(filtered.extractions) == 0
|
|
235
298
|
|
|
236
299
|
|
|
300
|
+
class TestDropContextCopies:
|
|
301
|
+
def test_drops_only_unaligned_extractions_found_in_context(self):
|
|
302
|
+
aligned = lx.data.Extraction(
|
|
303
|
+
extraction_class="target",
|
|
304
|
+
extraction_text="InhA",
|
|
305
|
+
char_interval=lx.data.CharInterval(start_pos=0, end_pos=4),
|
|
306
|
+
)
|
|
307
|
+
copied = lx.data.Extraction(extraction_class="compound_name", extraction_text="Isoniazid")
|
|
308
|
+
unaligned = lx.data.Extraction(extraction_class="compound_name", extraction_text="7a")
|
|
309
|
+
# In the context too, but also on the page: a missed alignment, not a copy.
|
|
310
|
+
on_page = lx.data.Extraction(extraction_class="disease", extraction_text="malaria")
|
|
311
|
+
doc = lx.data.AnnotatedDocument(
|
|
312
|
+
text="InhA IC50s for 7a; malaria panel",
|
|
313
|
+
extractions=[aligned, copied, unaligned, on_page],
|
|
314
|
+
)
|
|
315
|
+
kept = NERExtractor._drop_context_copies(
|
|
316
|
+
doc, "InhA inhibitors that match isoniazid, and a malaria screen"
|
|
317
|
+
)
|
|
318
|
+
assert kept.extractions == [aligned, unaligned, on_page]
|
|
319
|
+
|
|
320
|
+
|
|
237
321
|
class TestBuildExamples:
|
|
238
322
|
def test_extra_examples_appended(self):
|
|
239
323
|
extra = lx.data.ExampleData(text="extra", extractions=[])
|
|
@@ -284,8 +368,33 @@ class TestTBProfile:
|
|
|
284
368
|
for cls in TB_BIOLOGY.entity_classes:
|
|
285
369
|
assert cls in TB.entity_classes
|
|
286
370
|
|
|
371
|
+
def test_bioactivity_schema_has_assay(self):
|
|
372
|
+
"""langextract builds the output schema from example attribute keys, so an
|
|
373
|
+
`assay` slot exists only if some bioactivity example carries one."""
|
|
374
|
+
for profile in (TB, BIOACTIVITY, FULL):
|
|
375
|
+
keys = {
|
|
376
|
+
k
|
|
377
|
+
for ex in profile.examples
|
|
378
|
+
for e in ex.extractions
|
|
379
|
+
if e.extraction_class == "bioactivity"
|
|
380
|
+
for k in (e.attributes or {})
|
|
381
|
+
}
|
|
382
|
+
assert {"value", "unit", "assay_type", "compound_name", "assay"} <= keys, profile.name
|
|
383
|
+
|
|
384
|
+
def test_tb_bioactivity_schema_has_context_slots(self):
|
|
385
|
+
"""TB gives what a value was measured against (target, strain) and with
|
|
386
|
+
(combination) their own slots, apart from assay."""
|
|
387
|
+
keys = {
|
|
388
|
+
k
|
|
389
|
+
for ex in TB.examples
|
|
390
|
+
for e in ex.extractions
|
|
391
|
+
if e.extraction_class == "bioactivity"
|
|
392
|
+
for k in (e.attributes or {})
|
|
393
|
+
}
|
|
394
|
+
assert {"target", "strain", "combination"} <= keys
|
|
395
|
+
|
|
287
396
|
def test_tb_examples_count(self):
|
|
288
|
-
assert len(TB.examples) ==
|
|
397
|
+
assert len(TB.examples) == 7
|
|
289
398
|
assert len(TB_CHEMISTRY.examples) == 2
|
|
290
399
|
assert len(TB_BIOLOGY.examples) == 2
|
|
291
400
|
|
|
@@ -180,6 +180,13 @@ class TestDeriveIdPatterns:
|
|
|
180
180
|
assert not pdb.regex.search("in 2019 the trial")
|
|
181
181
|
assert not pdb.regex.search("n = 1234")
|
|
182
182
|
|
|
183
|
+
def test_mycobrowser_pattern_needs_locus_digits(self):
|
|
184
|
+
"""MTT (viability assay), MTX and MTD are not Mycobrowser loci."""
|
|
185
|
+
mt = next(p for p in _accessions(["MT18B_0001"]) if p.description == "Mycobrowser ID")
|
|
186
|
+
for locus in ["MT0005", "MT18B_0001", "MTB000001"]:
|
|
187
|
+
assert mt.regex.fullmatch(locus), locus
|
|
188
|
+
assert not mt.regex.search("HepG2 MTT, MTX and MTD read-outs")
|
|
189
|
+
|
|
183
190
|
def test_cas_check_digit_rejects_dates(self):
|
|
184
191
|
cas = next(p for p in derive_id_patterns({}) if p.description == "CAS number")
|
|
185
192
|
assert cas.validate("50-78-2") # aspirin
|
|
@@ -246,6 +253,13 @@ class TestGazetteerMatcher:
|
|
|
246
253
|
m = inha_matches[0]
|
|
247
254
|
assert text[m.char_start : m.char_end] == "InhA"
|
|
248
255
|
|
|
256
|
+
def test_char_offsets_survive_newlines(self):
|
|
257
|
+
"""normalize() folds a newline into a space; offsets after it must not drift."""
|
|
258
|
+
matcher = self._simple_matcher(fuzzy_threshold=0)
|
|
259
|
+
text = "Targets:\nINHA and\n\nDPRE1."
|
|
260
|
+
found = {m.canonical: text[m.char_start : m.char_end] for m in matcher.match(text)}
|
|
261
|
+
assert found == {"InhA": "INHA", "DprE1": "DPRE1"}
|
|
262
|
+
|
|
249
263
|
def test_regex_accession_matching(self):
|
|
250
264
|
matcher = GazetteerMatcher(
|
|
251
265
|
gazetteers={"target": ["InhA"]},
|
|
@@ -917,7 +917,7 @@ wheels = [
|
|
|
917
917
|
|
|
918
918
|
[[package]]
|
|
919
919
|
name = "langextract"
|
|
920
|
-
version = "1.
|
|
920
|
+
version = "1.7.0"
|
|
921
921
|
source = { registry = "https://pypi.org/simple" }
|
|
922
922
|
dependencies = [
|
|
923
923
|
{ name = "absl-py" },
|
|
@@ -940,9 +940,9 @@ dependencies = [
|
|
|
940
940
|
{ name = "tqdm" },
|
|
941
941
|
{ name = "typing-extensions" },
|
|
942
942
|
]
|
|
943
|
-
sdist = { url = "https://files.pythonhosted.org/packages/
|
|
943
|
+
sdist = { url = "https://files.pythonhosted.org/packages/e3/e7/6cf877cdd3d5d3a479c0688ff2b63c665c4319ae35ba00734c9fef36a1d2/langextract-1.7.0.tar.gz", hash = "sha256:cb6dc515583b0fdc0f9eff9dce61b2a3c7c84b0668add4dd4e7be364f4c188ba", size = 147514, upload-time = "2026-09-13T18:58:06.376Z" }
|
|
944
944
|
wheels = [
|
|
945
|
-
{ url = "https://files.pythonhosted.org/packages/
|
|
945
|
+
{ url = "https://files.pythonhosted.org/packages/ae/ec/2883a93872701045971ea77e30434d72556a557f188df575b154d179854c/langextract-1.7.0-py3-none-any.whl", hash = "sha256:908f8ec696ab13578cdc3862efcf1a4fda929ae426410d99e86977d50e52136d", size = 154297, upload-time = "2026-09-13T18:58:04.896Z" },
|
|
946
946
|
]
|
|
947
947
|
|
|
948
948
|
[[package]]
|
|
@@ -2170,7 +2170,7 @@ dev = [
|
|
|
2170
2170
|
|
|
2171
2171
|
[package.metadata]
|
|
2172
2172
|
requires-dist = [
|
|
2173
|
-
{ name = "langextract", specifier = ">=1.
|
|
2173
|
+
{ name = "langextract", specifier = ">=1.7.0" },
|
|
2174
2174
|
{ name = "pandas", marker = "extra == 'dataframe'", specifier = ">=1.5" },
|
|
2175
2175
|
{ name = "pyyaml", specifier = ">=6.0" },
|
|
2176
2176
|
{ name = "rapidfuzz", specifier = ">=3.0" },
|
|
@@ -1,95 +0,0 @@
|
|
|
1
|
-
"""Prompt strings for each built-in EntityProfile."""
|
|
2
|
-
|
|
3
|
-
CHEMISTRY_PROMPT = (
|
|
4
|
-
"Extract chemical entities from the text. "
|
|
5
|
-
"Include: compound names (generic names, IUPAC names, code names like 'Compound 5b', "
|
|
6
|
-
"brand names, and abbreviations), SMILES strings (the exact SMILES notation as written), "
|
|
7
|
-
"CAS registry numbers (e.g. '50-78-2'), and molecular formulas (e.g. 'C9H8O4'). "
|
|
8
|
-
"Do not infer or generate SMILES — only extract them if explicitly present in the text."
|
|
9
|
-
)
|
|
10
|
-
|
|
11
|
-
BIOLOGY_PROMPT = (
|
|
12
|
-
"Extract biological target entities from the text. "
|
|
13
|
-
"Include: protein targets (e.g. 'EGFR', 'CDK4/6', 'PD-L1'), gene names (e.g. 'KRAS', 'TP53', "
|
|
14
|
-
"'BRCA1'), receptor names, enzyme names, and pathway names. "
|
|
15
|
-
"For each target, capture the gene symbol if mentioned alongside a protein name. "
|
|
16
|
-
"Capture the organism context if specified (e.g. 'human', 'mouse')."
|
|
17
|
-
)
|
|
18
|
-
|
|
19
|
-
BIOACTIVITY_PROMPT = (
|
|
20
|
-
"Extract bioactivity measurements and assay data from the text. "
|
|
21
|
-
"Include: potency values (IC50, MIC, EC50, Ki, Kd, GI50, CC50), selectivity ratios, "
|
|
22
|
-
"percent inhibition values, and Hill coefficients. "
|
|
23
|
-
"For each value, capture the numeric value, unit (nM, µM, mM, µg/mL, ng/mL), and measurement type. "
|
|
24
|
-
"Also capture the compound — its name, code name, or identifier (e.g. 'Compound 7', 'BTZ043', "
|
|
25
|
-
"'CHEMBL1234', 'SACC-3000') — that the measurement applies to, as a 'compound_name' attribute, "
|
|
26
|
-
"if it can be identified from the surrounding context. "
|
|
27
|
-
"Also extract assay descriptions: cell lines used (e.g. 'HeLa', 'A549'), assay formats "
|
|
28
|
-
"(e.g. 'cell viability', 'binding assay', 'enzymatic assay'), and organisms."
|
|
29
|
-
)
|
|
30
|
-
|
|
31
|
-
DISEASE_PROMPT = (
|
|
32
|
-
"Extract disease names and clinical indications from the text. "
|
|
33
|
-
"Include: cancer types (e.g. 'non-small cell lung cancer', 'AML', 'NSCLC'), "
|
|
34
|
-
"non-oncology diseases (e.g. 'type 2 diabetes', 'rheumatoid arthritis'), "
|
|
35
|
-
"and therapeutic areas (e.g. 'oncology', 'CNS'). "
|
|
36
|
-
"Capture both full names and abbreviations. "
|
|
37
|
-
"For each disease, note the therapeutic area if discernible from context."
|
|
38
|
-
)
|
|
39
|
-
|
|
40
|
-
FULL_PROMPT = (
|
|
41
|
-
"Extract all drug discovery entities from the text. "
|
|
42
|
-
"This includes:\n"
|
|
43
|
-
"- Chemical entities: compound names (generic, IUPAC, code names, brand names), "
|
|
44
|
-
"SMILES strings (only if explicitly written), CAS numbers, molecular formulas.\n"
|
|
45
|
-
"- Biological targets: protein names, gene names, receptor names, enzyme names, pathways.\n"
|
|
46
|
-
"- Bioactivity data: IC50, MIC, EC50, Ki, Kd, and other potency/selectivity measurements "
|
|
47
|
-
"with their numeric values, units, and the compound name or identifier they apply to "
|
|
48
|
-
"(captured as a 'compound_name' attribute).\n"
|
|
49
|
-
"- Assay information: cell lines, assay formats, experimental organisms.\n"
|
|
50
|
-
"- Diseases and indications: cancer types, disease names, therapeutic areas.\n"
|
|
51
|
-
"- Mechanisms of action: binding modes, inhibition types, selectivity descriptions.\n"
|
|
52
|
-
"Extract only what is explicitly stated; do not infer or generate values."
|
|
53
|
-
)
|
|
54
|
-
|
|
55
|
-
# ── Tuberculosis early drug discovery prompts ──────────────────────────
|
|
56
|
-
|
|
57
|
-
TB_PROMPT = (
|
|
58
|
-
"Extract drug discovery entities from this tuberculosis research text.\n\n"
|
|
59
|
-
"For each bioactivity measurement (IC50, MIC, MIC90, EC50, CC50, etc.), capture the "
|
|
60
|
-
"compound name, code, or identifier it belongs to as a 'compound_name' attribute "
|
|
61
|
-
"(e.g. 'BTZ043 showed a MIC of 1 ng/mL' → compound_name: 'BTZ043').\n\n"
|
|
62
|
-
"DISAMBIGUATION RULES:\n"
|
|
63
|
-
"- Mycobacterial proteins (e.g. ClpC1, DprE1, InhA, AtpE, MmpL3, QcrB) "
|
|
64
|
-
"are biological targets, NOT compounds.\n"
|
|
65
|
-
"- Rv locus tags (Rv3790, Rv1484), UniProt IDs (P9WPS1), and PDB codes "
|
|
66
|
-
"are accession_number, not target or gene_name.\n"
|
|
67
|
-
"- Compound registry and programme identifiers (CHEMBL4521987, ZINC000012345678, "
|
|
68
|
-
"DB00945, SACC-3060, TBDA-01187, GSK3036656) are compound_name; only Rv locus "
|
|
69
|
-
"tags, UniProt IDs, PDB codes and RefSeq IDs are accession_number.\n"
|
|
70
|
-
"- Enzyme descriptions like 'enoyl-ACP reductase' are product, not target.\n"
|
|
71
|
-
"- 'cell wall', 'lipid metabolism' are functional_category, not mechanism_of_action.\n"
|
|
72
|
-
"- 'fragment screening', 'biochemical assay' are screening_method, not assay.\n"
|
|
73
|
-
"- Use target for proteins in a drug-targeting context, gene_name for loci, "
|
|
74
|
-
"protein_name for non-drug-target proteins.\n\n"
|
|
75
|
-
"Extract only what is explicitly stated; do not infer or generate values."
|
|
76
|
-
)
|
|
77
|
-
|
|
78
|
-
TB_CHEMISTRY_PROMPT = (
|
|
79
|
-
"Extract chemical entities from this tuberculosis drug discovery text. "
|
|
80
|
-
"Include compound names, SMILES (only if explicitly written), CAS numbers, "
|
|
81
|
-
"and molecular formulas. "
|
|
82
|
-
"Mycobacterial proteins (ClpC1, DprE1, InhA, AtpE, etc.) are NOT compounds. "
|
|
83
|
-
"Extract only what is explicitly stated; do not infer or generate values."
|
|
84
|
-
)
|
|
85
|
-
|
|
86
|
-
TB_BIOLOGY_PROMPT = (
|
|
87
|
-
"Extract biological entities from this tuberculosis research text. "
|
|
88
|
-
"Use target for proteins in a drug-targeting context, gene_name for loci, "
|
|
89
|
-
"protein_name for non-drug-target proteins. "
|
|
90
|
-
"Rv locus tags and UniProt IDs are accession_number. "
|
|
91
|
-
"Enzyme descriptions (e.g. 'enoyl-ACP reductase') are product. "
|
|
92
|
-
"Protein functional categories (e.g. 'cell wall', 'lipid metabolism') "
|
|
93
|
-
"are functional_category. "
|
|
94
|
-
"Extract only what is explicitly stated; do not infer or generate values."
|
|
95
|
-
)
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{structflo_ner-0.5.0 → structflo_ner-0.7.0}/structflo/ner/fast/gazetteers/accession_number.yml
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{structflo_ner-0.5.0 → structflo_ner-0.7.0}/structflo/ner/fast/gazetteers/functional_category.yml
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{structflo_ner-0.5.0 → structflo_ner-0.7.0}/structflo/ner/fast/gazetteers/screening_method.yml
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|